@nodaro/prompts 1.12.0 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/index.cjs +389 -194
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.d.cts +244 -29
  4. package/dist/index.d.ts +244 -29
  5. package/dist/index.js +377 -196
  6. package/dist/index.js.map +1 -1
  7. package/package.json +2 -2
  8. package/src/__tests__/assemble-image-input-cap.test.ts +37 -13
  9. package/src/__tests__/assemble-image-input.test.ts +100 -19
  10. package/src/__tests__/assemble-video-input-cap.test.ts +101 -15
  11. package/src/__tests__/assemble-video-input.test.ts +167 -33
  12. package/src/__tests__/direction-hint-token-safety.test.ts +21 -0
  13. package/src/__tests__/multi-picker-spec.test.ts +21 -1
  14. package/src/__tests__/person-regional-aesthetic.test.ts +2 -1
  15. package/src/__tests__/prompt-style-section.test.ts +345 -0
  16. package/src/__tests__/provider-prompt-doctrine.test.ts +39 -0
  17. package/src/__tests__/style-section-boundary.test.ts +179 -0
  18. package/src/__tests__/subject-fold.test.ts +32 -13
  19. package/src/assemble-image-input.ts +51 -26
  20. package/src/assemble-video-input.ts +54 -34
  21. package/src/direction-registry.ts +108 -37
  22. package/src/gemini-omni-inputs.ts +11 -3
  23. package/src/held-prop.ts +1 -0
  24. package/src/hint-shedding.ts +23 -4
  25. package/src/index.ts +1 -0
  26. package/src/person.ts +4 -0
  27. package/src/picker-analyzer-registry.ts +37 -0
  28. package/src/prompt-builder.ts +84 -27
  29. package/src/prompt-hint-join.ts +9 -0
  30. package/src/prompt-style-section.ts +256 -0
  31. package/src/prompt-wizard-categories.ts +6 -0
  32. package/src/provider-prompt-doctrine.ts +51 -2
  33. package/src/setting.ts +1 -0
  34. package/src/style.ts +1 -0
  35. package/src/styling.ts +5 -1
  36. package/src/video-reference-resolver.ts +5 -2
@@ -44,7 +44,6 @@ import {
44
44
  type StructuredPromptFields,
45
45
  } from "./prompt-builder-structured-fields.js"
46
46
  import {
47
- renderDirectionHints,
48
47
  IMAGE_HINT_MODE_DEFAULT,
49
48
  type DirectionFields,
50
49
  } from "./direction-registry.js"
@@ -53,7 +52,13 @@ import {
53
52
  SUBJECT_IMAGE_HINT_MODE_DEFAULT,
54
53
  type SubjectFields,
55
54
  } from "./subject-registry.js"
56
- import { joinPromptHints } from "./prompt-hint-join.js"
55
+ import {
56
+ asBodyClauses,
57
+ composeSectionedPrompt,
58
+ partitionStyleClauses,
59
+ sectionedClauseCosts,
60
+ type SlottedPromptClause,
61
+ } from "./prompt-style-section.js"
57
62
  import { keepableDirectionHints } from "./hint-shedding.js"
58
63
  import type { CharacterDef, ConnectedReference, IdentityMeta } from "@nodaro/shared"
59
64
 
@@ -174,17 +179,23 @@ export interface AssembleImageInput {
174
179
  * the TAIL: direction leaves before subject, which is the right order (who is
175
180
  * in the shot outranks how it is lit). `structuredFragment` is user CONTENT (a
176
181
  * Path-1 structured field the caller populated), so it is sticky and always
177
- * lands LAST, exactly as before.
182
+ * ends the BODY, exactly as before — the `[style]` section reads after it.
183
+ *
184
+ * Each clause carries the SLOT it reads in. The list order is the SURVIVAL
185
+ * order; it stopped being the string order when the look clauses lifted into
186
+ * the section (`prompt-style-section.ts`). On this surface every direction row
187
+ * is `look` — the registry has no image-surface motion row — so an image
188
+ * `[style]` section carries the whole direction fold.
178
189
  */
179
190
  interface ImageHintPieces {
180
191
  /** Subject clauses then direction clauses, each in its registry's fold order. */
181
- readonly hintClauses: readonly string[]
192
+ readonly hintClauses: readonly SlottedPromptClause[]
182
193
  /** The structured-field fragment ("" when nothing is populated). */
183
194
  readonly structuredFragment: string
184
195
  }
185
196
 
186
197
  /**
187
- * Render the fold's hint pieces once, so the cap-aware retry can re-join a
198
+ * Render the fold's hint pieces once, so the cap-aware retry can re-compose a
188
199
  * SUBSET of them without re-rendering the catalogs. Each renderer folds its own
189
200
  * channel in its registry's canonical table order (unknown keys and unknown ids
190
201
  * contribute nothing), and `renderStructuredFields` returns "" when nothing is
@@ -192,7 +203,7 @@ interface ImageHintPieces {
192
203
  *
193
204
  * SUBJECT LEADS DIRECTION: the subject is the noun phrase ("a woman in her
194
205
  * 30s, …") the cinematographic clauses then modify. With no `subject` the list
195
- * IS the direction fold, so every existing caller's prompt is byte-identical.
206
+ * IS the direction fold, so every existing caller sheds identically.
196
207
  */
197
208
  function renderImageHintPieces(
198
209
  subject: SubjectFields | undefined,
@@ -201,15 +212,17 @@ function renderImageHintPieces(
201
212
  ): ImageHintPieces {
202
213
  return {
203
214
  hintClauses: [
204
- ...renderSubjectHints(subject, {
205
- surface: "image",
206
- mode: SUBJECT_IMAGE_HINT_MODE_DEFAULT,
207
- }),
208
- ...renderDirectionHints(direction, {
215
+ ...asBodyClauses(
216
+ renderSubjectHints(subject, {
217
+ surface: "image",
218
+ mode: SUBJECT_IMAGE_HINT_MODE_DEFAULT,
219
+ }),
220
+ ),
221
+ ...partitionStyleClauses(direction, {
209
222
  surface: "image",
210
223
  mode: IMAGE_HINT_MODE_DEFAULT,
211
224
  }),
212
- ].filter((p) => p.length > 0),
225
+ ].filter((c) => c.text.length > 0),
213
226
  structuredFragment: structured ? renderStructuredFields(structured) : "",
214
227
  }
215
228
  }
@@ -218,32 +231,34 @@ function renderImageHintPieces(
218
231
  * Compose the subject + cinematic-direction hints and the structured-field
219
232
  * fragment with the user's prompt, keeping the FIRST `keptHintClauses` clauses
220
233
  * (the full count on the first pass; fewer only when the provider cap forced a
221
- * shed). The structured fragment always lands LAST.
234
+ * shed). The structured fragment always ends the body; the look clauses that
235
+ * survived follow it in the `[style]` section.
222
236
  *
223
237
  * EXACT NO-OP CONTRACT: when there are no subject/cinematic/structured hint
224
238
  * pieces (the platform-caller case for a node that carries no stored `subject`/
225
239
  * `direction`/`structured` — every workflow authored before the canvas honored
226
240
  * them), the
227
- * user's prompt is returned **verbatim, untrimmed** by `joinPromptHints`. This
228
- * is load-bearing for parity: the old platform path passed the prompt straight
229
- * to `buildImagePrompt`, which never trims, so trimming here would change the
241
+ * user's prompt is returned **verbatim, untrimmed**. This is load-bearing for
242
+ * parity: the old platform path passed the prompt straight to
243
+ * `buildImagePrompt`, which never trims, so trimming here would change the
230
244
  * assembled prompt (and the recorded `jobs.input_data`) byte-for-byte. Never
231
245
  * mutates inputs.
232
246
  *
233
- * A node that DOES carry `direction`/`structured` takes the join branch and is
234
- * therefore trimmed + `". "`-joined — intended, and asserted at the caller
235
- * level by the payload-builder before/after test.
247
+ * A node that DOES carry `direction`/`structured` takes the fold branch and is
248
+ * therefore trimmed — a `[style]` section counts as folded even when the body
249
+ * gained nothing. Intended, and asserted at the caller level by the
250
+ * payload-builder before/after test.
236
251
  */
237
252
  function composePromptText(
238
253
  userPrompt: string,
239
254
  pieces: ImageHintPieces,
240
255
  keptHintClauses: number,
241
256
  ): string {
242
- const hints = [
243
- ...pieces.hintClauses.slice(0, keptHintClauses),
257
+ return composeSectionedPrompt(
258
+ userPrompt,
259
+ pieces.hintClauses.slice(0, keptHintClauses),
244
260
  pieces.structuredFragment,
245
- ].filter((p) => p.length > 0)
246
- return joinPromptHints(userPrompt, hints)
261
+ )
247
262
  }
248
263
 
249
264
  /**
@@ -318,9 +333,19 @@ export function assembleImageInput(
318
333
  // point the body overflows on its own and the builder's clamp stands.
319
334
  let kept = pieces.hintClauses.length
320
335
  let fitted = assembleWith(kept)
321
- while (fitted.overflowChars > 0 && kept > 0) {
322
- kept = keepableDirectionHints(pieces.hintClauses, kept, fitted.overflowChars)
323
- fitted = assembleWith(kept)
336
+ if (fitted.overflowChars > 0) {
337
+ // Priced only on the overflow path: the deltas cost a composition per
338
+ // clause and the fits-first-time case is the common one.
339
+ const costs = sectionedClauseCosts(
340
+ input.userPrompt,
341
+ pieces.hintClauses,
342
+ pieces.structuredFragment,
343
+ )
344
+ const texts = pieces.hintClauses.map((c) => c.text)
345
+ while (fitted.overflowChars > 0 && kept > 0) {
346
+ kept = keepableDirectionHints(texts, kept, fitted.overflowChars, costs)
347
+ fitted = assembleWith(kept)
348
+ }
324
349
  }
325
350
  // `overflowChars` is assembly bookkeeping, not part of the callers' contract.
326
351
  const { overflowChars, ...result } = fitted
@@ -11,8 +11,9 @@
11
11
  *
12
12
  * WHERE IT RUNS (load-bearing): on the prompt BODY, BEFORE
13
13
  * `resolveVideoReferenceCore`. That resolver FRAMES the body — legacy prepends
14
- * its `Use these characters:` block, hybrid prepends the lock lines and APPENDS
15
- * the canonical role phrases and extras. Folding afterwards would push the
14
+ * its `Use these characters:` block, hybrid prepends the lock lines and extends
15
+ * the body's END with the canonical role phrases and extras (spliced in ahead of
16
+ * the `[style]` section, which stays last). Folding afterwards would push the
16
17
  * scene/look description PAST the identity directives, a worse version of the
17
18
  * bug this channel exists to fix. The image side is structurally identical
18
19
  * (`assembleImageInput` = `composePromptText` → `buildImagePrompt`).
@@ -24,6 +25,13 @@
24
25
  * gets dropped. See {@link VideoPromptCapOptions}. Both catalog channels —
25
26
  * SUBJECT and direction — fold into the one sheddable list that budget walks.
26
27
  *
28
+ * THE SHAPE IT EMITS: the body (prose, subject fold, MOTION clauses, structured
29
+ * fragment) and then a trailing `[style]` section carrying every LOOK clause —
30
+ * `prompt-style-section.ts` owns those bytes. Camera motion is shot prose, not
31
+ * style, so the whole motion family stays in the body; the boundary is the
32
+ * registry's `family` column, deliberately the same column the verbosity policy
33
+ * below splits on.
34
+ *
27
35
  * THE VERBOSITY POLICY LIVES HERE, NOT IN THE CLIENT: motion dimensions render
28
36
  * their compact professional term, look dimensions their full clause
29
37
  * (`VIDEO_HINT_MODE_DEFAULT`, resolved per row's `family` by the registry). The
@@ -49,10 +57,10 @@
49
57
  * `subject-registry.ts` for the subject channel) — ONE renderer per channel
50
58
  * serves both surfaces, so the image and video folds cannot drift.
51
59
  * Clients render their "will inject into prompt" preview by importing
52
- * `renderDirectionHints` + `joinPromptHints` directly.
60
+ * `renderSubjectHints` + `partitionStyleClauses` + `composeSectionedPrompt`
61
+ * directly (or `renderStyleSection` for the section alone).
53
62
  */
54
63
  import {
55
- renderDirectionHints,
56
64
  VIDEO_HINT_MODE_DEFAULT,
57
65
  type DirectionFields,
58
66
  type DirectionHintMode,
@@ -63,7 +71,12 @@ import {
63
71
  type SubjectFields,
64
72
  type SubjectHintMode,
65
73
  } from "./subject-registry.js"
66
- import { joinPromptHints } from "./prompt-hint-join.js"
74
+ import {
75
+ asBodyClauses,
76
+ composeSectionedPrompt,
77
+ partitionStyleClauses,
78
+ sectionedClauseCosts,
79
+ } from "./prompt-style-section.js"
67
80
  import { keepableDirectionHints } from "./hint-shedding.js"
68
81
  import {
69
82
  renderStructuredFields,
@@ -90,9 +103,10 @@ import {
90
103
  * module header — folding afterwards strands the scene description past the
91
104
  * identity directives). The resolver then ADDS binding text: legacy's
92
105
  * "Use these characters:" block, hybrid's lock lines and the canonical role
93
- * phrases it APPENDS. That added text is exactly what an order-blind tail cut
94
- * destroys first, so it must be inside the budget — but it must never be
95
- * shed. Measuring THROUGH the caller's framing gives both properties at once:
106
+ * phrases that end its body. None of it is sheddable and all of it is inside
107
+ * what the clamp measures, so a budget blind to it under-sheds and hands the
108
+ * remainder to the order-blind cut. Measuring THROUGH the caller's framing
109
+ * gives both properties at once:
96
110
  * the shed decision sees the final length, while the only thing it can drop
97
111
  * is a hint clause it rendered itself.
98
112
  *
@@ -121,10 +135,11 @@ export interface VideoPromptCapOptions {
121
135
  * Fold a video run's subject and cinematic-direction ids (and optional
122
136
  * structured fields) into its prompt body.
123
137
  *
124
- * The SUBJECT hints land first (who is in the shot — the noun phrase the
125
- * cinematography modifies), then the direction hints in the registry's
126
- * canonical table order (camera motion leads), and the structured fragment
127
- * lands LAST — the same ordering `composePromptText` uses for stills.
138
+ * IN THE BODY: the SUBJECT hints first (who is in the shot — the noun phrase the
139
+ * cinematography modifies), then the MOTION direction hints in the registry's
140
+ * canonical table order (camera motion leads), then the structured fragment —
141
+ * the same ordering `composePromptText` uses for stills. The LOOK hints leave
142
+ * the body for the `[style]` section that follows it.
128
143
  *
129
144
  * `subject` rides `opts` rather than a fourth positional parameter on purpose:
130
145
  * every existing caller passes `(prompt, direction)` or
@@ -194,33 +209,33 @@ export function composeVideoPromptText(
194
209
  // side uses (`renderImageHintPieces`): the subject is the noun phrase the
195
210
  // cinematography modifies, so losing "who is in the shot" to keep a
196
211
  // decorative grade would be the wrong trade. With no `subject` the list IS
197
- // the direction fold, so every pre-subject caller is byte-identical.
212
+ // the direction fold, so every pre-subject caller sheds identically.
213
+ //
214
+ // Each clause carries the SLOT it reads in, because survival order and string
215
+ // order are two different things once the look clauses lift into `[style]`.
198
216
  const hintClauses = [
199
- ...renderSubjectHints(opts?.subject, {
200
- surface: "video",
201
- mode: opts?.subjectHintMode ?? SUBJECT_VIDEO_HINT_MODE_DEFAULT,
202
- }),
203
- ...renderDirectionHints(direction, {
217
+ ...asBodyClauses(
218
+ renderSubjectHints(opts?.subject, {
219
+ surface: "video",
220
+ mode: opts?.subjectHintMode ?? SUBJECT_VIDEO_HINT_MODE_DEFAULT,
221
+ }),
222
+ ),
223
+ ...partitionStyleClauses(direction, {
204
224
  surface: "video",
205
225
  mode: opts?.hintMode ?? VIDEO_HINT_MODE_DEFAULT,
206
226
  }),
207
- ].filter((p) => p.length > 0)
208
- // User CONTENT, not a garnish: never sheddable, always last.
227
+ ].filter((c) => c.text.length > 0)
228
+ // User CONTENT, not a garnish: never sheddable, always last IN THE BODY (the
229
+ // `[style]` section reads after it).
209
230
  const structuredFragment = structured ? renderStructuredFields(structured) : ""
210
231
 
211
- const composeWith = (kept: number): string | undefined => {
212
- const hints = [...hintClauses.slice(0, kept), structuredFragment].filter(
213
- (p) => p.length > 0,
214
- )
215
- // Nothing to fold → the caller's value straight back, `undefined` included.
216
- // Do NOT collapse this into `joinPromptHints(userPrompt ?? "", hints)`: that
217
- // would turn an absent prompt into `""` and break the no-op contract above.
218
- // A FULL shed lands here too, which is what keeps the no-op contract intact
219
- // at `kept === 0` — the route's `composed !== prompt` guard then correctly
220
- // leaves `input_data.userPrompt` unpinned.
221
- if (hints.length === 0) return userPrompt
222
- return joinPromptHints(userPrompt ?? "", hints)
223
- }
232
+ // Nothing folded — no body hint AND no section — returns the caller's value
233
+ // straight back, `undefined` included. A FULL shed lands there too, which is
234
+ // what keeps the no-op contract intact at `kept === 0`: the route's
235
+ // `composed !== prompt` guard then correctly leaves `input_data.userPrompt`
236
+ // unpinned.
237
+ const composeWith = (kept: number): string | undefined =>
238
+ composeSectionedPrompt(userPrompt, hintClauses.slice(0, kept), structuredFragment)
224
239
 
225
240
  const cap = opts?.cap
226
241
  if (cap === undefined) return composeWith(hintClauses.length)
@@ -235,8 +250,13 @@ export function composeVideoPromptText(
235
250
  let kept = hintClauses.length
236
251
  let body = composeWith(kept)
237
252
  let framedLength = frame(body)?.length ?? 0
253
+ if (framedLength <= cap) return body
254
+ // Priced only on the overflow path: the deltas cost a composition per clause
255
+ // and the fits-first-time case is the common one.
256
+ const costs = sectionedClauseCosts(userPrompt, hintClauses, structuredFragment)
257
+ const texts = hintClauses.map((c) => c.text)
238
258
  while (framedLength > cap && kept > 0) {
239
- kept = keepableDirectionHints(hintClauses, kept, framedLength - cap)
259
+ kept = keepableDirectionHints(texts, kept, framedLength - cap, costs)
240
260
  body = composeWith(kept)
241
261
  framedLength = frame(body)?.length ?? 0
242
262
  }
@@ -72,8 +72,23 @@ import { getLoopSubjectPromptHint, getLoopSubjectTerm } from "./loop-subject.js"
72
72
 
73
73
  /** Which generation stages fold a dimension. */
74
74
  export type DirectionSurface = "image" | "video" | "both"
75
- /** Verbosity family — the video policy folds `motion` compact, `look` full. */
75
+ /**
76
+ * Verbosity family — the video policy folds `motion` compact, `look` full.
77
+ *
78
+ * It is ALSO the body/section split (`prompt-style-section.ts`): `motion` stays
79
+ * in the prompt body as shot prose, `look` moves to the `[style]` section. The
80
+ * two meanings are deliberately the same column: camera motion is part of the
81
+ * shot, not part of the look, on both axes, and a second column would let the
82
+ * verbosity policy and the section boundary drift apart one row at a time.
83
+ */
76
84
  export type DirectionFamily = "look" | "motion"
85
+ /**
86
+ * Which `[style]` line a LOOK row renders on. `"film"` = the four dimensions
87
+ * that describe the CAPTURE (stock, grade, style, era) plus the legacy camera
88
+ * format key; every other look row falls to the scene line. Absent on `motion`
89
+ * rows, which never reach the section at all.
90
+ */
91
+ export type DirectionStyleGroup = "film"
77
92
 
78
93
  export interface DirectionFieldSpec {
79
94
  /**
@@ -89,6 +104,12 @@ export interface DirectionFieldSpec {
89
104
  readonly surface: DirectionSurface
90
105
  /** Verbosity family. The video policy folds `motion` compact, `look` full. */
91
106
  readonly family: DirectionFamily
107
+ /**
108
+ * `[style]`-section line for a `look` row. Omitted = the scene line. Meaningless
109
+ * on a `motion` row (those stay in the body), which is why it is optional
110
+ * rather than a required column with a null member.
111
+ */
112
+ readonly styleGroup?: DirectionStyleGroup
92
113
  /** Ids honored per dimension. Extras are SLICED at render, never a 400. */
93
114
  readonly maxPicks: number
94
115
  /**
@@ -160,16 +181,20 @@ const temporal = perId(getTemporalPromptHint, getTemporalTerm)
160
181
  * would wrongly suppress a legal second selection. Overlap is handled instead
161
182
  * by the exact-string dedupe in `renderDirectionHints`.
162
183
  *
163
- * SECOND MEANING OF POSITION: BOTH cap-aware assemblers — `assembleImageInput`
164
- * (stills) and `composeVideoPromptText` (video) — shed hint clauses from the
165
- * TAIL of this order when a provider's prompt cap overflows, through the one
166
- * shared arithmetic in `hint-shedding.ts`. So a row's position is also its
167
- * survival order under the cap on EVERY surface: reordering rows for one
168
- * surface silently changes what the other drops first, and the row a
169
- * video-surface reorder would most likely touch (`cameraMotion`) leads the
170
- * fold. That is a consequence of reusing the fold order, not a ranking — this
171
- * table stays a compatibility order; anything that needs a real importance
172
- * ranking should add an explicit priority column rather than reorder these rows.
184
+ * SECOND MEANING OF POSITION — SURVIVAL, NOT STRING POSITION: BOTH cap-aware
185
+ * assemblers — `assembleImageInput` (stills) and `composeVideoPromptText`
186
+ * (video) — shed hint clauses from the TAIL of this order when a provider's
187
+ * prompt cap overflows, through the one shared arithmetic in
188
+ * `hint-shedding.ts`. So a row's position is its survival order under the cap on
189
+ * EVERY surface: reordering rows for one surface silently changes what the other
190
+ * drops first, and the row a video-surface reorder would most likely touch
191
+ * (`cameraMotion`) leads the fold. What position is NOT any more is the clause's
192
+ * place in the assembled STRING: every `look` row is lifted out of the body into
193
+ * the trailing `[style]` section (`prompt-style-section.ts`), so a look clause
194
+ * reads after every motion clause however early it folds. That is a consequence
195
+ * of reusing the fold order, not a ranking — this table stays a compatibility
196
+ * order; anything that needs a real importance ranking should add an explicit
197
+ * priority column rather than reorder these rows.
173
198
  *
174
199
  * WHERE THIS TABLE SITS IN THE COMBINED ORDER: both assemblers fold the SUBJECT
175
200
  * channel (`subject-registry.ts`) BEFORE this one and shed the combined list
@@ -189,7 +214,7 @@ export const DIRECTION_FIELDS = [
189
214
  { key: "compositionEffect", surface: "both", family: "look", maxPicks: 1, render: perId(getCompositionEffectPromptHint, getCompositionEffectTerm) },
190
215
 
191
216
  // Camera.
192
- { key: "cameraFormat", surface: "both", family: "look", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
217
+ { key: "cameraFormat", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
193
218
  { key: "lens", surface: "both", family: "look", maxPicks: 1, render: perId(getLensPromptHint, getLensTerm) },
194
219
 
195
220
  // Exposure (stills only — a video's exposure rides its own temporal levers).
@@ -203,12 +228,12 @@ export const DIRECTION_FIELDS = [
203
228
  { key: "lightingDirection", surface: "both", family: "look", maxPicks: 1, render: lighting },
204
229
  { key: "lightingRatio", surface: "both", family: "look", maxPicks: 1, render: lighting },
205
230
  { key: "colorTemperature", surface: "both", family: "look", maxPicks: 1, render: lighting },
206
- { key: "colorLook", surface: "both", family: "look", maxPicks: 1, render: perId(getColorLookPromptHint, getColorLookTerm) },
231
+ { key: "colorLook", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getColorLookPromptHint, getColorLookTerm) },
207
232
  { key: "atmosphere", surface: "both", family: "look", maxPicks: 2, render: viaListBuilder(buildAtmosphereHints) },
208
233
  { key: "postProcess", surface: "image", family: "look", maxPicks: 2, render: viaListBuilder(buildPostProcessHints) },
209
234
 
210
235
  // Style.
211
- { key: "style", surface: "both", family: "look", maxPicks: 1, render: perId(getStylePromptHint, getStyleTerm) },
236
+ { key: "style", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getStylePromptHint, getStyleTerm) },
212
237
  { key: "mood", surface: "both", family: "look", maxPicks: 2, render: viaMood },
213
238
  { key: "aesthetic", surface: "both", family: "look", maxPicks: 2, render: viaStringBuilder(buildAestheticHints) },
214
239
  { key: "photoGenre", surface: "image", family: "look", maxPicks: 1, render: perId(getPhotoGenrePromptHint, getPhotoGenreTerm) },
@@ -217,7 +242,7 @@ export const DIRECTION_FIELDS = [
217
242
 
218
243
  // Scene.
219
244
  { key: "setting", surface: "both", family: "look", maxPicks: 1, render: perId(getSettingPromptHint, getSettingTerm) },
220
- { key: "era", surface: "both", family: "look", maxPicks: 1, render: perId(getEraPromptHint, getEraTerm) },
245
+ { key: "era", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getEraPromptHint, getEraTerm) },
221
246
  { key: "backdrop", surface: "both", family: "look", maxPicks: 1, render: perId(getBackdropPromptHint, getBackdropTerm) },
222
247
 
223
248
  // Motion & time.
@@ -235,9 +260,17 @@ export const DIRECTION_FIELDS = [
235
260
  { key: "framingAngleId", surface: "both", family: "look", maxPicks: 1, render: framing },
236
261
  { key: "lightingId", surface: "both", family: "look", maxPicks: 1, render: lighting },
237
262
  { key: "lensId", surface: "both", family: "look", maxPicks: 1, render: perId(getLensPromptHint, getLensTerm) },
238
- { key: "cameraFormatId", surface: "both", family: "look", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
263
+ { key: "cameraFormatId", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
239
264
  ] as const satisfies ReadonlyArray<DirectionFieldSpec>
240
265
 
266
+ /**
267
+ * The table read at its DECLARED type. `styleGroup` is optional, so on the
268
+ * `as const` tuple only the rows that carry it have the property at all — a
269
+ * member-wise read would not compile. Every walk over the table goes through
270
+ * this binding.
271
+ */
272
+ const DIRECTION_SPECS: ReadonlyArray<DirectionFieldSpec> = DIRECTION_FIELDS
273
+
241
274
  export type DirectionFieldRow = (typeof DIRECTION_FIELDS)[number]
242
275
  export type DirectionKey = DirectionFieldRow["key"]
243
276
  export type ImageDirectionKey = Extract<DirectionFieldRow, { surface: "image" | "both" }>["key"]
@@ -260,6 +293,16 @@ export type DirectionFields = { readonly [K in DirectionKey]?: string | readonly
260
293
  */
261
294
  export const DIRECTION_KEYS: ReadonlyArray<DirectionKey> = DIRECTION_FIELDS.map((f) => f.key)
262
295
 
296
+ /**
297
+ * The rows that render on the `[style]` section's FILM line, in table order —
298
+ * derived from the table's `styleGroup` column so the grouping has exactly one
299
+ * definition. Exported for clients that render the section themselves; the
300
+ * platform's own renderer reads the column, not this list.
301
+ */
302
+ export const FILM_STYLE_KEYS: ReadonlyArray<DirectionKey> = DIRECTION_SPECS.filter(
303
+ (f) => f.styleGroup === "film",
304
+ ).map((f) => f.key as DirectionKey)
305
+
263
306
  /** Verbosity for a whole fold, or split per family. */
264
307
  export type DirectionHintMode =
265
308
  | PickerHintMode
@@ -321,7 +364,51 @@ function normalizeDirectionIds(value: unknown, maxPicks: number): string[] {
321
364
  export function directionFieldsForSurface(
322
365
  surface: "image" | "video",
323
366
  ): ReadonlyArray<DirectionFieldSpec> {
324
- return DIRECTION_FIELDS.filter((f) => f.surface === "both" || f.surface === surface)
367
+ return DIRECTION_SPECS.filter((f) => f.surface === "both" || f.surface === surface)
368
+ }
369
+
370
+ /** One rendered clause, still carrying the table attributes it came from. */
371
+ export interface DirectionHintClause {
372
+ readonly key: DirectionKey
373
+ readonly family: DirectionFamily
374
+ readonly styleGroup?: DirectionStyleGroup
375
+ readonly text: string
376
+ }
377
+
378
+ /**
379
+ * `renderDirectionHints` with the row each clause came from still attached —
380
+ * what the `[style]` section needs to decide which line a clause belongs on
381
+ * without a second table. Same order, same surface filter, same dedupe; the
382
+ * plain renderer is this one's `.text` projection, so the two cannot drift.
383
+ */
384
+ export function renderDirectionHintClauses(
385
+ direction: DirectionFields | undefined,
386
+ opts: { surface: "image" | "video"; mode?: DirectionHintMode },
387
+ ): DirectionHintClause[] {
388
+ if (!direction) return []
389
+ const mode = opts.mode ?? "full"
390
+ const out: DirectionHintClause[] = []
391
+ const seen = new Set<string>()
392
+ for (const spec of DIRECTION_SPECS) {
393
+ if (spec.surface !== "both" && spec.surface !== opts.surface) continue
394
+ const ids = normalizeDirectionIds(
395
+ (direction as Record<string, unknown>)[spec.key],
396
+ spec.maxPicks,
397
+ )
398
+ if (ids.length === 0) continue
399
+ for (const hint of spec.render(ids, modeForFamily(mode, spec.family))) {
400
+ if (hint.length > 0 && !seen.has(hint)) {
401
+ seen.add(hint)
402
+ out.push({
403
+ key: spec.key as DirectionKey,
404
+ family: spec.family,
405
+ ...(spec.styleGroup !== undefined ? { styleGroup: spec.styleGroup } : {}),
406
+ text: hint,
407
+ })
408
+ }
409
+ }
410
+ }
411
+ return out
325
412
  }
326
413
 
327
414
  /**
@@ -343,29 +430,13 @@ export function directionFieldsForSurface(
343
430
  * exactly as two wired picker nodes of one family behave today.
344
431
  *
345
432
  * Exported so a client's "will inject into prompt" preview renders the exact
346
- * server output instead of re-implementing the fold.
433
+ * clauses the server does instead of re-implementing the fold. A preview of the
434
+ * assembled STRING needs `prompt-style-section.ts` on top: the look clauses in
435
+ * this list do not read in this position any more.
347
436
  */
348
437
  export function renderDirectionHints(
349
438
  direction: DirectionFields | undefined,
350
439
  opts: { surface: "image" | "video"; mode?: DirectionHintMode },
351
440
  ): string[] {
352
- if (!direction) return []
353
- const mode = opts.mode ?? "full"
354
- const out: string[] = []
355
- const seen = new Set<string>()
356
- for (const spec of DIRECTION_FIELDS) {
357
- if (spec.surface !== "both" && spec.surface !== opts.surface) continue
358
- const ids = normalizeDirectionIds(
359
- (direction as Record<string, unknown>)[spec.key],
360
- spec.maxPicks,
361
- )
362
- if (ids.length === 0) continue
363
- for (const hint of spec.render(ids, modeForFamily(mode, spec.family))) {
364
- if (hint.length > 0 && !seen.has(hint)) {
365
- seen.add(hint)
366
- out.push(hint)
367
- }
368
- }
369
- }
370
- return out
441
+ return renderDirectionHintClauses(direction, opts).map((c) => c.text)
371
442
  }
@@ -40,6 +40,11 @@ export interface GeminiOmniI2vInputsArgs {
40
40
  refImageUrls?: Array<string | undefined>
41
41
  /** A connected source video occupies 2 of the 7 input slots (KIE quota). */
42
42
  videoConnected?: boolean
43
+ /** The Omni SKU this run targets — defaults to `gemini-omni-video` for
44
+ * back-compat. Both SKUs cap at 7 today, so passing it is behaviour-neutral;
45
+ * it stops the flash path from silently reading the pro model's quota if the
46
+ * two ever diverge. */
47
+ provider?: string
43
48
  }
44
49
 
45
50
  export interface GeminiOmniI2vInputsResult {
@@ -51,13 +56,16 @@ export interface GeminiOmniI2vInputsResult {
51
56
  droppedRefImages: number
52
57
  }
53
58
 
54
- /** The catalog-declared cap (7) — read from the shared limits map so the
59
+ /** The catalog-declared cap (7) — read PER SKU from the shared limits map so the
55
60
  * wire-contract number has one home; the literal is only the safety net. */
56
- const GEMINI_OMNI_INPUT_SLOTS = VIDEO_REF_LIMITS_BY_PROVIDER["gemini-omni-video"]?.images ?? 7
61
+ const DEFAULT_GEMINI_OMNI_PROVIDER = "gemini-omni-video"
62
+ function geminiOmniInputSlots(provider: string | undefined): number {
63
+ return VIDEO_REF_LIMITS_BY_PROVIDER[provider ?? DEFAULT_GEMINI_OMNI_PROVIDER]?.images ?? 7
64
+ }
57
65
 
58
66
  export function resolveGeminiOmniI2vInputs(args: GeminiOmniI2vInputsArgs): GeminiOmniI2vInputsResult {
59
67
  const refs = (args.refImageUrls ?? []).filter((u): u is string => typeof u === "string" && u.length > 0)
60
- const slots = GEMINI_OMNI_INPUT_SLOTS - (args.videoConnected ? 2 : 0)
68
+ const slots = geminiOmniInputSlots(args.provider) - (args.videoConnected ? 2 : 0)
61
69
  const refSlots = Math.max(0, slots - 1)
62
70
  const kept = refs.slice(0, refSlots)
63
71
  const droppedRefImages = refs.length - kept.length
package/src/held-prop.ts CHANGED
@@ -135,6 +135,7 @@ export const HELD_PROPS: ReadonlyArray<HeldProp> = [
135
135
  { id: "compass", label: "Compass", category: "occupational", description: "Vintage handheld nautical compass", promptHint: "holding a vintage brass nautical compass open in one cupped palm at chest height, the needle clearly visible as the eyes drift down to read the bearing", term: "holding an open brass nautical compass" },
136
136
  { id: "bow-and-arrow", label: "Bow and Arrow", category: "occupational", description: "Drawn archery bow with arrow nocked", promptHint: "holding an archery bow drawn at full tension with one hand on the grip and the other pulling the string back to the cheek, an arrow nocked and aimed forward", term: "drawing an archery bow with a nocked arrow" },
137
137
  { id: "shield", label: "Shield", category: "occupational", description: "Handheld medieval shield", promptHint: "holding a medieval shield raised across the body with one arm strapped through the back, the front face angled forward in a defensive stance", term: "holding a raised medieval shield" },
138
+ { id: "work-gloves", label: "Work Gloves", category: "occupational", description: "Worn leather work gloves held in hand", promptHint: "holding a worn pair of tan leather work gloves in both hands at waist height, the thick weathered leather clearly visible", term: "holding a pair of leather work gloves" },
138
139
  ] as const
139
140
 
140
141
  const heldPropById = new Map<string, HeldProp>(HELD_PROPS.map((p) => [p.id, p]))
@@ -19,6 +19,14 @@
19
19
  * prose instead — precisely the bug this machinery exists to prevent. Never
20
20
  * sheddable: the user's prose, the bound references and the framing text the
21
21
  * reference resolver adds, and the structured fragment (user CONTENT).
22
+ *
23
+ * THE LIST IS A SURVIVAL ORDER, NOT A STRING ORDER. It was both until the
24
+ * `[style]` section landed; now a look clause is lifted out of the body and
25
+ * reads after every motion clause however early it folds
26
+ * (`prompt-style-section.ts`). Position here still answers exactly one question
27
+ * — who leaves first — and `clauseCosts` is how the caller tells this function
28
+ * what a clause actually costs in a shape it can no longer infer from the
29
+ * clause text alone.
22
30
  */
23
31
  import { PROMPT_HINT_SEPARATOR } from "./prompt-hint-join.js"
24
32
 
@@ -49,20 +57,31 @@ import { PROMPT_HINT_SEPARATOR } from "./prompt-hint-join.js"
49
57
  * decorative `isoValue` clause; if that ever matters, the fix is an explicit
50
58
  * priority column on `DIRECTION_FIELDS`, not a second hand-kept list here.
51
59
  *
52
- * Deliberately approximate (assembly is not perfectly additive); the caller
53
- * re-assembles and re-checks, and this function strictly decreases `kept`
54
- * whenever `deficit > 0`, so that loop terminates.
60
+ * `clauseCosts[i]` is what clause `i` really adds to the assembled prompt.
61
+ * Without it each clause is charged its text plus one separator, which is what
62
+ * a clause folded inline costs — but a clause that lands in the `[style]`
63
+ * section carries section bytes too (the first one carries the whole header),
64
+ * and under-charging it makes this walk cover the deficit with MORE clauses
65
+ * than it needs. Both in-package callers pass exact composed-length deltas
66
+ * (`sectionedClauseCosts`); the default keeps the pre-section arithmetic for
67
+ * anyone else.
68
+ *
69
+ * Still deliberately approximate (assembly is not perfectly additive — a
70
+ * downstream frame can grow or shrink around the body); the caller re-assembles
71
+ * and re-checks, and this function strictly decreases `kept` whenever
72
+ * `deficit > 0`, so that loop terminates however the costs are priced.
55
73
  */
56
74
  export function keepableDirectionHints(
57
75
  hintClauses: readonly string[],
58
76
  kept: number,
59
77
  deficit: number,
78
+ clauseCosts?: readonly number[],
60
79
  ): number {
61
80
  let remaining = deficit
62
81
  let next = kept
63
82
  while (next > 0 && remaining > 0) {
64
83
  next -= 1
65
- remaining -= hintClauses[next]!.length + PROMPT_HINT_SEPARATOR.length
84
+ remaining -= clauseCosts?.[next] ?? hintClauses[next]!.length + PROMPT_HINT_SEPARATOR.length
66
85
  }
67
86
  return next
68
87
  }
package/src/index.ts CHANGED
@@ -21,6 +21,7 @@ export * from "./prompt-builder-structured-fields.js"
21
21
  export * from "./direction-registry.js"
22
22
  export * from "./subject-registry.js"
23
23
  export * from "./prompt-hint-join.js"
24
+ export * from "./prompt-style-section.js"
24
25
  export * from "./hint-shedding.js"
25
26
  export * from "./video-reference-resolver.js"
26
27
  export * from "./sound-aggregator.js"
package/src/person.ts CHANGED
@@ -809,6 +809,10 @@ export const PEOPLE: ReadonlyArray<Person> = [
809
809
  { id: "south-india-traditional", label: "South India Traditional", group: "Asia", dimension: "regional-aesthetic", description: "Tamil / Kerala temple-town classical aesthetic", promptHint: "a South Indian traditional aesthetic — Tamil / Kerala temple-town vibe, classical refinement", term: "south indian traditional aesthetic" },
810
810
  { id: "bangkok-street", label: "Bangkok Street", group: "Asia", dimension: "regional-aesthetic", description: "Bangkok Thai night-market neon-urban aesthetic", promptHint: "a Bangkok street aesthetic — Thai night-market energy, neon-and-warmth urban vibe", term: "bangkok street aesthetic" },
811
811
 
812
+ // ----- Central Asia -----
813
+ { id: "samarkand-silk-road", label: "Samarkand Silk Road", group: "Central Asia", dimension: "regional-aesthetic", description: "Uzbek Silk Road bazaar aesthetic (Samarkand / Bukhara)", promptHint: "a Central Asian Silk Road aesthetic — Samarkand / Bukhara bazaar vibe, ikat-and-suzani textiles, sun-baked adobe and blue-tiled madrasa mood", term: "samarkand silk road aesthetic" },
814
+ { id: "tashkent-modern", label: "Tashkent Modern", group: "Central Asia", dimension: "regional-aesthetic", description: "Contemporary Uzbek metropolitan Central Asian aesthetic", promptHint: "a modern Tashkent aesthetic — contemporary Uzbek metropolitan vibe, Soviet-modern-meets-Silk-Road blend, warm steppe-city confidence", term: "tashkent modern aesthetic" },
815
+
812
816
  // ----- Latin America -----
813
817
  { id: "carioca-rio", label: "Carioca (Rio)", group: "Latin America", dimension: "regional-aesthetic", description: "Rio de Janeiro beach-and-favela-music Brazilian aesthetic", promptHint: "a Carioca aesthetic — Rio de Janeiro beach-and-favela-music vibe, sun-warmed Brazilian energy", term: "carioca aesthetic" },
814
818
  { id: "paulista", label: "Paulista (São Paulo)", group: "Latin America", dimension: "regional-aesthetic", description: "São Paulo metropolitan Brazilian creative-class aesthetic", promptHint: "a Paulista aesthetic — São Paulo metropolitan vibe, urban-Brazilian creative-class polish", term: "paulista aesthetic" },