@nodaro/prompts 1.10.0 → 1.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/index.cjs +693 -53
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.d.cts +1084 -27
  4. package/dist/index.d.ts +1084 -27
  5. package/dist/index.js +664 -55
  6. package/dist/index.js.map +1 -1
  7. package/package.json +2 -2
  8. package/src/__tests__/__snapshots__/entity-convergence-image.test.ts.snap +19 -0
  9. package/src/__tests__/animal-getters-parity.test.ts +82 -0
  10. package/src/__tests__/assemble-image-input-cap.test.ts +212 -0
  11. package/src/__tests__/assemble-image-input.test.ts +93 -3
  12. package/src/__tests__/assemble-video-input-cap.test.ts +356 -0
  13. package/src/__tests__/assemble-video-input.test.ts +301 -0
  14. package/src/__tests__/direction-hint-token-safety.test.ts +113 -0
  15. package/src/__tests__/direction-registry.test.ts +393 -0
  16. package/src/__tests__/entity-convergence-image.test.ts +374 -0
  17. package/src/__tests__/image-convergence-image.test.ts +370 -0
  18. package/src/__tests__/location-convergence-image.test.ts +29 -1
  19. package/src/__tests__/location-default-role-image.test.ts +166 -0
  20. package/src/__tests__/mention-splice-spacing.test.ts +257 -0
  21. package/src/__tests__/read-node-direction.test.ts +154 -0
  22. package/src/__tests__/read-node-subject.test.ts +140 -0
  23. package/src/__tests__/subject-fold.test.ts +232 -0
  24. package/src/__tests__/subject-registry.test.ts +312 -0
  25. package/src/assemble-image-input.ts +160 -58
  26. package/src/assemble-video-input.ts +244 -0
  27. package/src/direction-registry.ts +371 -0
  28. package/src/hint-shedding.ts +68 -0
  29. package/src/index.ts +10 -2
  30. package/src/parameter-prompt-hint.ts +8 -7
  31. package/src/picker-catalogs.ts +14 -7
  32. package/src/prompt-builder.ts +728 -58
  33. package/src/prompt-hint-join.ts +30 -0
  34. package/src/read-node-direction.ts +233 -0
  35. package/src/subject-registry.ts +464 -0
@@ -16,13 +16,17 @@
16
16
  * This wrapper collapses them into one.
17
17
  *
18
18
  * THE NO-OP CONTRACT (load-bearing — the platform-caller parity relies on it):
19
- * the two platform callers (`execute-node` / `payload-builder`) compose their
20
- * prompt from the canvas graph themselves and pass NO cinematic `direction`
21
- * ids and NO `structured` fields. In that case `composePromptText` MUST return
22
- * the caller's `userPrompt` byte-for-byte unchanged, so the wrapper degenerates
23
- * to exactly the `buildImagePrompt(...)` call those sites make today. Studio
24
- * (and the MCP route) supply `direction` / `structured` and get the id-hint
25
- * composition on top.
19
+ * a node that carries NO stored `subject` / `direction` / `structured` (every
20
+ * workflow authored before the canvas honored them) still reaches here with all
21
+ * three absent, and `composePromptText` MUST return the caller's `userPrompt`
22
+ * byte-for-byte unchanged, so the wrapper degenerates to exactly the
23
+ * `buildImagePrompt(...)` call those sites made before. The platform callers
24
+ * (`execute-node` / `payload-builder`) compose their prompt from the canvas GRAPH themselves and
25
+ * ALSO forward a node's STORED `subject` / `direction` / `structured` when it
26
+ * carries them (`readSubjectFields` / `readDirectionFields` /
27
+ * `readStructuredFields`); those nodes get the id-hint composition on top,
28
+ * ADDITIVE to the graph-wired cinematography hints the caller already folded
29
+ * into `userPrompt`. Studio and the MCP route supply the levers directly.
26
30
  *
27
31
  * THE EMPTY-CHECK FLAG (also load-bearing for parity): `execute-node` rejects a
28
32
  * truly-empty assembled prompt (its "type one, mention a character, or connect
@@ -32,34 +36,45 @@
32
36
  * guard (frontend, Studio, route) pass `throwOnEmpty: true`.
33
37
  */
34
38
  import {
35
- buildImagePrompt,
39
+ buildImagePromptWithOverflow,
36
40
  type BuildImagePromptResult,
37
41
  } from "./prompt-builder.js"
38
- import { getFramingPromptHint } from "./framing.js"
39
- import { getLightingPromptHint } from "./lighting.js"
40
- import { getLensPromptHint } from "./lens.js"
41
- import { getCameraFormatPromptHint } from "./camera-format.js"
42
42
  import {
43
43
  renderStructuredFields,
44
44
  type StructuredPromptFields,
45
45
  } from "./prompt-builder-structured-fields.js"
46
+ import {
47
+ renderDirectionHints,
48
+ IMAGE_HINT_MODE_DEFAULT,
49
+ type DirectionFields,
50
+ } from "./direction-registry.js"
51
+ import {
52
+ renderSubjectHints,
53
+ SUBJECT_IMAGE_HINT_MODE_DEFAULT,
54
+ type SubjectFields,
55
+ } from "./subject-registry.js"
56
+ import { joinPromptHints } from "./prompt-hint-join.js"
57
+ import { keepableDirectionHints } from "./hint-shedding.js"
46
58
  import type { CharacterDef, ConnectedReference, IdentityMeta } from "@nodaro/shared"
47
59
 
48
60
  /**
49
- * Flat cinematic-direction ids the Studio framing UI (and the MCP route)
50
- * expose — all optional. Promoted here from Studio's `assembly.ts` so the
51
- * id → hint composition lives in one place. The platform callers pass none of
52
- * these (they fold their hints from the graph into `userPrompt` themselves).
61
+ * Flat cinematic-direction ids the Studio framing UI, the MCP route and the
62
+ * canvas node data expose — all optional. The dimensions, their canonical fold
63
+ * ORDER and their per-catalog rendering live in `direction-registry.ts`; this
64
+ * re-export keeps the import path stable for existing consumers. The platform
65
+ * callers fold their GRAPH-WIRED hints into `userPrompt` themselves and pass
66
+ * these only when the node carries them as stored data (Studio-emitted graphs,
67
+ * spec D3).
53
68
  */
54
- export interface DirectionFields {
55
- /** Shot Type — the FRAMINGS shot-size/coverage/composition/vantage dimensions. */
56
- framingId?: string
57
- /** Angle — the FRAMINGS angle dimension (separate pill, so it can coexist with Shot Type). */
58
- framingAngleId?: string
59
- lightingId?: string
60
- lensId?: string
61
- cameraFormatId?: string
62
- }
69
+ export type { DirectionFields }
70
+
71
+ /**
72
+ * Flat SUBJECT ids (Person / Styling / prop catalogs) — the companion channel
73
+ * to `direction`, describing WHO is in the shot rather than how it is shot. Its
74
+ * table, fold order and per-catalog rendering live in `subject-registry.ts`;
75
+ * this re-export keeps one import path for a consumer that takes both levers.
76
+ */
77
+ export type { SubjectFields }
63
78
 
64
79
  /**
65
80
  * Input to `assembleImageInput`. A faithful SUPERSET of what the two platform
@@ -80,10 +95,18 @@ export interface AssembleImageInput {
80
95
  connectedReferences?: ConnectedReference[]
81
96
  /**
82
97
  * Flat cinematic-direction ids → folded into the prompt as hints. Studio /
83
- * MCP-route use; the platform callers pass none (so `composePromptText` is a
84
- * no-op for them and the result is byte-identical to today).
98
+ * MCP-route use, and the platform callers' narrow-read of a node's STORED
99
+ * `data.direction`; absent on a node that carries none (so `composePromptText`
100
+ * is a no-op for it and the result is byte-identical to today).
85
101
  */
86
102
  direction?: DirectionFields
103
+ /**
104
+ * Flat subject ids (Person / Styling / props) → folded into the prompt AHEAD
105
+ * of the direction clauses: the subject is the noun phrase the cinematography
106
+ * then modifies. Same provenance as `direction` — Studio / MCP-route, or the
107
+ * platform callers' narrow-read of a node's STORED `data.subject`.
108
+ */
109
+ subject?: SubjectFields
87
110
  /** Path-1 structured fields → composed fragment appended to the prompt. */
88
111
  structured?: StructuredPromptFields
89
112
  /**
@@ -141,55 +164,118 @@ export interface AssembleImageInput {
141
164
  }
142
165
 
143
166
  /**
144
- * Compose the cinematic-direction hints + structured-field fragment with the
145
- * user's prompt. Each `get*PromptHint` returns "" on a miss, and
146
- * `renderStructuredFields` returns "" when nothing is populated.
167
+ * The hint pieces a fold contributes, split by whether the assembler may SHED
168
+ * them under a provider prompt cap.
147
169
  *
148
- * EXACT NO-OP CONTRACT: when there are no cinematic/structured hint pieces (the
149
- * platform-caller case — execute-node / payload-builder never pass `direction`/
150
- * `structured`), the user's prompt is returned **verbatim, untrimmed**. This is
151
- * load-bearing for parity: the old platform path passed the prompt straight to
152
- * `buildImagePrompt`, which never trims, so trimming here would change the
153
- * assembled prompt (and the recorded `jobs.input_data`) byte-for-byte. We only
154
- * trim the user prompt when joining it WITH hints, so it reads cleanly
155
- * ("prompt. hint", not "prompt . hint"). Never mutates inputs.
170
+ * `hintClauses` are the catalog-rendered clauses — the SUBJECT fold first, then
171
+ * the cinematic direction fold — decorative garnish next to a reference
172
+ * directive or the user's own prose, and the only thing this assembler drops
173
+ * when the prompt won't fit. They are ONE list because the shed walks it from
174
+ * the TAIL: direction leaves before subject, which is the right order (who is
175
+ * in the shot outranks how it is lit). `structuredFragment` is user CONTENT (a
176
+ * Path-1 structured field the caller populated), so it is sticky and always
177
+ * lands LAST, exactly as before.
156
178
  */
157
- function composePromptText(
158
- userPrompt: string,
179
+ interface ImageHintPieces {
180
+ /** Subject clauses then direction clauses, each in its registry's fold order. */
181
+ readonly hintClauses: readonly string[]
182
+ /** The structured-field fragment ("" when nothing is populated). */
183
+ readonly structuredFragment: string
184
+ }
185
+
186
+ /**
187
+ * Render the fold's hint pieces once, so the cap-aware retry can re-join a
188
+ * SUBSET of them without re-rendering the catalogs. Each renderer folds its own
189
+ * channel in its registry's canonical table order (unknown keys and unknown ids
190
+ * contribute nothing), and `renderStructuredFields` returns "" when nothing is
191
+ * populated. Never mutates inputs.
192
+ *
193
+ * SUBJECT LEADS DIRECTION: the subject is the noun phrase ("a woman in her
194
+ * 30s, …") the cinematographic clauses then modify. With no `subject` the list
195
+ * IS the direction fold, so every existing caller's prompt is byte-identical.
196
+ */
197
+ function renderImageHintPieces(
198
+ subject: SubjectFields | undefined,
159
199
  direction: DirectionFields | undefined,
160
200
  structured: StructuredPromptFields | undefined,
201
+ ): ImageHintPieces {
202
+ return {
203
+ hintClauses: [
204
+ ...renderSubjectHints(subject, {
205
+ surface: "image",
206
+ mode: SUBJECT_IMAGE_HINT_MODE_DEFAULT,
207
+ }),
208
+ ...renderDirectionHints(direction, {
209
+ surface: "image",
210
+ mode: IMAGE_HINT_MODE_DEFAULT,
211
+ }),
212
+ ].filter((p) => p.length > 0),
213
+ structuredFragment: structured ? renderStructuredFields(structured) : "",
214
+ }
215
+ }
216
+
217
+ /**
218
+ * Compose the subject + cinematic-direction hints and the structured-field
219
+ * fragment with the user's prompt, keeping the FIRST `keptHintClauses` clauses
220
+ * (the full count on the first pass; fewer only when the provider cap forced a
221
+ * shed). The structured fragment always lands LAST.
222
+ *
223
+ * EXACT NO-OP CONTRACT: when there are no subject/cinematic/structured hint
224
+ * pieces (the platform-caller case for a node that carries no stored `subject`/
225
+ * `direction`/`structured` — every workflow authored before the canvas honored
226
+ * them), the
227
+ * user's prompt is returned **verbatim, untrimmed** by `joinPromptHints`. This
228
+ * is load-bearing for parity: the old platform path passed the prompt straight
229
+ * to `buildImagePrompt`, which never trims, so trimming here would change the
230
+ * assembled prompt (and the recorded `jobs.input_data`) byte-for-byte. Never
231
+ * mutates inputs.
232
+ *
233
+ * A node that DOES carry `direction`/`structured` takes the join branch and is
234
+ * therefore trimmed + `". "`-joined — intended, and asserted at the caller
235
+ * level by the payload-builder before/after test.
236
+ */
237
+ function composePromptText(
238
+ userPrompt: string,
239
+ pieces: ImageHintPieces,
240
+ keptHintClauses: number,
161
241
  ): string {
162
242
  const hints = [
163
- getFramingPromptHint(direction?.framingId),
164
- getFramingPromptHint(direction?.framingAngleId),
165
- getLightingPromptHint(direction?.lightingId),
166
- getLensPromptHint(direction?.lensId),
167
- getCameraFormatPromptHint(direction?.cameraFormatId),
168
- structured ? renderStructuredFields(structured) : "",
243
+ ...pieces.hintClauses.slice(0, keptHintClauses),
244
+ pieces.structuredFragment,
169
245
  ].filter((p) => p.length > 0)
170
- // No hints → verbatim (exact no-op = platform parity). With hints → trim the
171
- // user prompt so the ". " join is clean. The trailing filter drops a blank
172
- // user prompt so the join never starts with ". " (parity-critical — don't
173
- // remove it as "redundant": `hints` is pre-filtered but `userPrompt` is not).
174
- if (hints.length === 0) return userPrompt
175
- return [userPrompt.trim(), ...hints].filter((p) => p.length > 0).join(". ")
246
+ return joinPromptHints(userPrompt, hints)
176
247
  }
177
248
 
178
249
  /**
179
250
  * Assemble a node's image-generation inputs into a `BuildImagePromptResult`
180
251
  * (`{ prompt, nativeNegativePrompt, referenceImageUrls }`).
181
252
  *
182
- * Order: (1) compose the prompt text (no-op when no direction/structured),
253
+ * Order: (1) compose the prompt text (no-op when no subject/direction/structured),
183
254
  * (2) `buildImagePrompt(...)` — exactly the call the three sites make today,
184
- * (3) optional post-assembly empty-prompt throw (gated by `throwOnEmpty`).
255
+ * (3) shed hint clauses and re-assemble while the provider cap overflows,
256
+ * (4) optional post-assembly empty-prompt throw (gated by `throwOnEmpty`).
257
+ *
258
+ * TRUNCATION ORDERING (step 3): `buildImagePrompt`'s cap clamp cuts the TAIL,
259
+ * which is ORDER-BLIND — on a low-cap provider (seedream = 3000) a maximal
260
+ * direction fold renders ~3.3K characters of clauses and the cut can sever a
261
+ * reference directive, mention-resolved text or the user's own prose while a
262
+ * decorative clause survives. So the ASSEMBLER decides instead: it knows which
263
+ * clauses are hints because it just built them, and drops them last-folded
264
+ * first until the prompt fits. Everything else — references, prose, the
265
+ * structured fragment, the Style/Avoid suffixes — outranks a hint. A body that
266
+ * still overflows with ZERO hints (long prose or many directives on its own)
267
+ * falls back to the builder's clamp, unchanged.
268
+ *
269
+ * UNDER-CAP PARITY: the first pass folds every hint, so a prompt that fits is
270
+ * byte-identical to before — the retry only ever runs on an over-cap assembly.
185
271
  */
186
272
  export function assembleImageInput(
187
273
  input: AssembleImageInput,
188
274
  ): BuildImagePromptResult {
189
- const prompt = composePromptText(input.userPrompt, input.direction, input.structured)
275
+ const pieces = renderImageHintPieces(input.subject, input.direction, input.structured)
190
276
 
191
- const result = buildImagePrompt({
192
- prompt,
277
+ const assembleWith = (keptHintClauses: number) => buildImagePromptWithOverflow({
278
+ prompt: composePromptText(input.userPrompt, pieces, keptHintClauses),
193
279
  provider: input.provider,
194
280
  ...(input.connectedReferences !== undefined
195
281
  ? { connectedReferences: input.connectedReferences }
@@ -223,8 +309,24 @@ export function assembleImageInput(
223
309
  : {}),
224
310
  })
225
311
 
312
+ // Fold everything first (the under-cap byte-parity pass), then shed hints
313
+ // from the tail of the COMBINED fold order (subject clauses first in the list,
314
+ // therefore last to leave) while the assembled prompt overflows the provider
315
+ // cap. `keepableDirectionHints` — the one shed arithmetic, shared with
316
+ // `composeVideoPromptText` — strictly decreases `kept` whenever there IS an
317
+ // overflow, so this terminates at `kept === 0` in the worst case, at which
318
+ // point the body overflows on its own and the builder's clamp stands.
319
+ let kept = pieces.hintClauses.length
320
+ let fitted = assembleWith(kept)
321
+ while (fitted.overflowChars > 0 && kept > 0) {
322
+ kept = keepableDirectionHints(pieces.hintClauses, kept, fitted.overflowChars)
323
+ fitted = assembleWith(kept)
324
+ }
325
+ // `overflowChars` is assembly bookkeeping, not part of the callers' contract.
326
+ const { overflowChars, ...result } = fitted
327
+
226
328
  // Post-assembly empty-prompt check (opt-in): a bound entity / `@`-mention /
227
- // direction chip could have filled the assembled prompt even if the user
329
+ // subject or direction chip could have filled the assembled prompt even if the user
228
330
  // typed nothing — so only reject when the FINAL prompt is truly empty.
229
331
  if (input.throwOnEmpty && !result.prompt.trim()) {
230
332
  throw new Error(
@@ -0,0 +1,244 @@
1
+ /**
2
+ * `composeVideoPromptText` — the video twin of `assemble-image-input.ts`'s
3
+ * `composePromptText`: fold cinematic-direction picker IDS into the prompt BODY,
4
+ * server-side, at the model call.
5
+ *
6
+ * WHY THIS EXISTS: `/v1/generate-video` had no structured direction channel, so
7
+ * every client baked the hint TEXT itself. A copied scene then carried stale
8
+ * catalog wording forever, a re-generate double-baked it, and each client
9
+ * re-implemented the fold with its own separator and its own order. The wire
10
+ * now carries ids; the platform renders the clauses.
11
+ *
12
+ * WHERE IT RUNS (load-bearing): on the prompt BODY, BEFORE
13
+ * `resolveVideoReferenceCore`. That resolver FRAMES the body — legacy prepends
14
+ * its `Use these characters:` block, hybrid prepends the lock lines and APPENDS
15
+ * the canonical role phrases and extras. Folding afterwards would push the
16
+ * scene/look description PAST the identity directives, a worse version of the
17
+ * bug this channel exists to fix. The image side is structurally identical
18
+ * (`assembleImageInput` = `composePromptText` → `buildImagePrompt`).
19
+ *
20
+ * That ordering is also why cap-aware shedding here takes a `frame` callback
21
+ * rather than a provider id: the shed must run at the FOLD site (before the
22
+ * resolver) but be decided on the RESOLVED length (after it), so the binding
23
+ * text the resolver adds is inside the budget and can never be the thing that
24
+ * gets dropped. See {@link VideoPromptCapOptions}. Both catalog channels —
25
+ * SUBJECT and direction — fold into the one sheddable list that budget walks.
26
+ *
27
+ * THE VERBOSITY POLICY LIVES HERE, NOT IN THE CLIENT: motion dimensions render
28
+ * their compact professional term, look dimensions their full clause
29
+ * (`VIDEO_HINT_MODE_DEFAULT`, resolved per row's `family` by the registry). The
30
+ * SUBJECT fold has its own policy — compact on video
31
+ * (`SUBJECT_VIDEO_HINT_MODE_DEFAULT`), because a fully specified person at full
32
+ * verbosity is ~30 paragraph clauses and the start frame already carries the
33
+ * subject's identity into the clip.
34
+ * It is a threaded PARAMETER with a pure default — never deployment state:
35
+ * `__tests__/content-free-contract.test.ts` hard-fails any environment read
36
+ * under `packages/prompts/src`, and this module has nothing to read anyway.
37
+ *
38
+ * EXACT NO-OP CONTRACT: with no subject, no direction and no structured fields
39
+ * the caller's
40
+ * `userPrompt` comes back VERBATIM AND UNTRIMMED — `undefined` included, since
41
+ * a video prompt is optional on the route. That is what keeps every existing
42
+ * caller byte-identical (the "backward-compatible: no connectedReferences →
43
+ * prompt + flat refs pass through unchanged" oracle in
44
+ * `backend/src/routes/__tests__/generate-video.test.ts`, restated locally in
45
+ * `__tests__/assemble-video-input.test.ts`).
46
+ *
47
+ * WHAT IS DELIBERATELY NOT HERE: the dimension table, the fold order, the
48
+ * dedupe and the surface filter all live in `direction-registry.ts` (and
49
+ * `subject-registry.ts` for the subject channel) — ONE renderer per channel
50
+ * serves both surfaces, so the image and video folds cannot drift.
51
+ * Clients render their "will inject into prompt" preview by importing
52
+ * `renderDirectionHints` + `joinPromptHints` directly.
53
+ */
54
+ import {
55
+ renderDirectionHints,
56
+ VIDEO_HINT_MODE_DEFAULT,
57
+ type DirectionFields,
58
+ type DirectionHintMode,
59
+ } from "./direction-registry.js"
60
+ import {
61
+ renderSubjectHints,
62
+ SUBJECT_VIDEO_HINT_MODE_DEFAULT,
63
+ type SubjectFields,
64
+ type SubjectHintMode,
65
+ } from "./subject-registry.js"
66
+ import { joinPromptHints } from "./prompt-hint-join.js"
67
+ import { keepableDirectionHints } from "./hint-shedding.js"
68
+ import {
69
+ renderStructuredFields,
70
+ type StructuredPromptFields,
71
+ } from "./prompt-builder-structured-fields.js"
72
+
73
+ /**
74
+ * Cap-aware shedding, opt-in. Absent → the composer is exactly what it always
75
+ * was (every existing caller stays byte-identical, and the no-op path below is
76
+ * never even reached differently).
77
+ *
78
+ * WHY A NUMBER AND A CALLBACK, NOT A PROVIDER ID — the two halves of the video
79
+ * surface's problem, which the image half did not have:
80
+ *
81
+ * - `cap` is the caller's EFFECTIVE ceiling, not `getMaxVideoPromptChars` read
82
+ * here. The routes compute it with `effectiveVideoPromptCeiling`, which
83
+ * mirrors `applyVideoNegativePrompt`'s reservation of the `"\nAvoid: …"`
84
+ * suffix for a provider with no native negative param. Re-deriving the cap
85
+ * inside this package would put a second copy of that reservation one
86
+ * refactor away from drifting from the clamp it is supposed to predict.
87
+ *
88
+ * - `frame` is the REFERENCE RESOLVER, and it is what makes the shed correct
89
+ * end-to-end. The fold runs BEFORE `resolveVideoReferenceCore` (see the
90
+ * module header — folding afterwards strands the scene description past the
91
+ * identity directives). The resolver then ADDS binding text: legacy's
92
+ * "Use these characters:" block, hybrid's lock lines and the canonical role
93
+ * phrases it APPENDS. That added text is exactly what an order-blind tail cut
94
+ * destroys first, so it must be inside the budget — but it must never be
95
+ * shed. Measuring THROUGH the caller's framing gives both properties at once:
96
+ * the shed decision sees the final length, while the only thing it can drop
97
+ * is a hint clause it rendered itself.
98
+ *
99
+ * Re-framing a SUBSET of the hints is sound because a hint can never change how
100
+ * the resolver reads the rest of the body: no registered catalog hint, term or
101
+ * label contains a `{image:N}` / `{ref:` / `@slug:N` shape
102
+ * (`__tests__/direction-hint-token-safety.test.ts` pins that for every catalog),
103
+ * so dropping one cannot renumber or unbind a reference.
104
+ */
105
+ export interface VideoPromptCapOptions {
106
+ /**
107
+ * The maximum length the FRAMED prompt may reach. Sheds only while the framed
108
+ * body exceeds it; `undefined` (the default) disables shedding entirely.
109
+ */
110
+ readonly cap?: number
111
+ /**
112
+ * The downstream framing the cap is measured through — the caller's reference
113
+ * assembly. Identity when omitted (a caller with a cap but no references).
114
+ * Must be PURE: it is called once per shed iteration, and the caller re-runs
115
+ * its own real assembly on the returned body afterwards.
116
+ */
117
+ readonly frame?: (body: string | undefined) => string | undefined
118
+ }
119
+
120
+ /**
121
+ * Fold a video run's subject and cinematic-direction ids (and optional
122
+ * structured fields) into its prompt body.
123
+ *
124
+ * The SUBJECT hints land first (who is in the shot — the noun phrase the
125
+ * cinematography modifies), then the direction hints in the registry's
126
+ * canonical table order (camera motion leads), and the structured fragment
127
+ * lands LAST — the same ordering `composePromptText` uses for stills.
128
+ *
129
+ * `subject` rides `opts` rather than a fourth positional parameter on purpose:
130
+ * every existing caller passes `(prompt, direction)` or
131
+ * `(prompt, direction, structured)` positionally, and a new positional would
132
+ * have made the two levers' order a memorization test.
133
+ *
134
+ * TRUNCATION ORDERING (opt-in via `opts.cap`): the provider clamp
135
+ * (`applyVideoNegativePrompt`) slices the prompt TAIL, which is ORDER-BLIND —
136
+ * on a low-cap provider (kling = 1000) a broad direction renders more than the
137
+ * whole ceiling and the cut severs reference bindings and the end of the user's
138
+ * prose while decorative clauses survive. With a cap the composer decides
139
+ * instead: it knows which clauses are hints because it just rendered them, and
140
+ * drops them LAST-FOLDED FIRST until the framed prompt fits. Everything else —
141
+ * the user's prose, the structured fragment (user CONTENT, never a garnish) and
142
+ * every byte the resolver's framing adds — outranks a hint.
143
+ *
144
+ * SUBJECT CLAUSES ARE SHED CANDIDATES TOO, and they shed AFTER the direction
145
+ * clauses. Both folds are catalog decoration of the same class — ids the
146
+ * platform rendered into wording — so exempting one would just move the
147
+ * overflow into the order-blind clamp, which is the bug this machinery exists
148
+ * to prevent. They ride the SAME `hintClauses` list the shed already walks
149
+ * (subject first, direction second, tail-first shedding), so there is exactly
150
+ * one shed arithmetic (`hint-shedding.ts`) across both channels and both
151
+ * surfaces. Neither channel ever sheds before the prose, the references or the
152
+ * structured fragment.
153
+ *
154
+ * WHAT THE BUDGET DELIBERATELY EXCLUDES: the route's later opt-in identity
155
+ * injection (an async DB read that appends a canonical description) and any
156
+ * registered `applyPromptPolicies` transform both run AFTER the reference
157
+ * assembly and are not modelled here. Pricing them in would mean folding an
158
+ * await into this pure composer; instead the provider clamp stays their last
159
+ * resort, exactly as today. Same for a body that still overflows with ZERO
160
+ * hints left — long prose, or many bound references on their own.
161
+ *
162
+ * UNDER-CAP PARITY: the first pass folds every hint, so a prompt that fits is
163
+ * byte-identical to a capless call, and a caller with no
164
+ * `subject`/`direction`/`structured` takes the same exact no-op path it always
165
+ * did.
166
+ *
167
+ * @param userPrompt The user's prompt. Optional: an image-to-video run may
168
+ * legitimately have none, and it is returned as-is when nothing folds.
169
+ * @param direction Flat catalog ids. Unknown keys, off-surface keys (an
170
+ * image-only dimension sent to a video run) and unknown ids all contribute
171
+ * nothing — never a throw.
172
+ * @param structured Path-1 structured fields. Not a `/v1/generate-video` wire
173
+ * field today; the canvas orchestrator passes it directly.
174
+ * @param opts.hintMode Override the direction verbosity policy (a whole-fold
175
+ * `PickerHintMode`, or a `{ look, motion }` split).
176
+ * @param opts.subject Flat subject ids (Person / Styling / props), same
177
+ * inertness contract as `direction`.
178
+ * @param opts.subjectHintMode Override the subject verbosity policy.
179
+ * @param opts.cap / `opts.frame` See {@link VideoPromptCapOptions}.
180
+ */
181
+ export function composeVideoPromptText(
182
+ userPrompt: string | undefined,
183
+ direction: DirectionFields | undefined,
184
+ structured?: StructuredPromptFields,
185
+ opts?: {
186
+ readonly hintMode?: DirectionHintMode
187
+ readonly subject?: SubjectFields
188
+ readonly subjectHintMode?: SubjectHintMode
189
+ } & VideoPromptCapOptions,
190
+ ): string | undefined {
191
+ // ONE sheddable list, subject FIRST then direction — because the shed walks it
192
+ // from the TAIL, so this order IS the survival order: a direction clause
193
+ // leaves before a subject clause. Deliberate, and the same order the image
194
+ // side uses (`renderImageHintPieces`): the subject is the noun phrase the
195
+ // cinematography modifies, so losing "who is in the shot" to keep a
196
+ // decorative grade would be the wrong trade. With no `subject` the list IS
197
+ // the direction fold, so every pre-subject caller is byte-identical.
198
+ const hintClauses = [
199
+ ...renderSubjectHints(opts?.subject, {
200
+ surface: "video",
201
+ mode: opts?.subjectHintMode ?? SUBJECT_VIDEO_HINT_MODE_DEFAULT,
202
+ }),
203
+ ...renderDirectionHints(direction, {
204
+ surface: "video",
205
+ mode: opts?.hintMode ?? VIDEO_HINT_MODE_DEFAULT,
206
+ }),
207
+ ].filter((p) => p.length > 0)
208
+ // User CONTENT, not a garnish: never sheddable, always last.
209
+ const structuredFragment = structured ? renderStructuredFields(structured) : ""
210
+
211
+ const composeWith = (kept: number): string | undefined => {
212
+ const hints = [...hintClauses.slice(0, kept), structuredFragment].filter(
213
+ (p) => p.length > 0,
214
+ )
215
+ // Nothing to fold → the caller's value straight back, `undefined` included.
216
+ // Do NOT collapse this into `joinPromptHints(userPrompt ?? "", hints)`: that
217
+ // would turn an absent prompt into `""` and break the no-op contract above.
218
+ // A FULL shed lands here too, which is what keeps the no-op contract intact
219
+ // at `kept === 0` — the route's `composed !== prompt` guard then correctly
220
+ // leaves `input_data.userPrompt` unpinned.
221
+ if (hints.length === 0) return userPrompt
222
+ return joinPromptHints(userPrompt ?? "", hints)
223
+ }
224
+
225
+ const cap = opts?.cap
226
+ if (cap === undefined) return composeWith(hintClauses.length)
227
+
228
+ // Fold everything first (the under-cap byte-parity pass), then shed from the
229
+ // tail of the fold order while the FRAMED prompt overflows the ceiling.
230
+ // `keepableDirectionHints` — the ONE shed arithmetic, shared with the image
231
+ // assembler — strictly decreases `kept` whenever there is a deficit, so this
232
+ // terminates at `kept === 0` in the worst case, at which point nothing
233
+ // droppable is left and the provider clamp stands.
234
+ const frame = opts?.frame ?? ((body: string | undefined) => body)
235
+ let kept = hintClauses.length
236
+ let body = composeWith(kept)
237
+ let framedLength = frame(body)?.length ?? 0
238
+ while (framedLength > cap && kept > 0) {
239
+ kept = keepableDirectionHints(hintClauses, kept, framedLength - cap)
240
+ body = composeWith(kept)
241
+ framedLength = frame(body)?.length ?? 0
242
+ }
243
+ return body
244
+ }