@nodaro/prompts 1.10.0 → 1.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +693 -53
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1084 -27
- package/dist/index.d.ts +1084 -27
- package/dist/index.js +664 -55
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/__tests__/__snapshots__/entity-convergence-image.test.ts.snap +19 -0
- package/src/__tests__/animal-getters-parity.test.ts +82 -0
- package/src/__tests__/assemble-image-input-cap.test.ts +212 -0
- package/src/__tests__/assemble-image-input.test.ts +93 -3
- package/src/__tests__/assemble-video-input-cap.test.ts +356 -0
- package/src/__tests__/assemble-video-input.test.ts +301 -0
- package/src/__tests__/direction-hint-token-safety.test.ts +113 -0
- package/src/__tests__/direction-registry.test.ts +393 -0
- package/src/__tests__/entity-convergence-image.test.ts +374 -0
- package/src/__tests__/image-convergence-image.test.ts +370 -0
- package/src/__tests__/location-convergence-image.test.ts +29 -1
- package/src/__tests__/location-default-role-image.test.ts +166 -0
- package/src/__tests__/mention-splice-spacing.test.ts +257 -0
- package/src/__tests__/read-node-direction.test.ts +154 -0
- package/src/__tests__/read-node-subject.test.ts +140 -0
- package/src/__tests__/subject-fold.test.ts +232 -0
- package/src/__tests__/subject-registry.test.ts +312 -0
- package/src/assemble-image-input.ts +160 -58
- package/src/assemble-video-input.ts +244 -0
- package/src/direction-registry.ts +371 -0
- package/src/hint-shedding.ts +68 -0
- package/src/index.ts +10 -2
- package/src/parameter-prompt-hint.ts +8 -7
- package/src/picker-catalogs.ts +14 -7
- package/src/prompt-builder.ts +728 -58
- package/src/prompt-hint-join.ts +30 -0
- package/src/read-node-direction.ts +233 -0
- package/src/subject-registry.ts +464 -0
|
@@ -16,13 +16,17 @@
|
|
|
16
16
|
* This wrapper collapses them into one.
|
|
17
17
|
*
|
|
18
18
|
* THE NO-OP CONTRACT (load-bearing — the platform-caller parity relies on it):
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
* (
|
|
25
|
-
*
|
|
19
|
+
* a node that carries NO stored `subject` / `direction` / `structured` (every
|
|
20
|
+
* workflow authored before the canvas honored them) still reaches here with all
|
|
21
|
+
* three absent, and `composePromptText` MUST return the caller's `userPrompt`
|
|
22
|
+
* byte-for-byte unchanged, so the wrapper degenerates to exactly the
|
|
23
|
+
* `buildImagePrompt(...)` call those sites made before. The platform callers
|
|
24
|
+
* (`execute-node` / `payload-builder`) compose their prompt from the canvas GRAPH themselves and
|
|
25
|
+
* ALSO forward a node's STORED `subject` / `direction` / `structured` when it
|
|
26
|
+
* carries them (`readSubjectFields` / `readDirectionFields` /
|
|
27
|
+
* `readStructuredFields`); those nodes get the id-hint composition on top,
|
|
28
|
+
* ADDITIVE to the graph-wired cinematography hints the caller already folded
|
|
29
|
+
* into `userPrompt`. Studio and the MCP route supply the levers directly.
|
|
26
30
|
*
|
|
27
31
|
* THE EMPTY-CHECK FLAG (also load-bearing for parity): `execute-node` rejects a
|
|
28
32
|
* truly-empty assembled prompt (its "type one, mention a character, or connect
|
|
@@ -32,34 +36,45 @@
|
|
|
32
36
|
* guard (frontend, Studio, route) pass `throwOnEmpty: true`.
|
|
33
37
|
*/
|
|
34
38
|
import {
|
|
35
|
-
|
|
39
|
+
buildImagePromptWithOverflow,
|
|
36
40
|
type BuildImagePromptResult,
|
|
37
41
|
} from "./prompt-builder.js"
|
|
38
|
-
import { getFramingPromptHint } from "./framing.js"
|
|
39
|
-
import { getLightingPromptHint } from "./lighting.js"
|
|
40
|
-
import { getLensPromptHint } from "./lens.js"
|
|
41
|
-
import { getCameraFormatPromptHint } from "./camera-format.js"
|
|
42
42
|
import {
|
|
43
43
|
renderStructuredFields,
|
|
44
44
|
type StructuredPromptFields,
|
|
45
45
|
} from "./prompt-builder-structured-fields.js"
|
|
46
|
+
import {
|
|
47
|
+
renderDirectionHints,
|
|
48
|
+
IMAGE_HINT_MODE_DEFAULT,
|
|
49
|
+
type DirectionFields,
|
|
50
|
+
} from "./direction-registry.js"
|
|
51
|
+
import {
|
|
52
|
+
renderSubjectHints,
|
|
53
|
+
SUBJECT_IMAGE_HINT_MODE_DEFAULT,
|
|
54
|
+
type SubjectFields,
|
|
55
|
+
} from "./subject-registry.js"
|
|
56
|
+
import { joinPromptHints } from "./prompt-hint-join.js"
|
|
57
|
+
import { keepableDirectionHints } from "./hint-shedding.js"
|
|
46
58
|
import type { CharacterDef, ConnectedReference, IdentityMeta } from "@nodaro/shared"
|
|
47
59
|
|
|
48
60
|
/**
|
|
49
|
-
* Flat cinematic-direction ids the Studio framing UI
|
|
50
|
-
* expose — all optional.
|
|
51
|
-
*
|
|
52
|
-
*
|
|
61
|
+
* Flat cinematic-direction ids the Studio framing UI, the MCP route and the
|
|
62
|
+
* canvas node data expose — all optional. The dimensions, their canonical fold
|
|
63
|
+
* ORDER and their per-catalog rendering live in `direction-registry.ts`; this
|
|
64
|
+
* re-export keeps the import path stable for existing consumers. The platform
|
|
65
|
+
* callers fold their GRAPH-WIRED hints into `userPrompt` themselves and pass
|
|
66
|
+
* these only when the node carries them as stored data (Studio-emitted graphs,
|
|
67
|
+
* spec D3).
|
|
53
68
|
*/
|
|
54
|
-
export
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
}
|
|
69
|
+
export type { DirectionFields }
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Flat SUBJECT ids (Person / Styling / prop catalogs) — the companion channel
|
|
73
|
+
* to `direction`, describing WHO is in the shot rather than how it is shot. Its
|
|
74
|
+
* table, fold order and per-catalog rendering live in `subject-registry.ts`;
|
|
75
|
+
* this re-export keeps one import path for a consumer that takes both levers.
|
|
76
|
+
*/
|
|
77
|
+
export type { SubjectFields }
|
|
63
78
|
|
|
64
79
|
/**
|
|
65
80
|
* Input to `assembleImageInput`. A faithful SUPERSET of what the two platform
|
|
@@ -80,10 +95,18 @@ export interface AssembleImageInput {
|
|
|
80
95
|
connectedReferences?: ConnectedReference[]
|
|
81
96
|
/**
|
|
82
97
|
* Flat cinematic-direction ids → folded into the prompt as hints. Studio /
|
|
83
|
-
* MCP-route use
|
|
84
|
-
*
|
|
98
|
+
* MCP-route use, and the platform callers' narrow-read of a node's STORED
|
|
99
|
+
* `data.direction`; absent on a node that carries none (so `composePromptText`
|
|
100
|
+
* is a no-op for it and the result is byte-identical to today).
|
|
85
101
|
*/
|
|
86
102
|
direction?: DirectionFields
|
|
103
|
+
/**
|
|
104
|
+
* Flat subject ids (Person / Styling / props) → folded into the prompt AHEAD
|
|
105
|
+
* of the direction clauses: the subject is the noun phrase the cinematography
|
|
106
|
+
* then modifies. Same provenance as `direction` — Studio / MCP-route, or the
|
|
107
|
+
* platform callers' narrow-read of a node's STORED `data.subject`.
|
|
108
|
+
*/
|
|
109
|
+
subject?: SubjectFields
|
|
87
110
|
/** Path-1 structured fields → composed fragment appended to the prompt. */
|
|
88
111
|
structured?: StructuredPromptFields
|
|
89
112
|
/**
|
|
@@ -141,55 +164,118 @@ export interface AssembleImageInput {
|
|
|
141
164
|
}
|
|
142
165
|
|
|
143
166
|
/**
|
|
144
|
-
*
|
|
145
|
-
*
|
|
146
|
-
* `renderStructuredFields` returns "" when nothing is populated.
|
|
167
|
+
* The hint pieces a fold contributes, split by whether the assembler may SHED
|
|
168
|
+
* them under a provider prompt cap.
|
|
147
169
|
*
|
|
148
|
-
*
|
|
149
|
-
*
|
|
150
|
-
*
|
|
151
|
-
*
|
|
152
|
-
*
|
|
153
|
-
*
|
|
154
|
-
*
|
|
155
|
-
*
|
|
170
|
+
* `hintClauses` are the catalog-rendered clauses — the SUBJECT fold first, then
|
|
171
|
+
* the cinematic direction fold — decorative garnish next to a reference
|
|
172
|
+
* directive or the user's own prose, and the only thing this assembler drops
|
|
173
|
+
* when the prompt won't fit. They are ONE list because the shed walks it from
|
|
174
|
+
* the TAIL: direction leaves before subject, which is the right order (who is
|
|
175
|
+
* in the shot outranks how it is lit). `structuredFragment` is user CONTENT (a
|
|
176
|
+
* Path-1 structured field the caller populated), so it is sticky and always
|
|
177
|
+
* lands LAST, exactly as before.
|
|
156
178
|
*/
|
|
157
|
-
|
|
158
|
-
|
|
179
|
+
interface ImageHintPieces {
|
|
180
|
+
/** Subject clauses then direction clauses, each in its registry's fold order. */
|
|
181
|
+
readonly hintClauses: readonly string[]
|
|
182
|
+
/** The structured-field fragment ("" when nothing is populated). */
|
|
183
|
+
readonly structuredFragment: string
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Render the fold's hint pieces once, so the cap-aware retry can re-join a
|
|
188
|
+
* SUBSET of them without re-rendering the catalogs. Each renderer folds its own
|
|
189
|
+
* channel in its registry's canonical table order (unknown keys and unknown ids
|
|
190
|
+
* contribute nothing), and `renderStructuredFields` returns "" when nothing is
|
|
191
|
+
* populated. Never mutates inputs.
|
|
192
|
+
*
|
|
193
|
+
* SUBJECT LEADS DIRECTION: the subject is the noun phrase ("a woman in her
|
|
194
|
+
* 30s, …") the cinematographic clauses then modify. With no `subject` the list
|
|
195
|
+
* IS the direction fold, so every existing caller's prompt is byte-identical.
|
|
196
|
+
*/
|
|
197
|
+
function renderImageHintPieces(
|
|
198
|
+
subject: SubjectFields | undefined,
|
|
159
199
|
direction: DirectionFields | undefined,
|
|
160
200
|
structured: StructuredPromptFields | undefined,
|
|
201
|
+
): ImageHintPieces {
|
|
202
|
+
return {
|
|
203
|
+
hintClauses: [
|
|
204
|
+
...renderSubjectHints(subject, {
|
|
205
|
+
surface: "image",
|
|
206
|
+
mode: SUBJECT_IMAGE_HINT_MODE_DEFAULT,
|
|
207
|
+
}),
|
|
208
|
+
...renderDirectionHints(direction, {
|
|
209
|
+
surface: "image",
|
|
210
|
+
mode: IMAGE_HINT_MODE_DEFAULT,
|
|
211
|
+
}),
|
|
212
|
+
].filter((p) => p.length > 0),
|
|
213
|
+
structuredFragment: structured ? renderStructuredFields(structured) : "",
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
/**
|
|
218
|
+
* Compose the subject + cinematic-direction hints and the structured-field
|
|
219
|
+
* fragment with the user's prompt, keeping the FIRST `keptHintClauses` clauses
|
|
220
|
+
* (the full count on the first pass; fewer only when the provider cap forced a
|
|
221
|
+
* shed). The structured fragment always lands LAST.
|
|
222
|
+
*
|
|
223
|
+
* EXACT NO-OP CONTRACT: when there are no subject/cinematic/structured hint
|
|
224
|
+
* pieces (the platform-caller case for a node that carries no stored `subject`/
|
|
225
|
+
* `direction`/`structured` — every workflow authored before the canvas honored
|
|
226
|
+
* them), the
|
|
227
|
+
* user's prompt is returned **verbatim, untrimmed** by `joinPromptHints`. This
|
|
228
|
+
* is load-bearing for parity: the old platform path passed the prompt straight
|
|
229
|
+
* to `buildImagePrompt`, which never trims, so trimming here would change the
|
|
230
|
+
* assembled prompt (and the recorded `jobs.input_data`) byte-for-byte. Never
|
|
231
|
+
* mutates inputs.
|
|
232
|
+
*
|
|
233
|
+
* A node that DOES carry `direction`/`structured` takes the join branch and is
|
|
234
|
+
* therefore trimmed + `". "`-joined — intended, and asserted at the caller
|
|
235
|
+
* level by the payload-builder before/after test.
|
|
236
|
+
*/
|
|
237
|
+
function composePromptText(
|
|
238
|
+
userPrompt: string,
|
|
239
|
+
pieces: ImageHintPieces,
|
|
240
|
+
keptHintClauses: number,
|
|
161
241
|
): string {
|
|
162
242
|
const hints = [
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
getLightingPromptHint(direction?.lightingId),
|
|
166
|
-
getLensPromptHint(direction?.lensId),
|
|
167
|
-
getCameraFormatPromptHint(direction?.cameraFormatId),
|
|
168
|
-
structured ? renderStructuredFields(structured) : "",
|
|
243
|
+
...pieces.hintClauses.slice(0, keptHintClauses),
|
|
244
|
+
pieces.structuredFragment,
|
|
169
245
|
].filter((p) => p.length > 0)
|
|
170
|
-
|
|
171
|
-
// user prompt so the ". " join is clean. The trailing filter drops a blank
|
|
172
|
-
// user prompt so the join never starts with ". " (parity-critical — don't
|
|
173
|
-
// remove it as "redundant": `hints` is pre-filtered but `userPrompt` is not).
|
|
174
|
-
if (hints.length === 0) return userPrompt
|
|
175
|
-
return [userPrompt.trim(), ...hints].filter((p) => p.length > 0).join(". ")
|
|
246
|
+
return joinPromptHints(userPrompt, hints)
|
|
176
247
|
}
|
|
177
248
|
|
|
178
249
|
/**
|
|
179
250
|
* Assemble a node's image-generation inputs into a `BuildImagePromptResult`
|
|
180
251
|
* (`{ prompt, nativeNegativePrompt, referenceImageUrls }`).
|
|
181
252
|
*
|
|
182
|
-
* Order: (1) compose the prompt text (no-op when no direction/structured),
|
|
253
|
+
* Order: (1) compose the prompt text (no-op when no subject/direction/structured),
|
|
183
254
|
* (2) `buildImagePrompt(...)` — exactly the call the three sites make today,
|
|
184
|
-
* (3)
|
|
255
|
+
* (3) shed hint clauses and re-assemble while the provider cap overflows,
|
|
256
|
+
* (4) optional post-assembly empty-prompt throw (gated by `throwOnEmpty`).
|
|
257
|
+
*
|
|
258
|
+
* TRUNCATION ORDERING (step 3): `buildImagePrompt`'s cap clamp cuts the TAIL,
|
|
259
|
+
* which is ORDER-BLIND — on a low-cap provider (seedream = 3000) a maximal
|
|
260
|
+
* direction fold renders ~3.3K characters of clauses and the cut can sever a
|
|
261
|
+
* reference directive, mention-resolved text or the user's own prose while a
|
|
262
|
+
* decorative clause survives. So the ASSEMBLER decides instead: it knows which
|
|
263
|
+
* clauses are hints because it just built them, and drops them last-folded
|
|
264
|
+
* first until the prompt fits. Everything else — references, prose, the
|
|
265
|
+
* structured fragment, the Style/Avoid suffixes — outranks a hint. A body that
|
|
266
|
+
* still overflows with ZERO hints (long prose or many directives on its own)
|
|
267
|
+
* falls back to the builder's clamp, unchanged.
|
|
268
|
+
*
|
|
269
|
+
* UNDER-CAP PARITY: the first pass folds every hint, so a prompt that fits is
|
|
270
|
+
* byte-identical to before — the retry only ever runs on an over-cap assembly.
|
|
185
271
|
*/
|
|
186
272
|
export function assembleImageInput(
|
|
187
273
|
input: AssembleImageInput,
|
|
188
274
|
): BuildImagePromptResult {
|
|
189
|
-
const
|
|
275
|
+
const pieces = renderImageHintPieces(input.subject, input.direction, input.structured)
|
|
190
276
|
|
|
191
|
-
const
|
|
192
|
-
prompt,
|
|
277
|
+
const assembleWith = (keptHintClauses: number) => buildImagePromptWithOverflow({
|
|
278
|
+
prompt: composePromptText(input.userPrompt, pieces, keptHintClauses),
|
|
193
279
|
provider: input.provider,
|
|
194
280
|
...(input.connectedReferences !== undefined
|
|
195
281
|
? { connectedReferences: input.connectedReferences }
|
|
@@ -223,8 +309,24 @@ export function assembleImageInput(
|
|
|
223
309
|
: {}),
|
|
224
310
|
})
|
|
225
311
|
|
|
312
|
+
// Fold everything first (the under-cap byte-parity pass), then shed hints
|
|
313
|
+
// from the tail of the COMBINED fold order (subject clauses first in the list,
|
|
314
|
+
// therefore last to leave) while the assembled prompt overflows the provider
|
|
315
|
+
// cap. `keepableDirectionHints` — the one shed arithmetic, shared with
|
|
316
|
+
// `composeVideoPromptText` — strictly decreases `kept` whenever there IS an
|
|
317
|
+
// overflow, so this terminates at `kept === 0` in the worst case, at which
|
|
318
|
+
// point the body overflows on its own and the builder's clamp stands.
|
|
319
|
+
let kept = pieces.hintClauses.length
|
|
320
|
+
let fitted = assembleWith(kept)
|
|
321
|
+
while (fitted.overflowChars > 0 && kept > 0) {
|
|
322
|
+
kept = keepableDirectionHints(pieces.hintClauses, kept, fitted.overflowChars)
|
|
323
|
+
fitted = assembleWith(kept)
|
|
324
|
+
}
|
|
325
|
+
// `overflowChars` is assembly bookkeeping, not part of the callers' contract.
|
|
326
|
+
const { overflowChars, ...result } = fitted
|
|
327
|
+
|
|
226
328
|
// Post-assembly empty-prompt check (opt-in): a bound entity / `@`-mention /
|
|
227
|
-
// direction chip could have filled the assembled prompt even if the user
|
|
329
|
+
// subject or direction chip could have filled the assembled prompt even if the user
|
|
228
330
|
// typed nothing — so only reject when the FINAL prompt is truly empty.
|
|
229
331
|
if (input.throwOnEmpty && !result.prompt.trim()) {
|
|
230
332
|
throw new Error(
|
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `composeVideoPromptText` — the video twin of `assemble-image-input.ts`'s
|
|
3
|
+
* `composePromptText`: fold cinematic-direction picker IDS into the prompt BODY,
|
|
4
|
+
* server-side, at the model call.
|
|
5
|
+
*
|
|
6
|
+
* WHY THIS EXISTS: `/v1/generate-video` had no structured direction channel, so
|
|
7
|
+
* every client baked the hint TEXT itself. A copied scene then carried stale
|
|
8
|
+
* catalog wording forever, a re-generate double-baked it, and each client
|
|
9
|
+
* re-implemented the fold with its own separator and its own order. The wire
|
|
10
|
+
* now carries ids; the platform renders the clauses.
|
|
11
|
+
*
|
|
12
|
+
* WHERE IT RUNS (load-bearing): on the prompt BODY, BEFORE
|
|
13
|
+
* `resolveVideoReferenceCore`. That resolver FRAMES the body — legacy prepends
|
|
14
|
+
* its `Use these characters:` block, hybrid prepends the lock lines and APPENDS
|
|
15
|
+
* the canonical role phrases and extras. Folding afterwards would push the
|
|
16
|
+
* scene/look description PAST the identity directives, a worse version of the
|
|
17
|
+
* bug this channel exists to fix. The image side is structurally identical
|
|
18
|
+
* (`assembleImageInput` = `composePromptText` → `buildImagePrompt`).
|
|
19
|
+
*
|
|
20
|
+
* That ordering is also why cap-aware shedding here takes a `frame` callback
|
|
21
|
+
* rather than a provider id: the shed must run at the FOLD site (before the
|
|
22
|
+
* resolver) but be decided on the RESOLVED length (after it), so the binding
|
|
23
|
+
* text the resolver adds is inside the budget and can never be the thing that
|
|
24
|
+
* gets dropped. See {@link VideoPromptCapOptions}. Both catalog channels —
|
|
25
|
+
* SUBJECT and direction — fold into the one sheddable list that budget walks.
|
|
26
|
+
*
|
|
27
|
+
* THE VERBOSITY POLICY LIVES HERE, NOT IN THE CLIENT: motion dimensions render
|
|
28
|
+
* their compact professional term, look dimensions their full clause
|
|
29
|
+
* (`VIDEO_HINT_MODE_DEFAULT`, resolved per row's `family` by the registry). The
|
|
30
|
+
* SUBJECT fold has its own policy — compact on video
|
|
31
|
+
* (`SUBJECT_VIDEO_HINT_MODE_DEFAULT`), because a fully specified person at full
|
|
32
|
+
* verbosity is ~30 paragraph clauses and the start frame already carries the
|
|
33
|
+
* subject's identity into the clip.
|
|
34
|
+
* It is a threaded PARAMETER with a pure default — never deployment state:
|
|
35
|
+
* `__tests__/content-free-contract.test.ts` hard-fails any environment read
|
|
36
|
+
* under `packages/prompts/src`, and this module has nothing to read anyway.
|
|
37
|
+
*
|
|
38
|
+
* EXACT NO-OP CONTRACT: with no subject, no direction and no structured fields
|
|
39
|
+
* the caller's
|
|
40
|
+
* `userPrompt` comes back VERBATIM AND UNTRIMMED — `undefined` included, since
|
|
41
|
+
* a video prompt is optional on the route. That is what keeps every existing
|
|
42
|
+
* caller byte-identical (the "backward-compatible: no connectedReferences →
|
|
43
|
+
* prompt + flat refs pass through unchanged" oracle in
|
|
44
|
+
* `backend/src/routes/__tests__/generate-video.test.ts`, restated locally in
|
|
45
|
+
* `__tests__/assemble-video-input.test.ts`).
|
|
46
|
+
*
|
|
47
|
+
* WHAT IS DELIBERATELY NOT HERE: the dimension table, the fold order, the
|
|
48
|
+
* dedupe and the surface filter all live in `direction-registry.ts` (and
|
|
49
|
+
* `subject-registry.ts` for the subject channel) — ONE renderer per channel
|
|
50
|
+
* serves both surfaces, so the image and video folds cannot drift.
|
|
51
|
+
* Clients render their "will inject into prompt" preview by importing
|
|
52
|
+
* `renderDirectionHints` + `joinPromptHints` directly.
|
|
53
|
+
*/
|
|
54
|
+
import {
|
|
55
|
+
renderDirectionHints,
|
|
56
|
+
VIDEO_HINT_MODE_DEFAULT,
|
|
57
|
+
type DirectionFields,
|
|
58
|
+
type DirectionHintMode,
|
|
59
|
+
} from "./direction-registry.js"
|
|
60
|
+
import {
|
|
61
|
+
renderSubjectHints,
|
|
62
|
+
SUBJECT_VIDEO_HINT_MODE_DEFAULT,
|
|
63
|
+
type SubjectFields,
|
|
64
|
+
type SubjectHintMode,
|
|
65
|
+
} from "./subject-registry.js"
|
|
66
|
+
import { joinPromptHints } from "./prompt-hint-join.js"
|
|
67
|
+
import { keepableDirectionHints } from "./hint-shedding.js"
|
|
68
|
+
import {
|
|
69
|
+
renderStructuredFields,
|
|
70
|
+
type StructuredPromptFields,
|
|
71
|
+
} from "./prompt-builder-structured-fields.js"
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Cap-aware shedding, opt-in. Absent → the composer is exactly what it always
|
|
75
|
+
* was (every existing caller stays byte-identical, and the no-op path below is
|
|
76
|
+
* never even reached differently).
|
|
77
|
+
*
|
|
78
|
+
* WHY A NUMBER AND A CALLBACK, NOT A PROVIDER ID — the two halves of the video
|
|
79
|
+
* surface's problem, which the image half did not have:
|
|
80
|
+
*
|
|
81
|
+
* - `cap` is the caller's EFFECTIVE ceiling, not `getMaxVideoPromptChars` read
|
|
82
|
+
* here. The routes compute it with `effectiveVideoPromptCeiling`, which
|
|
83
|
+
* mirrors `applyVideoNegativePrompt`'s reservation of the `"\nAvoid: …"`
|
|
84
|
+
* suffix for a provider with no native negative param. Re-deriving the cap
|
|
85
|
+
* inside this package would put a second copy of that reservation one
|
|
86
|
+
* refactor away from drifting from the clamp it is supposed to predict.
|
|
87
|
+
*
|
|
88
|
+
* - `frame` is the REFERENCE RESOLVER, and it is what makes the shed correct
|
|
89
|
+
* end-to-end. The fold runs BEFORE `resolveVideoReferenceCore` (see the
|
|
90
|
+
* module header — folding afterwards strands the scene description past the
|
|
91
|
+
* identity directives). The resolver then ADDS binding text: legacy's
|
|
92
|
+
* "Use these characters:" block, hybrid's lock lines and the canonical role
|
|
93
|
+
* phrases it APPENDS. That added text is exactly what an order-blind tail cut
|
|
94
|
+
* destroys first, so it must be inside the budget — but it must never be
|
|
95
|
+
* shed. Measuring THROUGH the caller's framing gives both properties at once:
|
|
96
|
+
* the shed decision sees the final length, while the only thing it can drop
|
|
97
|
+
* is a hint clause it rendered itself.
|
|
98
|
+
*
|
|
99
|
+
* Re-framing a SUBSET of the hints is sound because a hint can never change how
|
|
100
|
+
* the resolver reads the rest of the body: no registered catalog hint, term or
|
|
101
|
+
* label contains a `{image:N}` / `{ref:` / `@slug:N` shape
|
|
102
|
+
* (`__tests__/direction-hint-token-safety.test.ts` pins that for every catalog),
|
|
103
|
+
* so dropping one cannot renumber or unbind a reference.
|
|
104
|
+
*/
|
|
105
|
+
export interface VideoPromptCapOptions {
|
|
106
|
+
/**
|
|
107
|
+
* The maximum length the FRAMED prompt may reach. Sheds only while the framed
|
|
108
|
+
* body exceeds it; `undefined` (the default) disables shedding entirely.
|
|
109
|
+
*/
|
|
110
|
+
readonly cap?: number
|
|
111
|
+
/**
|
|
112
|
+
* The downstream framing the cap is measured through — the caller's reference
|
|
113
|
+
* assembly. Identity when omitted (a caller with a cap but no references).
|
|
114
|
+
* Must be PURE: it is called once per shed iteration, and the caller re-runs
|
|
115
|
+
* its own real assembly on the returned body afterwards.
|
|
116
|
+
*/
|
|
117
|
+
readonly frame?: (body: string | undefined) => string | undefined
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Fold a video run's subject and cinematic-direction ids (and optional
|
|
122
|
+
* structured fields) into its prompt body.
|
|
123
|
+
*
|
|
124
|
+
* The SUBJECT hints land first (who is in the shot — the noun phrase the
|
|
125
|
+
* cinematography modifies), then the direction hints in the registry's
|
|
126
|
+
* canonical table order (camera motion leads), and the structured fragment
|
|
127
|
+
* lands LAST — the same ordering `composePromptText` uses for stills.
|
|
128
|
+
*
|
|
129
|
+
* `subject` rides `opts` rather than a fourth positional parameter on purpose:
|
|
130
|
+
* every existing caller passes `(prompt, direction)` or
|
|
131
|
+
* `(prompt, direction, structured)` positionally, and a new positional would
|
|
132
|
+
* have made the two levers' order a memorization test.
|
|
133
|
+
*
|
|
134
|
+
* TRUNCATION ORDERING (opt-in via `opts.cap`): the provider clamp
|
|
135
|
+
* (`applyVideoNegativePrompt`) slices the prompt TAIL, which is ORDER-BLIND —
|
|
136
|
+
* on a low-cap provider (kling = 1000) a broad direction renders more than the
|
|
137
|
+
* whole ceiling and the cut severs reference bindings and the end of the user's
|
|
138
|
+
* prose while decorative clauses survive. With a cap the composer decides
|
|
139
|
+
* instead: it knows which clauses are hints because it just rendered them, and
|
|
140
|
+
* drops them LAST-FOLDED FIRST until the framed prompt fits. Everything else —
|
|
141
|
+
* the user's prose, the structured fragment (user CONTENT, never a garnish) and
|
|
142
|
+
* every byte the resolver's framing adds — outranks a hint.
|
|
143
|
+
*
|
|
144
|
+
* SUBJECT CLAUSES ARE SHED CANDIDATES TOO, and they shed AFTER the direction
|
|
145
|
+
* clauses. Both folds are catalog decoration of the same class — ids the
|
|
146
|
+
* platform rendered into wording — so exempting one would just move the
|
|
147
|
+
* overflow into the order-blind clamp, which is the bug this machinery exists
|
|
148
|
+
* to prevent. They ride the SAME `hintClauses` list the shed already walks
|
|
149
|
+
* (subject first, direction second, tail-first shedding), so there is exactly
|
|
150
|
+
* one shed arithmetic (`hint-shedding.ts`) across both channels and both
|
|
151
|
+
* surfaces. Neither channel ever sheds before the prose, the references or the
|
|
152
|
+
* structured fragment.
|
|
153
|
+
*
|
|
154
|
+
* WHAT THE BUDGET DELIBERATELY EXCLUDES: the route's later opt-in identity
|
|
155
|
+
* injection (an async DB read that appends a canonical description) and any
|
|
156
|
+
* registered `applyPromptPolicies` transform both run AFTER the reference
|
|
157
|
+
* assembly and are not modelled here. Pricing them in would mean folding an
|
|
158
|
+
* await into this pure composer; instead the provider clamp stays their last
|
|
159
|
+
* resort, exactly as today. Same for a body that still overflows with ZERO
|
|
160
|
+
* hints left — long prose, or many bound references on their own.
|
|
161
|
+
*
|
|
162
|
+
* UNDER-CAP PARITY: the first pass folds every hint, so a prompt that fits is
|
|
163
|
+
* byte-identical to a capless call, and a caller with no
|
|
164
|
+
* `subject`/`direction`/`structured` takes the same exact no-op path it always
|
|
165
|
+
* did.
|
|
166
|
+
*
|
|
167
|
+
* @param userPrompt The user's prompt. Optional: an image-to-video run may
|
|
168
|
+
* legitimately have none, and it is returned as-is when nothing folds.
|
|
169
|
+
* @param direction Flat catalog ids. Unknown keys, off-surface keys (an
|
|
170
|
+
* image-only dimension sent to a video run) and unknown ids all contribute
|
|
171
|
+
* nothing — never a throw.
|
|
172
|
+
* @param structured Path-1 structured fields. Not a `/v1/generate-video` wire
|
|
173
|
+
* field today; the canvas orchestrator passes it directly.
|
|
174
|
+
* @param opts.hintMode Override the direction verbosity policy (a whole-fold
|
|
175
|
+
* `PickerHintMode`, or a `{ look, motion }` split).
|
|
176
|
+
* @param opts.subject Flat subject ids (Person / Styling / props), same
|
|
177
|
+
* inertness contract as `direction`.
|
|
178
|
+
* @param opts.subjectHintMode Override the subject verbosity policy.
|
|
179
|
+
* @param opts.cap / `opts.frame` See {@link VideoPromptCapOptions}.
|
|
180
|
+
*/
|
|
181
|
+
export function composeVideoPromptText(
|
|
182
|
+
userPrompt: string | undefined,
|
|
183
|
+
direction: DirectionFields | undefined,
|
|
184
|
+
structured?: StructuredPromptFields,
|
|
185
|
+
opts?: {
|
|
186
|
+
readonly hintMode?: DirectionHintMode
|
|
187
|
+
readonly subject?: SubjectFields
|
|
188
|
+
readonly subjectHintMode?: SubjectHintMode
|
|
189
|
+
} & VideoPromptCapOptions,
|
|
190
|
+
): string | undefined {
|
|
191
|
+
// ONE sheddable list, subject FIRST then direction — because the shed walks it
|
|
192
|
+
// from the TAIL, so this order IS the survival order: a direction clause
|
|
193
|
+
// leaves before a subject clause. Deliberate, and the same order the image
|
|
194
|
+
// side uses (`renderImageHintPieces`): the subject is the noun phrase the
|
|
195
|
+
// cinematography modifies, so losing "who is in the shot" to keep a
|
|
196
|
+
// decorative grade would be the wrong trade. With no `subject` the list IS
|
|
197
|
+
// the direction fold, so every pre-subject caller is byte-identical.
|
|
198
|
+
const hintClauses = [
|
|
199
|
+
...renderSubjectHints(opts?.subject, {
|
|
200
|
+
surface: "video",
|
|
201
|
+
mode: opts?.subjectHintMode ?? SUBJECT_VIDEO_HINT_MODE_DEFAULT,
|
|
202
|
+
}),
|
|
203
|
+
...renderDirectionHints(direction, {
|
|
204
|
+
surface: "video",
|
|
205
|
+
mode: opts?.hintMode ?? VIDEO_HINT_MODE_DEFAULT,
|
|
206
|
+
}),
|
|
207
|
+
].filter((p) => p.length > 0)
|
|
208
|
+
// User CONTENT, not a garnish: never sheddable, always last.
|
|
209
|
+
const structuredFragment = structured ? renderStructuredFields(structured) : ""
|
|
210
|
+
|
|
211
|
+
const composeWith = (kept: number): string | undefined => {
|
|
212
|
+
const hints = [...hintClauses.slice(0, kept), structuredFragment].filter(
|
|
213
|
+
(p) => p.length > 0,
|
|
214
|
+
)
|
|
215
|
+
// Nothing to fold → the caller's value straight back, `undefined` included.
|
|
216
|
+
// Do NOT collapse this into `joinPromptHints(userPrompt ?? "", hints)`: that
|
|
217
|
+
// would turn an absent prompt into `""` and break the no-op contract above.
|
|
218
|
+
// A FULL shed lands here too, which is what keeps the no-op contract intact
|
|
219
|
+
// at `kept === 0` — the route's `composed !== prompt` guard then correctly
|
|
220
|
+
// leaves `input_data.userPrompt` unpinned.
|
|
221
|
+
if (hints.length === 0) return userPrompt
|
|
222
|
+
return joinPromptHints(userPrompt ?? "", hints)
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
const cap = opts?.cap
|
|
226
|
+
if (cap === undefined) return composeWith(hintClauses.length)
|
|
227
|
+
|
|
228
|
+
// Fold everything first (the under-cap byte-parity pass), then shed from the
|
|
229
|
+
// tail of the fold order while the FRAMED prompt overflows the ceiling.
|
|
230
|
+
// `keepableDirectionHints` — the ONE shed arithmetic, shared with the image
|
|
231
|
+
// assembler — strictly decreases `kept` whenever there is a deficit, so this
|
|
232
|
+
// terminates at `kept === 0` in the worst case, at which point nothing
|
|
233
|
+
// droppable is left and the provider clamp stands.
|
|
234
|
+
const frame = opts?.frame ?? ((body: string | undefined) => body)
|
|
235
|
+
let kept = hintClauses.length
|
|
236
|
+
let body = composeWith(kept)
|
|
237
|
+
let framedLength = frame(body)?.length ?? 0
|
|
238
|
+
while (framedLength > cap && kept > 0) {
|
|
239
|
+
kept = keepableDirectionHints(hintClauses, kept, framedLength - cap)
|
|
240
|
+
body = composeWith(kept)
|
|
241
|
+
framedLength = frame(body)?.length ?? 0
|
|
242
|
+
}
|
|
243
|
+
return body
|
|
244
|
+
}
|