@nodaro/prompts 1.11.0 → 1.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +627 -177
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +726 -33
- package/dist/index.d.ts +726 -33
- package/dist/index.js +598 -179
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/__tests__/__snapshots__/entity-convergence-image.test.ts.snap +19 -0
- package/src/__tests__/animal-getters-parity.test.ts +82 -0
- package/src/__tests__/assemble-image-input-cap.test.ts +236 -0
- package/src/__tests__/assemble-image-input.test.ts +100 -19
- package/src/__tests__/assemble-video-input-cap.test.ts +442 -0
- package/src/__tests__/assemble-video-input.test.ts +167 -33
- package/src/__tests__/direction-hint-token-safety.test.ts +21 -0
- package/src/__tests__/entity-convergence-image.test.ts +374 -0
- package/src/__tests__/location-convergence-image.test.ts +29 -1
- package/src/__tests__/location-default-role-image.test.ts +166 -0
- package/src/__tests__/mention-splice-spacing.test.ts +257 -0
- package/src/__tests__/prompt-style-section.test.ts +345 -0
- package/src/__tests__/read-node-subject.test.ts +140 -0
- package/src/__tests__/style-section-boundary.test.ts +179 -0
- package/src/__tests__/subject-fold.test.ts +251 -0
- package/src/__tests__/subject-registry.test.ts +312 -0
- package/src/assemble-image-input.ts +169 -41
- package/src/assemble-video-input.ts +200 -25
- package/src/direction-registry.ts +116 -28
- package/src/hint-shedding.ts +87 -0
- package/src/index.ts +3 -0
- package/src/parameter-prompt-hint.ts +8 -7
- package/src/picker-catalogs.ts +14 -7
- package/src/prompt-builder.ts +628 -88
- package/src/prompt-hint-join.ts +9 -0
- package/src/prompt-style-section.ts +256 -0
- package/src/read-node-direction.ts +60 -1
- package/src/subject-registry.ts +464 -0
- package/src/video-reference-resolver.ts +5 -2
|
@@ -11,20 +11,40 @@
|
|
|
11
11
|
*
|
|
12
12
|
* WHERE IT RUNS (load-bearing): on the prompt BODY, BEFORE
|
|
13
13
|
* `resolveVideoReferenceCore`. That resolver FRAMES the body — legacy prepends
|
|
14
|
-
* its `Use these characters:` block, hybrid prepends the lock lines and
|
|
15
|
-
* the canonical role phrases and extras
|
|
14
|
+
* its `Use these characters:` block, hybrid prepends the lock lines and extends
|
|
15
|
+
* the body's END with the canonical role phrases and extras (spliced in ahead of
|
|
16
|
+
* the `[style]` section, which stays last). Folding afterwards would push the
|
|
16
17
|
* scene/look description PAST the identity directives, a worse version of the
|
|
17
18
|
* bug this channel exists to fix. The image side is structurally identical
|
|
18
19
|
* (`assembleImageInput` = `composePromptText` → `buildImagePrompt`).
|
|
19
20
|
*
|
|
21
|
+
* That ordering is also why cap-aware shedding here takes a `frame` callback
|
|
22
|
+
* rather than a provider id: the shed must run at the FOLD site (before the
|
|
23
|
+
* resolver) but be decided on the RESOLVED length (after it), so the binding
|
|
24
|
+
* text the resolver adds is inside the budget and can never be the thing that
|
|
25
|
+
* gets dropped. See {@link VideoPromptCapOptions}. Both catalog channels —
|
|
26
|
+
* SUBJECT and direction — fold into the one sheddable list that budget walks.
|
|
27
|
+
*
|
|
28
|
+
* THE SHAPE IT EMITS: the body (prose, subject fold, MOTION clauses, structured
|
|
29
|
+
* fragment) and then a trailing `[style]` section carrying every LOOK clause —
|
|
30
|
+
* `prompt-style-section.ts` owns those bytes. Camera motion is shot prose, not
|
|
31
|
+
* style, so the whole motion family stays in the body; the boundary is the
|
|
32
|
+
* registry's `family` column, deliberately the same column the verbosity policy
|
|
33
|
+
* below splits on.
|
|
34
|
+
*
|
|
20
35
|
* THE VERBOSITY POLICY LIVES HERE, NOT IN THE CLIENT: motion dimensions render
|
|
21
36
|
* their compact professional term, look dimensions their full clause
|
|
22
|
-
* (`VIDEO_HINT_MODE_DEFAULT`, resolved per row's `family` by the registry).
|
|
37
|
+
* (`VIDEO_HINT_MODE_DEFAULT`, resolved per row's `family` by the registry). The
|
|
38
|
+
* SUBJECT fold has its own policy — compact on video
|
|
39
|
+
* (`SUBJECT_VIDEO_HINT_MODE_DEFAULT`), because a fully specified person at full
|
|
40
|
+
* verbosity is ~30 paragraph clauses and the start frame already carries the
|
|
41
|
+
* subject's identity into the clip.
|
|
23
42
|
* It is a threaded PARAMETER with a pure default — never deployment state:
|
|
24
43
|
* `__tests__/content-free-contract.test.ts` hard-fails any environment read
|
|
25
44
|
* under `packages/prompts/src`, and this module has nothing to read anyway.
|
|
26
45
|
*
|
|
27
|
-
* EXACT NO-OP CONTRACT: with no direction and no structured fields
|
|
46
|
+
* EXACT NO-OP CONTRACT: with no subject, no direction and no structured fields
|
|
47
|
+
* the caller's
|
|
28
48
|
* `userPrompt` comes back VERBATIM AND UNTRIMMED — `undefined` included, since
|
|
29
49
|
* a video prompt is optional on the route. That is what keeps every existing
|
|
30
50
|
* caller byte-identical (the "backward-compatible: no connectedReferences →
|
|
@@ -33,30 +53,131 @@
|
|
|
33
53
|
* `__tests__/assemble-video-input.test.ts`).
|
|
34
54
|
*
|
|
35
55
|
* WHAT IS DELIBERATELY NOT HERE: the dimension table, the fold order, the
|
|
36
|
-
* dedupe and the surface filter all live in `direction-registry.ts`
|
|
37
|
-
*
|
|
56
|
+
* dedupe and the surface filter all live in `direction-registry.ts` (and
|
|
57
|
+
* `subject-registry.ts` for the subject channel) — ONE renderer per channel
|
|
58
|
+
* serves both surfaces, so the image and video folds cannot drift.
|
|
38
59
|
* Clients render their "will inject into prompt" preview by importing
|
|
39
|
-
* `
|
|
60
|
+
* `renderSubjectHints` + `partitionStyleClauses` + `composeSectionedPrompt`
|
|
61
|
+
* directly (or `renderStyleSection` for the section alone).
|
|
40
62
|
*/
|
|
41
63
|
import {
|
|
42
|
-
renderDirectionHints,
|
|
43
64
|
VIDEO_HINT_MODE_DEFAULT,
|
|
44
65
|
type DirectionFields,
|
|
45
66
|
type DirectionHintMode,
|
|
46
67
|
} from "./direction-registry.js"
|
|
47
|
-
import {
|
|
68
|
+
import {
|
|
69
|
+
renderSubjectHints,
|
|
70
|
+
SUBJECT_VIDEO_HINT_MODE_DEFAULT,
|
|
71
|
+
type SubjectFields,
|
|
72
|
+
type SubjectHintMode,
|
|
73
|
+
} from "./subject-registry.js"
|
|
74
|
+
import {
|
|
75
|
+
asBodyClauses,
|
|
76
|
+
composeSectionedPrompt,
|
|
77
|
+
partitionStyleClauses,
|
|
78
|
+
sectionedClauseCosts,
|
|
79
|
+
} from "./prompt-style-section.js"
|
|
80
|
+
import { keepableDirectionHints } from "./hint-shedding.js"
|
|
48
81
|
import {
|
|
49
82
|
renderStructuredFields,
|
|
50
83
|
type StructuredPromptFields,
|
|
51
84
|
} from "./prompt-builder-structured-fields.js"
|
|
52
85
|
|
|
53
86
|
/**
|
|
54
|
-
*
|
|
55
|
-
*
|
|
87
|
+
* Cap-aware shedding, opt-in. Absent → the composer is exactly what it always
|
|
88
|
+
* was (every existing caller stays byte-identical, and the no-op path below is
|
|
89
|
+
* never even reached differently).
|
|
90
|
+
*
|
|
91
|
+
* WHY A NUMBER AND A CALLBACK, NOT A PROVIDER ID — the two halves of the video
|
|
92
|
+
* surface's problem, which the image half did not have:
|
|
93
|
+
*
|
|
94
|
+
* - `cap` is the caller's EFFECTIVE ceiling, not `getMaxVideoPromptChars` read
|
|
95
|
+
* here. The routes compute it with `effectiveVideoPromptCeiling`, which
|
|
96
|
+
* mirrors `applyVideoNegativePrompt`'s reservation of the `"\nAvoid: …"`
|
|
97
|
+
* suffix for a provider with no native negative param. Re-deriving the cap
|
|
98
|
+
* inside this package would put a second copy of that reservation one
|
|
99
|
+
* refactor away from drifting from the clamp it is supposed to predict.
|
|
100
|
+
*
|
|
101
|
+
* - `frame` is the REFERENCE RESOLVER, and it is what makes the shed correct
|
|
102
|
+
* end-to-end. The fold runs BEFORE `resolveVideoReferenceCore` (see the
|
|
103
|
+
* module header — folding afterwards strands the scene description past the
|
|
104
|
+
* identity directives). The resolver then ADDS binding text: legacy's
|
|
105
|
+
* "Use these characters:" block, hybrid's lock lines and the canonical role
|
|
106
|
+
* phrases that end its body. None of it is sheddable and all of it is inside
|
|
107
|
+
* what the clamp measures, so a budget blind to it under-sheds and hands the
|
|
108
|
+
* remainder to the order-blind cut. Measuring THROUGH the caller's framing
|
|
109
|
+
* gives both properties at once:
|
|
110
|
+
* the shed decision sees the final length, while the only thing it can drop
|
|
111
|
+
* is a hint clause it rendered itself.
|
|
112
|
+
*
|
|
113
|
+
* Re-framing a SUBSET of the hints is sound because a hint can never change how
|
|
114
|
+
* the resolver reads the rest of the body: no registered catalog hint, term or
|
|
115
|
+
* label contains a `{image:N}` / `{ref:` / `@slug:N` shape
|
|
116
|
+
* (`__tests__/direction-hint-token-safety.test.ts` pins that for every catalog),
|
|
117
|
+
* so dropping one cannot renumber or unbind a reference.
|
|
118
|
+
*/
|
|
119
|
+
export interface VideoPromptCapOptions {
|
|
120
|
+
/**
|
|
121
|
+
* The maximum length the FRAMED prompt may reach. Sheds only while the framed
|
|
122
|
+
* body exceeds it; `undefined` (the default) disables shedding entirely.
|
|
123
|
+
*/
|
|
124
|
+
readonly cap?: number
|
|
125
|
+
/**
|
|
126
|
+
* The downstream framing the cap is measured through — the caller's reference
|
|
127
|
+
* assembly. Identity when omitted (a caller with a cap but no references).
|
|
128
|
+
* Must be PURE: it is called once per shed iteration, and the caller re-runs
|
|
129
|
+
* its own real assembly on the returned body afterwards.
|
|
130
|
+
*/
|
|
131
|
+
readonly frame?: (body: string | undefined) => string | undefined
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Fold a video run's subject and cinematic-direction ids (and optional
|
|
136
|
+
* structured fields) into its prompt body.
|
|
137
|
+
*
|
|
138
|
+
* IN THE BODY: the SUBJECT hints first (who is in the shot — the noun phrase the
|
|
139
|
+
* cinematography modifies), then the MOTION direction hints in the registry's
|
|
140
|
+
* canonical table order (camera motion leads), then the structured fragment —
|
|
141
|
+
* the same ordering `composePromptText` uses for stills. The LOOK hints leave
|
|
142
|
+
* the body for the `[style]` section that follows it.
|
|
143
|
+
*
|
|
144
|
+
* `subject` rides `opts` rather than a fourth positional parameter on purpose:
|
|
145
|
+
* every existing caller passes `(prompt, direction)` or
|
|
146
|
+
* `(prompt, direction, structured)` positionally, and a new positional would
|
|
147
|
+
* have made the two levers' order a memorization test.
|
|
148
|
+
*
|
|
149
|
+
* TRUNCATION ORDERING (opt-in via `opts.cap`): the provider clamp
|
|
150
|
+
* (`applyVideoNegativePrompt`) slices the prompt TAIL, which is ORDER-BLIND —
|
|
151
|
+
* on a low-cap provider (kling = 1000) a broad direction renders more than the
|
|
152
|
+
* whole ceiling and the cut severs reference bindings and the end of the user's
|
|
153
|
+
* prose while decorative clauses survive. With a cap the composer decides
|
|
154
|
+
* instead: it knows which clauses are hints because it just rendered them, and
|
|
155
|
+
* drops them LAST-FOLDED FIRST until the framed prompt fits. Everything else —
|
|
156
|
+
* the user's prose, the structured fragment (user CONTENT, never a garnish) and
|
|
157
|
+
* every byte the resolver's framing adds — outranks a hint.
|
|
56
158
|
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
159
|
+
* SUBJECT CLAUSES ARE SHED CANDIDATES TOO, and they shed AFTER the direction
|
|
160
|
+
* clauses. Both folds are catalog decoration of the same class — ids the
|
|
161
|
+
* platform rendered into wording — so exempting one would just move the
|
|
162
|
+
* overflow into the order-blind clamp, which is the bug this machinery exists
|
|
163
|
+
* to prevent. They ride the SAME `hintClauses` list the shed already walks
|
|
164
|
+
* (subject first, direction second, tail-first shedding), so there is exactly
|
|
165
|
+
* one shed arithmetic (`hint-shedding.ts`) across both channels and both
|
|
166
|
+
* surfaces. Neither channel ever sheds before the prose, the references or the
|
|
167
|
+
* structured fragment.
|
|
168
|
+
*
|
|
169
|
+
* WHAT THE BUDGET DELIBERATELY EXCLUDES: the route's later opt-in identity
|
|
170
|
+
* injection (an async DB read that appends a canonical description) and any
|
|
171
|
+
* registered `applyPromptPolicies` transform both run AFTER the reference
|
|
172
|
+
* assembly and are not modelled here. Pricing them in would mean folding an
|
|
173
|
+
* await into this pure composer; instead the provider clamp stays their last
|
|
174
|
+
* resort, exactly as today. Same for a body that still overflows with ZERO
|
|
175
|
+
* hints left — long prose, or many bound references on their own.
|
|
176
|
+
*
|
|
177
|
+
* UNDER-CAP PARITY: the first pass folds every hint, so a prompt that fits is
|
|
178
|
+
* byte-identical to a capless call, and a caller with no
|
|
179
|
+
* `subject`/`direction`/`structured` takes the same exact no-op path it always
|
|
180
|
+
* did.
|
|
60
181
|
*
|
|
61
182
|
* @param userPrompt The user's prompt. Optional: an image-to-video run may
|
|
62
183
|
* legitimately have none, and it is returned as-is when nothing folds.
|
|
@@ -65,25 +186,79 @@ import {
|
|
|
65
186
|
* nothing — never a throw.
|
|
66
187
|
* @param structured Path-1 structured fields. Not a `/v1/generate-video` wire
|
|
67
188
|
* field today; the canvas orchestrator passes it directly.
|
|
68
|
-
* @param opts.hintMode Override the verbosity policy (a whole-fold
|
|
189
|
+
* @param opts.hintMode Override the direction verbosity policy (a whole-fold
|
|
69
190
|
* `PickerHintMode`, or a `{ look, motion }` split).
|
|
191
|
+
* @param opts.subject Flat subject ids (Person / Styling / props), same
|
|
192
|
+
* inertness contract as `direction`.
|
|
193
|
+
* @param opts.subjectHintMode Override the subject verbosity policy.
|
|
194
|
+
* @param opts.cap / `opts.frame` See {@link VideoPromptCapOptions}.
|
|
70
195
|
*/
|
|
71
196
|
export function composeVideoPromptText(
|
|
72
197
|
userPrompt: string | undefined,
|
|
73
198
|
direction: DirectionFields | undefined,
|
|
74
199
|
structured?: StructuredPromptFields,
|
|
75
|
-
opts?: {
|
|
200
|
+
opts?: {
|
|
201
|
+
readonly hintMode?: DirectionHintMode
|
|
202
|
+
readonly subject?: SubjectFields
|
|
203
|
+
readonly subjectHintMode?: SubjectHintMode
|
|
204
|
+
} & VideoPromptCapOptions,
|
|
76
205
|
): string | undefined {
|
|
77
|
-
|
|
78
|
-
|
|
206
|
+
// ONE sheddable list, subject FIRST then direction — because the shed walks it
|
|
207
|
+
// from the TAIL, so this order IS the survival order: a direction clause
|
|
208
|
+
// leaves before a subject clause. Deliberate, and the same order the image
|
|
209
|
+
// side uses (`renderImageHintPieces`): the subject is the noun phrase the
|
|
210
|
+
// cinematography modifies, so losing "who is in the shot" to keep a
|
|
211
|
+
// decorative grade would be the wrong trade. With no `subject` the list IS
|
|
212
|
+
// the direction fold, so every pre-subject caller sheds identically.
|
|
213
|
+
//
|
|
214
|
+
// Each clause carries the SLOT it reads in, because survival order and string
|
|
215
|
+
// order are two different things once the look clauses lift into `[style]`.
|
|
216
|
+
const hintClauses = [
|
|
217
|
+
...asBodyClauses(
|
|
218
|
+
renderSubjectHints(opts?.subject, {
|
|
219
|
+
surface: "video",
|
|
220
|
+
mode: opts?.subjectHintMode ?? SUBJECT_VIDEO_HINT_MODE_DEFAULT,
|
|
221
|
+
}),
|
|
222
|
+
),
|
|
223
|
+
...partitionStyleClauses(direction, {
|
|
79
224
|
surface: "video",
|
|
80
225
|
mode: opts?.hintMode ?? VIDEO_HINT_MODE_DEFAULT,
|
|
81
226
|
}),
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
//
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
227
|
+
].filter((c) => c.text.length > 0)
|
|
228
|
+
// User CONTENT, not a garnish: never sheddable, always last IN THE BODY (the
|
|
229
|
+
// `[style]` section reads after it).
|
|
230
|
+
const structuredFragment = structured ? renderStructuredFields(structured) : ""
|
|
231
|
+
|
|
232
|
+
// Nothing folded — no body hint AND no section — returns the caller's value
|
|
233
|
+
// straight back, `undefined` included. A FULL shed lands there too, which is
|
|
234
|
+
// what keeps the no-op contract intact at `kept === 0`: the route's
|
|
235
|
+
// `composed !== prompt` guard then correctly leaves `input_data.userPrompt`
|
|
236
|
+
// unpinned.
|
|
237
|
+
const composeWith = (kept: number): string | undefined =>
|
|
238
|
+
composeSectionedPrompt(userPrompt, hintClauses.slice(0, kept), structuredFragment)
|
|
239
|
+
|
|
240
|
+
const cap = opts?.cap
|
|
241
|
+
if (cap === undefined) return composeWith(hintClauses.length)
|
|
242
|
+
|
|
243
|
+
// Fold everything first (the under-cap byte-parity pass), then shed from the
|
|
244
|
+
// tail of the fold order while the FRAMED prompt overflows the ceiling.
|
|
245
|
+
// `keepableDirectionHints` — the ONE shed arithmetic, shared with the image
|
|
246
|
+
// assembler — strictly decreases `kept` whenever there is a deficit, so this
|
|
247
|
+
// terminates at `kept === 0` in the worst case, at which point nothing
|
|
248
|
+
// droppable is left and the provider clamp stands.
|
|
249
|
+
const frame = opts?.frame ?? ((body: string | undefined) => body)
|
|
250
|
+
let kept = hintClauses.length
|
|
251
|
+
let body = composeWith(kept)
|
|
252
|
+
let framedLength = frame(body)?.length ?? 0
|
|
253
|
+
if (framedLength <= cap) return body
|
|
254
|
+
// Priced only on the overflow path: the deltas cost a composition per clause
|
|
255
|
+
// and the fits-first-time case is the common one.
|
|
256
|
+
const costs = sectionedClauseCosts(userPrompt, hintClauses, structuredFragment)
|
|
257
|
+
const texts = hintClauses.map((c) => c.text)
|
|
258
|
+
while (framedLength > cap && kept > 0) {
|
|
259
|
+
kept = keepableDirectionHints(texts, kept, framedLength - cap, costs)
|
|
260
|
+
body = composeWith(kept)
|
|
261
|
+
framedLength = frame(body)?.length ?? 0
|
|
262
|
+
}
|
|
263
|
+
return body
|
|
89
264
|
}
|
|
@@ -32,7 +32,8 @@
|
|
|
32
32
|
* token, not a bare id. A single-id channel cannot carry it.
|
|
33
33
|
* - Subject / Styling / prop dimensions (`animal`, `heldProp`, `material`,
|
|
34
34
|
* Person, Styling) — a separate `subject` channel, deliberately out of scope
|
|
35
|
-
* here.
|
|
35
|
+
* here. It now exists: `subject-registry.ts`, same table-driven shape, its
|
|
36
|
+
* key set DISJOINT from this one (pinned by a test) so nothing folds twice.
|
|
36
37
|
*
|
|
37
38
|
* PACK BLINDNESS (parity, not a regression): `get*PromptHint` reads the frozen
|
|
38
39
|
* base arrays, so ids added by a deployment-registered catalog pack resolve to
|
|
@@ -71,8 +72,23 @@ import { getLoopSubjectPromptHint, getLoopSubjectTerm } from "./loop-subject.js"
|
|
|
71
72
|
|
|
72
73
|
/** Which generation stages fold a dimension. */
|
|
73
74
|
export type DirectionSurface = "image" | "video" | "both"
|
|
74
|
-
/**
|
|
75
|
+
/**
|
|
76
|
+
* Verbosity family — the video policy folds `motion` compact, `look` full.
|
|
77
|
+
*
|
|
78
|
+
* It is ALSO the body/section split (`prompt-style-section.ts`): `motion` stays
|
|
79
|
+
* in the prompt body as shot prose, `look` moves to the `[style]` section. The
|
|
80
|
+
* two meanings are deliberately the same column: camera motion is part of the
|
|
81
|
+
* shot, not part of the look, on both axes, and a second column would let the
|
|
82
|
+
* verbosity policy and the section boundary drift apart one row at a time.
|
|
83
|
+
*/
|
|
75
84
|
export type DirectionFamily = "look" | "motion"
|
|
85
|
+
/**
|
|
86
|
+
* Which `[style]` line a LOOK row renders on. `"film"` = the four dimensions
|
|
87
|
+
* that describe the CAPTURE (stock, grade, style, era) plus the legacy camera
|
|
88
|
+
* format key; every other look row falls to the scene line. Absent on `motion`
|
|
89
|
+
* rows, which never reach the section at all.
|
|
90
|
+
*/
|
|
91
|
+
export type DirectionStyleGroup = "film"
|
|
76
92
|
|
|
77
93
|
export interface DirectionFieldSpec {
|
|
78
94
|
/**
|
|
@@ -88,6 +104,12 @@ export interface DirectionFieldSpec {
|
|
|
88
104
|
readonly surface: DirectionSurface
|
|
89
105
|
/** Verbosity family. The video policy folds `motion` compact, `look` full. */
|
|
90
106
|
readonly family: DirectionFamily
|
|
107
|
+
/**
|
|
108
|
+
* `[style]`-section line for a `look` row. Omitted = the scene line. Meaningless
|
|
109
|
+
* on a `motion` row (those stay in the body), which is why it is optional
|
|
110
|
+
* rather than a required column with a null member.
|
|
111
|
+
*/
|
|
112
|
+
readonly styleGroup?: DirectionStyleGroup
|
|
91
113
|
/** Ids honored per dimension. Extras are SLICED at render, never a 400. */
|
|
92
114
|
readonly maxPicks: number
|
|
93
115
|
/**
|
|
@@ -158,6 +180,26 @@ const temporal = perId(getTemporalPromptHint, getTemporalTerm)
|
|
|
158
180
|
* so they are NOT aliases of `shotSize` / `lightingStyle`, and an alias table
|
|
159
181
|
* would wrongly suppress a legal second selection. Overlap is handled instead
|
|
160
182
|
* by the exact-string dedupe in `renderDirectionHints`.
|
|
183
|
+
*
|
|
184
|
+
* SECOND MEANING OF POSITION — SURVIVAL, NOT STRING POSITION: BOTH cap-aware
|
|
185
|
+
* assemblers — `assembleImageInput` (stills) and `composeVideoPromptText`
|
|
186
|
+
* (video) — shed hint clauses from the TAIL of this order when a provider's
|
|
187
|
+
* prompt cap overflows, through the one shared arithmetic in
|
|
188
|
+
* `hint-shedding.ts`. So a row's position is its survival order under the cap on
|
|
189
|
+
* EVERY surface: reordering rows for one surface silently changes what the other
|
|
190
|
+
* drops first, and the row a video-surface reorder would most likely touch
|
|
191
|
+
* (`cameraMotion`) leads the fold. What position is NOT any more is the clause's
|
|
192
|
+
* place in the assembled STRING: every `look` row is lifted out of the body into
|
|
193
|
+
* the trailing `[style]` section (`prompt-style-section.ts`), so a look clause
|
|
194
|
+
* reads after every motion clause however early it folds. That is a consequence
|
|
195
|
+
* of reusing the fold order, not a ranking — this table stays a compatibility
|
|
196
|
+
* order; anything that needs a real importance ranking should add an explicit
|
|
197
|
+
* priority column rather than reorder these rows.
|
|
198
|
+
*
|
|
199
|
+
* WHERE THIS TABLE SITS IN THE COMBINED ORDER: both assemblers fold the SUBJECT
|
|
200
|
+
* channel (`subject-registry.ts`) BEFORE this one and shed the combined list
|
|
201
|
+
* tail-first, so every direction row here is dropped before any subject clause.
|
|
202
|
+
* Deliberate — see `hint-shedding.ts` for the argument.
|
|
161
203
|
*/
|
|
162
204
|
export const DIRECTION_FIELDS = [
|
|
163
205
|
{ key: "cameraMotion", surface: "video", family: "motion", maxPicks: 1, render: perId(getCameraMotionPromptHint, getCameraMotionTerm) },
|
|
@@ -172,7 +214,7 @@ export const DIRECTION_FIELDS = [
|
|
|
172
214
|
{ key: "compositionEffect", surface: "both", family: "look", maxPicks: 1, render: perId(getCompositionEffectPromptHint, getCompositionEffectTerm) },
|
|
173
215
|
|
|
174
216
|
// Camera.
|
|
175
|
-
{ key: "cameraFormat", surface: "both", family: "look", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
|
|
217
|
+
{ key: "cameraFormat", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
|
|
176
218
|
{ key: "lens", surface: "both", family: "look", maxPicks: 1, render: perId(getLensPromptHint, getLensTerm) },
|
|
177
219
|
|
|
178
220
|
// Exposure (stills only — a video's exposure rides its own temporal levers).
|
|
@@ -186,12 +228,12 @@ export const DIRECTION_FIELDS = [
|
|
|
186
228
|
{ key: "lightingDirection", surface: "both", family: "look", maxPicks: 1, render: lighting },
|
|
187
229
|
{ key: "lightingRatio", surface: "both", family: "look", maxPicks: 1, render: lighting },
|
|
188
230
|
{ key: "colorTemperature", surface: "both", family: "look", maxPicks: 1, render: lighting },
|
|
189
|
-
{ key: "colorLook", surface: "both", family: "look", maxPicks: 1, render: perId(getColorLookPromptHint, getColorLookTerm) },
|
|
231
|
+
{ key: "colorLook", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getColorLookPromptHint, getColorLookTerm) },
|
|
190
232
|
{ key: "atmosphere", surface: "both", family: "look", maxPicks: 2, render: viaListBuilder(buildAtmosphereHints) },
|
|
191
233
|
{ key: "postProcess", surface: "image", family: "look", maxPicks: 2, render: viaListBuilder(buildPostProcessHints) },
|
|
192
234
|
|
|
193
235
|
// Style.
|
|
194
|
-
{ key: "style", surface: "both", family: "look", maxPicks: 1, render: perId(getStylePromptHint, getStyleTerm) },
|
|
236
|
+
{ key: "style", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getStylePromptHint, getStyleTerm) },
|
|
195
237
|
{ key: "mood", surface: "both", family: "look", maxPicks: 2, render: viaMood },
|
|
196
238
|
{ key: "aesthetic", surface: "both", family: "look", maxPicks: 2, render: viaStringBuilder(buildAestheticHints) },
|
|
197
239
|
{ key: "photoGenre", surface: "image", family: "look", maxPicks: 1, render: perId(getPhotoGenrePromptHint, getPhotoGenreTerm) },
|
|
@@ -200,7 +242,7 @@ export const DIRECTION_FIELDS = [
|
|
|
200
242
|
|
|
201
243
|
// Scene.
|
|
202
244
|
{ key: "setting", surface: "both", family: "look", maxPicks: 1, render: perId(getSettingPromptHint, getSettingTerm) },
|
|
203
|
-
{ key: "era", surface: "both", family: "look", maxPicks: 1, render: perId(getEraPromptHint, getEraTerm) },
|
|
245
|
+
{ key: "era", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getEraPromptHint, getEraTerm) },
|
|
204
246
|
{ key: "backdrop", surface: "both", family: "look", maxPicks: 1, render: perId(getBackdropPromptHint, getBackdropTerm) },
|
|
205
247
|
|
|
206
248
|
// Motion & time.
|
|
@@ -218,9 +260,17 @@ export const DIRECTION_FIELDS = [
|
|
|
218
260
|
{ key: "framingAngleId", surface: "both", family: "look", maxPicks: 1, render: framing },
|
|
219
261
|
{ key: "lightingId", surface: "both", family: "look", maxPicks: 1, render: lighting },
|
|
220
262
|
{ key: "lensId", surface: "both", family: "look", maxPicks: 1, render: perId(getLensPromptHint, getLensTerm) },
|
|
221
|
-
{ key: "cameraFormatId", surface: "both", family: "look", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
|
|
263
|
+
{ key: "cameraFormatId", surface: "both", family: "look", styleGroup: "film", maxPicks: 1, render: perId(getCameraFormatPromptHint, getCameraFormatTerm) },
|
|
222
264
|
] as const satisfies ReadonlyArray<DirectionFieldSpec>
|
|
223
265
|
|
|
266
|
+
/**
|
|
267
|
+
* The table read at its DECLARED type. `styleGroup` is optional, so on the
|
|
268
|
+
* `as const` tuple only the rows that carry it have the property at all — a
|
|
269
|
+
* member-wise read would not compile. Every walk over the table goes through
|
|
270
|
+
* this binding.
|
|
271
|
+
*/
|
|
272
|
+
const DIRECTION_SPECS: ReadonlyArray<DirectionFieldSpec> = DIRECTION_FIELDS
|
|
273
|
+
|
|
224
274
|
export type DirectionFieldRow = (typeof DIRECTION_FIELDS)[number]
|
|
225
275
|
export type DirectionKey = DirectionFieldRow["key"]
|
|
226
276
|
export type ImageDirectionKey = Extract<DirectionFieldRow, { surface: "image" | "both" }>["key"]
|
|
@@ -243,6 +293,16 @@ export type DirectionFields = { readonly [K in DirectionKey]?: string | readonly
|
|
|
243
293
|
*/
|
|
244
294
|
export const DIRECTION_KEYS: ReadonlyArray<DirectionKey> = DIRECTION_FIELDS.map((f) => f.key)
|
|
245
295
|
|
|
296
|
+
/**
|
|
297
|
+
* The rows that render on the `[style]` section's FILM line, in table order —
|
|
298
|
+
* derived from the table's `styleGroup` column so the grouping has exactly one
|
|
299
|
+
* definition. Exported for clients that render the section themselves; the
|
|
300
|
+
* platform's own renderer reads the column, not this list.
|
|
301
|
+
*/
|
|
302
|
+
export const FILM_STYLE_KEYS: ReadonlyArray<DirectionKey> = DIRECTION_SPECS.filter(
|
|
303
|
+
(f) => f.styleGroup === "film",
|
|
304
|
+
).map((f) => f.key as DirectionKey)
|
|
305
|
+
|
|
246
306
|
/** Verbosity for a whole fold, or split per family. */
|
|
247
307
|
export type DirectionHintMode =
|
|
248
308
|
| PickerHintMode
|
|
@@ -304,7 +364,51 @@ function normalizeDirectionIds(value: unknown, maxPicks: number): string[] {
|
|
|
304
364
|
export function directionFieldsForSurface(
|
|
305
365
|
surface: "image" | "video",
|
|
306
366
|
): ReadonlyArray<DirectionFieldSpec> {
|
|
307
|
-
return
|
|
367
|
+
return DIRECTION_SPECS.filter((f) => f.surface === "both" || f.surface === surface)
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
/** One rendered clause, still carrying the table attributes it came from. */
|
|
371
|
+
export interface DirectionHintClause {
|
|
372
|
+
readonly key: DirectionKey
|
|
373
|
+
readonly family: DirectionFamily
|
|
374
|
+
readonly styleGroup?: DirectionStyleGroup
|
|
375
|
+
readonly text: string
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
/**
|
|
379
|
+
* `renderDirectionHints` with the row each clause came from still attached —
|
|
380
|
+
* what the `[style]` section needs to decide which line a clause belongs on
|
|
381
|
+
* without a second table. Same order, same surface filter, same dedupe; the
|
|
382
|
+
* plain renderer is this one's `.text` projection, so the two cannot drift.
|
|
383
|
+
*/
|
|
384
|
+
export function renderDirectionHintClauses(
|
|
385
|
+
direction: DirectionFields | undefined,
|
|
386
|
+
opts: { surface: "image" | "video"; mode?: DirectionHintMode },
|
|
387
|
+
): DirectionHintClause[] {
|
|
388
|
+
if (!direction) return []
|
|
389
|
+
const mode = opts.mode ?? "full"
|
|
390
|
+
const out: DirectionHintClause[] = []
|
|
391
|
+
const seen = new Set<string>()
|
|
392
|
+
for (const spec of DIRECTION_SPECS) {
|
|
393
|
+
if (spec.surface !== "both" && spec.surface !== opts.surface) continue
|
|
394
|
+
const ids = normalizeDirectionIds(
|
|
395
|
+
(direction as Record<string, unknown>)[spec.key],
|
|
396
|
+
spec.maxPicks,
|
|
397
|
+
)
|
|
398
|
+
if (ids.length === 0) continue
|
|
399
|
+
for (const hint of spec.render(ids, modeForFamily(mode, spec.family))) {
|
|
400
|
+
if (hint.length > 0 && !seen.has(hint)) {
|
|
401
|
+
seen.add(hint)
|
|
402
|
+
out.push({
|
|
403
|
+
key: spec.key as DirectionKey,
|
|
404
|
+
family: spec.family,
|
|
405
|
+
...(spec.styleGroup !== undefined ? { styleGroup: spec.styleGroup } : {}),
|
|
406
|
+
text: hint,
|
|
407
|
+
})
|
|
408
|
+
}
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
return out
|
|
308
412
|
}
|
|
309
413
|
|
|
310
414
|
/**
|
|
@@ -326,29 +430,13 @@ export function directionFieldsForSurface(
|
|
|
326
430
|
* exactly as two wired picker nodes of one family behave today.
|
|
327
431
|
*
|
|
328
432
|
* Exported so a client's "will inject into prompt" preview renders the exact
|
|
329
|
-
* server
|
|
433
|
+
* clauses the server does instead of re-implementing the fold. A preview of the
|
|
434
|
+
* assembled STRING needs `prompt-style-section.ts` on top: the look clauses in
|
|
435
|
+
* this list do not read in this position any more.
|
|
330
436
|
*/
|
|
331
437
|
export function renderDirectionHints(
|
|
332
438
|
direction: DirectionFields | undefined,
|
|
333
439
|
opts: { surface: "image" | "video"; mode?: DirectionHintMode },
|
|
334
440
|
): string[] {
|
|
335
|
-
|
|
336
|
-
const mode = opts.mode ?? "full"
|
|
337
|
-
const out: string[] = []
|
|
338
|
-
const seen = new Set<string>()
|
|
339
|
-
for (const spec of DIRECTION_FIELDS) {
|
|
340
|
-
if (spec.surface !== "both" && spec.surface !== opts.surface) continue
|
|
341
|
-
const ids = normalizeDirectionIds(
|
|
342
|
-
(direction as Record<string, unknown>)[spec.key],
|
|
343
|
-
spec.maxPicks,
|
|
344
|
-
)
|
|
345
|
-
if (ids.length === 0) continue
|
|
346
|
-
for (const hint of spec.render(ids, modeForFamily(mode, spec.family))) {
|
|
347
|
-
if (hint.length > 0 && !seen.has(hint)) {
|
|
348
|
-
seen.add(hint)
|
|
349
|
-
out.push(hint)
|
|
350
|
-
}
|
|
351
|
-
}
|
|
352
|
-
}
|
|
353
|
-
return out
|
|
441
|
+
return renderDirectionHintClauses(direction, opts).map((c) => c.text)
|
|
354
442
|
}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The shed arithmetic shared by the image (`assembleImageInput`) and video
|
|
3
|
+
* (`composeVideoPromptText`) cap-aware assemblers, so the two surfaces cannot
|
|
4
|
+
* drift in WHICH clause goes first when a provider's prompt cap overflows.
|
|
5
|
+
*
|
|
6
|
+
* Only the arithmetic lives here. Each surface keeps its own loop, because what
|
|
7
|
+
* they MEASURE differs: the image side reads `buildImagePrompt`'s
|
|
8
|
+
* `overflowChars` (the cap clamp reports how much it cut), while the video side
|
|
9
|
+
* measures the resolver-FRAMED body against the route's effective ceiling. Both
|
|
10
|
+
* hand this function the same question — "how many of the first `kept` clauses
|
|
11
|
+
* may stay if `deficit` characters have to leave the body?" — and both re-assemble
|
|
12
|
+
* and re-check afterwards.
|
|
13
|
+
*
|
|
14
|
+
* WHAT COUNTS AS A SHEDDABLE CLAUSE (both surfaces, one answer): every clause
|
|
15
|
+
* the platform RENDERED from catalog ids — the SUBJECT fold and the cinematic
|
|
16
|
+
* DIRECTION fold alike. They are decoration of the same class, so exempting
|
|
17
|
+
* either would not save it: the overflow would simply land in the provider's
|
|
18
|
+
* order-blind tail clamp, severing reference bindings or the end of the user's
|
|
19
|
+
* prose instead — precisely the bug this machinery exists to prevent. Never
|
|
20
|
+
* sheddable: the user's prose, the bound references and the framing text the
|
|
21
|
+
* reference resolver adds, and the structured fragment (user CONTENT).
|
|
22
|
+
*
|
|
23
|
+
* THE LIST IS A SURVIVAL ORDER, NOT A STRING ORDER. It was both until the
|
|
24
|
+
* `[style]` section landed; now a look clause is lifted out of the body and
|
|
25
|
+
* reads after every motion clause however early it folds
|
|
26
|
+
* (`prompt-style-section.ts`). Position here still answers exactly one question
|
|
27
|
+
* — who leaves first — and `clauseCosts` is how the caller tells this function
|
|
28
|
+
* what a clause actually costs in a shape it can no longer infer from the
|
|
29
|
+
* clause text alone.
|
|
30
|
+
*/
|
|
31
|
+
import { PROMPT_HINT_SEPARATOR } from "./prompt-hint-join.js"
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* How many of the first `kept` hint clauses may STAY if `deficit`
|
|
35
|
+
* characters have to leave the body. Walks the fold order from the TAIL,
|
|
36
|
+
* subtracting each clause plus the separator it brought, and stops as soon as
|
|
37
|
+
* enough has been reclaimed.
|
|
38
|
+
*
|
|
39
|
+
* The name is historical (direction was the first and for a while the only
|
|
40
|
+
* channel); the list both callers pass is now the COMBINED fold —
|
|
41
|
+
* `[...subject, ...direction]` on both surfaces — so the shed order is that
|
|
42
|
+
* combined order REVERSED: the direction block empties first, then the subject
|
|
43
|
+
* block. Deliberate, and the reason the two folds share one list: a fully
|
|
44
|
+
* specified person renders ~30 clauses, so a subject fold left unsheddable
|
|
45
|
+
* would be the single biggest way to push an overflow into the order-blind
|
|
46
|
+
* clamp, while a decorative grade or ISO value survives.
|
|
47
|
+
*
|
|
48
|
+
* Within the direction block the order is `DIRECTION_FIELDS` order REVERSED
|
|
49
|
+
* (and within the subject block, `SUBJECT_FIELDS` reversed). Note what that
|
|
50
|
+
* is and is not: each table's order is a COMPATIBILITY order (grouped by family,
|
|
51
|
+
* with the legacy `DirectionFields` block pinned last so every pre-registry
|
|
52
|
+
* caller's fold stays byte-identical) — it is NOT a ranking of how load-bearing
|
|
53
|
+
* a dimension is, and this function does not claim one. Tail-first is chosen
|
|
54
|
+
* because it is deterministic, matches the fold order the API documents, and
|
|
55
|
+
* needs no second ordering to drift out of sync with the table. A caller mixing
|
|
56
|
+
* legacy keys with the newer ones can therefore lose e.g. `lightingId` before a
|
|
57
|
+
* decorative `isoValue` clause; if that ever matters, the fix is an explicit
|
|
58
|
+
* priority column on `DIRECTION_FIELDS`, not a second hand-kept list here.
|
|
59
|
+
*
|
|
60
|
+
* `clauseCosts[i]` is what clause `i` really adds to the assembled prompt.
|
|
61
|
+
* Without it each clause is charged its text plus one separator, which is what
|
|
62
|
+
* a clause folded inline costs — but a clause that lands in the `[style]`
|
|
63
|
+
* section carries section bytes too (the first one carries the whole header),
|
|
64
|
+
* and under-charging it makes this walk cover the deficit with MORE clauses
|
|
65
|
+
* than it needs. Both in-package callers pass exact composed-length deltas
|
|
66
|
+
* (`sectionedClauseCosts`); the default keeps the pre-section arithmetic for
|
|
67
|
+
* anyone else.
|
|
68
|
+
*
|
|
69
|
+
* Still deliberately approximate (assembly is not perfectly additive — a
|
|
70
|
+
* downstream frame can grow or shrink around the body); the caller re-assembles
|
|
71
|
+
* and re-checks, and this function strictly decreases `kept` whenever
|
|
72
|
+
* `deficit > 0`, so that loop terminates however the costs are priced.
|
|
73
|
+
*/
|
|
74
|
+
export function keepableDirectionHints(
|
|
75
|
+
hintClauses: readonly string[],
|
|
76
|
+
kept: number,
|
|
77
|
+
deficit: number,
|
|
78
|
+
clauseCosts?: readonly number[],
|
|
79
|
+
): number {
|
|
80
|
+
let remaining = deficit
|
|
81
|
+
let next = kept
|
|
82
|
+
while (next > 0 && remaining > 0) {
|
|
83
|
+
next -= 1
|
|
84
|
+
remaining -= clauseCosts?.[next] ?? hintClauses[next]!.length + PROMPT_HINT_SEPARATOR.length
|
|
85
|
+
}
|
|
86
|
+
return next
|
|
87
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -19,7 +19,10 @@ export * from "./brand-tokens.js"
|
|
|
19
19
|
export * from "./prompt-builder.js"
|
|
20
20
|
export * from "./prompt-builder-structured-fields.js"
|
|
21
21
|
export * from "./direction-registry.js"
|
|
22
|
+
export * from "./subject-registry.js"
|
|
22
23
|
export * from "./prompt-hint-join.js"
|
|
24
|
+
export * from "./prompt-style-section.js"
|
|
25
|
+
export * from "./hint-shedding.js"
|
|
23
26
|
export * from "./video-reference-resolver.js"
|
|
24
27
|
export * from "./sound-aggregator.js"
|
|
25
28
|
export * from "./assemble-suno-input.js"
|
|
@@ -38,7 +38,7 @@ import { composeCameraMotionHintFromConnections } from "./camera-motions.js"
|
|
|
38
38
|
import { composeTransitionHintFromConnections, type TransitionDuration, type TransitionIntensity, type TransitionPosition, type TransitionTiming } from "./transitions.js"
|
|
39
39
|
import { composeCharacterFxHintFromConnections, type CharacterFxDuration, type CharacterFxIntensity, type CharacterFxPosition, type CharacterFxTiming } from "./character-fx.js"
|
|
40
40
|
import { buildMaterialHints } from "./materials.js"
|
|
41
|
-
import {
|
|
41
|
+
import { getAnimalPromptHint, getAnimalTerm } from "@nodaro/shared"
|
|
42
42
|
import { getVehicle } from "@nodaro/shared"
|
|
43
43
|
import { getWeapon } from "@nodaro/shared"
|
|
44
44
|
import { getFurniture } from "@nodaro/shared"
|
|
@@ -322,15 +322,16 @@ function resolveBaseHint(
|
|
|
322
322
|
return withCustomText(data, byMode(mode, getLoopSubjectPromptHint, getLoopSubjectTerm)(asStr(data.loopSubject)))
|
|
323
323
|
case "material":
|
|
324
324
|
return withCustomText(data, buildMaterialHints(data.material, mode))
|
|
325
|
-
|
|
326
|
-
|
|
325
|
+
// Animal is the one Object-entity catalog whose phrasing has a single
|
|
326
|
+
// owner: `@nodaro/shared`'s `getAnimalPromptHint` / `getAnimalTerm`, which
|
|
327
|
+
// the picker-catalog funnel calls too. Both getters already return "" on a
|
|
328
|
+
// miss, so the entry lookup and the `animal ? … : ""` guard are the
|
|
329
|
+
// getters' job now, not this switch's.
|
|
330
|
+
case "animal":
|
|
327
331
|
return withCustomText(
|
|
328
332
|
data,
|
|
329
|
-
animal
|
|
330
|
-
? byMode(mode, `featuring a ${animal.label.toLowerCase()}, ${animal.description}`, objectEntityTerm(animal))
|
|
331
|
-
: "",
|
|
333
|
+
byMode(mode, getAnimalPromptHint, getAnimalTerm)(asStr(data.animal)),
|
|
332
334
|
)
|
|
333
|
-
}
|
|
334
335
|
case "vehicle": {
|
|
335
336
|
const vehicle = getVehicle(asStr(data.vehicle))
|
|
336
337
|
return withCustomText(
|