@nodaro/prompts 1.10.0 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +318 -12
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +594 -25
- package/dist/index.d.ts +594 -25
- package/dist/index.js +306 -14
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/__tests__/assemble-image-input.test.ts +93 -3
- package/src/__tests__/assemble-video-input.test.ts +301 -0
- package/src/__tests__/direction-hint-token-safety.test.ts +113 -0
- package/src/__tests__/direction-registry.test.ts +393 -0
- package/src/__tests__/image-convergence-image.test.ts +370 -0
- package/src/__tests__/read-node-direction.test.ts +154 -0
- package/src/assemble-image-input.ts +45 -46
- package/src/assemble-video-input.ts +89 -0
- package/src/direction-registry.ts +354 -0
- package/src/index.ts +8 -2
- package/src/prompt-builder.ts +188 -1
- package/src/prompt-hint-join.ts +30 -0
- package/src/read-node-direction.ts +174 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nodaro/prompts",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.11.0",
|
|
4
4
|
"description": "Nodaro's prompt-engineering layer — person/picker catalogs with prompt hints, identity-lock clauses, entity prompt builders, brand presets, and prompt/reference assembly shared by the Nodaro platform and SDK.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "FSL-1.1-Apache-2.0",
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
"test": "vitest run"
|
|
21
21
|
},
|
|
22
22
|
"dependencies": {
|
|
23
|
-
"@nodaro/shared": "^2.
|
|
23
|
+
"@nodaro/shared": "^2.16.0"
|
|
24
24
|
},
|
|
25
25
|
"devDependencies": {
|
|
26
26
|
"tsup": "^8.5.0",
|
|
@@ -3,6 +3,10 @@ import { assembleImageInput } from "../assemble-image-input.js"
|
|
|
3
3
|
import { buildImagePrompt } from "../prompt-builder.js"
|
|
4
4
|
import { getFramingPromptHint } from "../framing.js"
|
|
5
5
|
import { getLightingPromptHint } from "../lighting.js"
|
|
6
|
+
import { getLensPromptHint } from "../lens.js"
|
|
7
|
+
import { getCameraFormatPromptHint } from "../camera-format.js"
|
|
8
|
+
import { getStylePromptHint } from "../style.js"
|
|
9
|
+
import { buildMoodHints } from "../mood.js"
|
|
6
10
|
import type { ConnectedReference } from "@nodaro/shared"
|
|
7
11
|
|
|
8
12
|
/**
|
|
@@ -11,9 +15,9 @@ import type { ConnectedReference } from "@nodaro/shared"
|
|
|
11
15
|
* `generate-image` assembly through it. These tests pin BOTH layers:
|
|
12
16
|
* (a) the id-based composition (direction / structured) — ported from
|
|
13
17
|
* Studio's `assembly.test.ts` as the oracle, and
|
|
14
|
-
* (b) the BY-CONSTRUCTION PARITY the caller refactor relies on:
|
|
15
|
-
* direction/structured, the wrapper === the old inline
|
|
16
|
-
* call + empty-check, byte-for-byte.
|
|
18
|
+
* (b) the BY-CONSTRUCTION PARITY the caller refactor relies on: for a node
|
|
19
|
+
* that carries no direction/structured, the wrapper === the old inline
|
|
20
|
+
* `buildImagePrompt` call + empty-check, byte-for-byte.
|
|
17
21
|
*/
|
|
18
22
|
|
|
19
23
|
// flux-2-max supports reference images (used to assert refs survive the gate).
|
|
@@ -93,6 +97,92 @@ describe("assembleImageInput — id-based composition (Studio oracle)", () => {
|
|
|
93
97
|
})
|
|
94
98
|
expect(result.prompt).toBe("a portrait. Subject: 30 years old, woman, calm expression.")
|
|
95
99
|
})
|
|
100
|
+
|
|
101
|
+
// ── The direction registry (the fold moved into `direction-registry.ts`) ──
|
|
102
|
+
|
|
103
|
+
it("returns the prompt VERBATIM and UNTRIMMED for a direction that renders nothing", () => {
|
|
104
|
+
// The no-op branch is what the platform-caller parity contract rests on: an
|
|
105
|
+
// empty (or all-empty-valued) `direction` must not trip the join, or the
|
|
106
|
+
// prompt would silently get trimmed.
|
|
107
|
+
for (const direction of [{}, { style: "" }, { mood: [] }, { style: "__no_such_style__" }]) {
|
|
108
|
+
const result = assembleImageInput({
|
|
109
|
+
userPrompt: " a knight \n",
|
|
110
|
+
provider: REF_PROVIDER,
|
|
111
|
+
direction,
|
|
112
|
+
})
|
|
113
|
+
expect(result.prompt).toBe(" a knight \n")
|
|
114
|
+
}
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
it("folds a registry key that predates no legacy field (style) end to end", () => {
|
|
118
|
+
const result = assembleImageInput({
|
|
119
|
+
userPrompt: "a knight",
|
|
120
|
+
provider: REF_PROVIDER,
|
|
121
|
+
direction: { style: "anime" },
|
|
122
|
+
})
|
|
123
|
+
expect(result.prompt).toBe(`a knight. ${getStylePromptHint("anime")}`)
|
|
124
|
+
})
|
|
125
|
+
|
|
126
|
+
it("blends a multi-pick dimension into ONE clause", () => {
|
|
127
|
+
const blended = buildMoodHints({ mood: ["happy", "joyful"] }, "full")
|
|
128
|
+
expect(blended).toHaveLength(1)
|
|
129
|
+
const result = assembleImageInput({
|
|
130
|
+
userPrompt: "a knight",
|
|
131
|
+
provider: REF_PROVIDER,
|
|
132
|
+
direction: { mood: ["happy", "joyful"] },
|
|
133
|
+
})
|
|
134
|
+
expect(result.prompt).toBe(`a knight. ${blended[0]}`)
|
|
135
|
+
})
|
|
136
|
+
|
|
137
|
+
it("folds in TABLE order, not the caller's object-literal order", () => {
|
|
138
|
+
const result = assembleImageInput({
|
|
139
|
+
userPrompt: "a knight",
|
|
140
|
+
provider: REF_PROVIDER,
|
|
141
|
+
direction: { style: "anime", shotSize: "wide-shot" },
|
|
142
|
+
})
|
|
143
|
+
expect(result.prompt).toBe(
|
|
144
|
+
`a knight. ${getFramingPromptHint("wide-shot")}. ${getStylePromptHint("anime")}`,
|
|
145
|
+
)
|
|
146
|
+
})
|
|
147
|
+
|
|
148
|
+
it("keeps the five pre-registry keys byte-identical to the old inlined fold", () => {
|
|
149
|
+
const direction = {
|
|
150
|
+
framingId: "wide-shot",
|
|
151
|
+
framingAngleId: "low-angle",
|
|
152
|
+
lightingId: "golden-hour",
|
|
153
|
+
lensId: "wide-24mm",
|
|
154
|
+
cameraFormatId: "16mm-film",
|
|
155
|
+
}
|
|
156
|
+
const result = assembleImageInput({
|
|
157
|
+
userPrompt: "a knight",
|
|
158
|
+
provider: REF_PROVIDER,
|
|
159
|
+
direction,
|
|
160
|
+
})
|
|
161
|
+
// The exact string the pre-registry `composePromptText` produced: the same
|
|
162
|
+
// five clauses, in the same order, joined with the same ". ".
|
|
163
|
+
expect(result.prompt).toBe(
|
|
164
|
+
[
|
|
165
|
+
"a knight",
|
|
166
|
+
getFramingPromptHint("wide-shot"),
|
|
167
|
+
getFramingPromptHint("low-angle"),
|
|
168
|
+
getLightingPromptHint("golden-hour"),
|
|
169
|
+
getLensPromptHint("wide-24mm"),
|
|
170
|
+
getCameraFormatPromptHint("16mm-film"),
|
|
171
|
+
].join(". "),
|
|
172
|
+
)
|
|
173
|
+
})
|
|
174
|
+
|
|
175
|
+
it("keeps the structured fragment LAST, after every direction clause", () => {
|
|
176
|
+
const result = assembleImageInput({
|
|
177
|
+
userPrompt: "a portrait",
|
|
178
|
+
provider: REF_PROVIDER,
|
|
179
|
+
direction: { style: "anime" },
|
|
180
|
+
structured: { person: { age: 30, gender: "woman", expression: "calm" } },
|
|
181
|
+
})
|
|
182
|
+
expect(result.prompt).toBe(
|
|
183
|
+
`a portrait. ${getStylePromptHint("anime")}. Subject: 30 years old, woman, calm expression.`,
|
|
184
|
+
)
|
|
185
|
+
})
|
|
96
186
|
})
|
|
97
187
|
|
|
98
188
|
describe("assembleImageInput — empty-prompt throw (opt-in)", () => {
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
import { describe, it, expect } from "vitest"
|
|
2
|
+
import { composeVideoPromptText } from "../assemble-video-input.js"
|
|
3
|
+
import { directionFieldsForSurface } from "../direction-registry.js"
|
|
4
|
+
import { getStylePromptHint, getStyleTerm } from "../style.js"
|
|
5
|
+
import { getTransitionPromptHint, getTransitionTerm } from "../transitions.js"
|
|
6
|
+
import { getCameraMotionPromptHint, getCameraMotionTerm } from "../camera-motions.js"
|
|
7
|
+
import { getFramingPromptHint } from "../framing.js"
|
|
8
|
+
import { getLightingPromptHint } from "../lighting.js"
|
|
9
|
+
import { buildMoodHints } from "../mood.js"
|
|
10
|
+
import { buildAestheticHints } from "../aesthetic.js"
|
|
11
|
+
import { buildAtmosphereHints } from "../atmosphere.js"
|
|
12
|
+
import { buildPhotographerHints } from "../photographer.js"
|
|
13
|
+
import { renderStructuredFields } from "../prompt-builder-structured-fields.js"
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* `composeVideoPromptText` is the video route's ONLY prompt-composition step,
|
|
17
|
+
* so two contracts matter here above everything else:
|
|
18
|
+
*
|
|
19
|
+
* 1. THE NO-OP CONTRACT — with no direction the caller's prompt comes back
|
|
20
|
+
* verbatim and untrimmed, `undefined` included. This is the local
|
|
21
|
+
* restatement of the route-level byte-parity oracle ("backward-compatible:
|
|
22
|
+
* no connectedReferences → prompt + flat refs pass through unchanged" in
|
|
23
|
+
* `backend/src/routes/__tests__/generate-video.test.ts`), and it is what
|
|
24
|
+
* makes this whole leg land dark.
|
|
25
|
+
* 2. THE VERBOSITY POLICY — look dimensions render their full clause, motion
|
|
26
|
+
* dimensions their compact professional term. That split moved from the
|
|
27
|
+
* client to the platform, so it is pinned in both directions.
|
|
28
|
+
*
|
|
29
|
+
* Real catalog ids throughout: every `get*PromptHint` returns `""` on a miss,
|
|
30
|
+
* so a made-up id would make most assertions vacuously pass.
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
// ── Real ids, one per dimension used below ──────────────────────────────────
|
|
34
|
+
const STYLE = "cinematic" // look
|
|
35
|
+
const TRANSITION = "cross-dissolve" // motion
|
|
36
|
+
const CAMERA_MOTION = "handheld" // motion
|
|
37
|
+
const SHOT_SIZE = "wide-shot" // look, framing catalog
|
|
38
|
+
const TIME_OF_DAY = "dawn" // look, lighting catalog (time-of-day category)
|
|
39
|
+
const LIGHTING_STYLE = "three-point" // look, lighting catalog (style category)
|
|
40
|
+
const PHOTOGRAPHER = "tim-walker" // IMAGE-ONLY dimension
|
|
41
|
+
const NO_SUCH_ID = "__no_such_id__"
|
|
42
|
+
|
|
43
|
+
describe("composeVideoPromptText — the no-op contract", () => {
|
|
44
|
+
it("returns a prompt verbatim when no direction is passed", () => {
|
|
45
|
+
expect(composeVideoPromptText("a knight rides at dusk", undefined)).toBe(
|
|
46
|
+
"a knight rides at dusk",
|
|
47
|
+
)
|
|
48
|
+
})
|
|
49
|
+
|
|
50
|
+
it("returns a whitespace-only prompt verbatim and UNTRIMMED", () => {
|
|
51
|
+
expect(composeVideoPromptText(" \n", undefined)).toBe(" \n")
|
|
52
|
+
})
|
|
53
|
+
|
|
54
|
+
it("preserves `undefined` (the video prompt is optional)", () => {
|
|
55
|
+
expect(composeVideoPromptText(undefined, undefined)).toBeUndefined()
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
it("treats an empty direction object as no direction", () => {
|
|
59
|
+
expect(composeVideoPromptText("a knight", {})).toBe("a knight")
|
|
60
|
+
expect(composeVideoPromptText(undefined, {})).toBeUndefined()
|
|
61
|
+
})
|
|
62
|
+
})
|
|
63
|
+
|
|
64
|
+
describe("composeVideoPromptText — the verbosity policy", () => {
|
|
65
|
+
it("renders a LOOK dimension as its full clause", () => {
|
|
66
|
+
expect(composeVideoPromptText("a knight", { style: STYLE })).toBe(
|
|
67
|
+
`a knight. ${getStylePromptHint(STYLE)}`,
|
|
68
|
+
)
|
|
69
|
+
})
|
|
70
|
+
|
|
71
|
+
it("renders a MOTION dimension as its compact term, not its full hint", () => {
|
|
72
|
+
const out = composeVideoPromptText("a knight", { transition: TRANSITION })
|
|
73
|
+
expect(out).toBe(`a knight. ${getTransitionTerm(TRANSITION)}`)
|
|
74
|
+
expect(out).not.toContain(getTransitionPromptHint(TRANSITION))
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
it("applies both halves of the split policy in ONE fold", () => {
|
|
78
|
+
const out = composeVideoPromptText("a knight", {
|
|
79
|
+
style: STYLE,
|
|
80
|
+
transition: TRANSITION,
|
|
81
|
+
})
|
|
82
|
+
expect(out).toContain(getStylePromptHint(STYLE))
|
|
83
|
+
expect(out).toContain(getTransitionTerm(TRANSITION))
|
|
84
|
+
expect(out).not.toContain(getTransitionPromptHint(TRANSITION))
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
it("honors a whole-fold `hintMode` override in both directions", () => {
|
|
88
|
+
// "full" promotes the motion family to its full clause…
|
|
89
|
+
expect(
|
|
90
|
+
composeVideoPromptText("a knight", { transition: TRANSITION }, undefined, {
|
|
91
|
+
hintMode: "full",
|
|
92
|
+
}),
|
|
93
|
+
).toBe(`a knight. ${getTransitionPromptHint(TRANSITION)}`)
|
|
94
|
+
// …and "compact" demotes the look family to its term.
|
|
95
|
+
expect(
|
|
96
|
+
composeVideoPromptText("a knight", { style: STYLE }, undefined, {
|
|
97
|
+
hintMode: "compact",
|
|
98
|
+
}),
|
|
99
|
+
).toBe(`a knight. ${getStyleTerm(STYLE)}`)
|
|
100
|
+
})
|
|
101
|
+
})
|
|
102
|
+
|
|
103
|
+
describe("composeVideoPromptText — ordering", () => {
|
|
104
|
+
it("puts camera motion first (the order Studio and the orchestrator both emit)", () => {
|
|
105
|
+
const out = composeVideoPromptText("a knight", {
|
|
106
|
+
style: STYLE,
|
|
107
|
+
cameraMotion: CAMERA_MOTION,
|
|
108
|
+
})!
|
|
109
|
+
expect(out.indexOf(getCameraMotionTerm(CAMERA_MOTION))).toBeLessThan(
|
|
110
|
+
out.indexOf(getStylePromptHint(STYLE)),
|
|
111
|
+
)
|
|
112
|
+
// Compact motion again — the camera-motion row is `family: "motion"`.
|
|
113
|
+
expect(out).not.toContain(getCameraMotionPromptHint(CAMERA_MOTION))
|
|
114
|
+
})
|
|
115
|
+
|
|
116
|
+
it("folds in TABLE order, not the caller's object order", () => {
|
|
117
|
+
// `shotSize` (row 2) precedes `style` (row 22) however the object is written.
|
|
118
|
+
const out = composeVideoPromptText("a knight", {
|
|
119
|
+
style: STYLE,
|
|
120
|
+
shotSize: SHOT_SIZE,
|
|
121
|
+
})!
|
|
122
|
+
expect(out.indexOf(getFramingPromptHint(SHOT_SIZE))).toBeLessThan(
|
|
123
|
+
out.indexOf(getStylePromptHint(STYLE)),
|
|
124
|
+
)
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
it("appends the structured fragment AFTER every direction hint", () => {
|
|
128
|
+
const structured = { mood: "wistful" }
|
|
129
|
+
const fragment = renderStructuredFields(structured)
|
|
130
|
+
expect(fragment.length).toBeGreaterThan(0)
|
|
131
|
+
const out = composeVideoPromptText("a knight", { style: STYLE }, structured)!
|
|
132
|
+
expect(out).toBe(`a knight. ${getStylePromptHint(STYLE)}. ${fragment}`)
|
|
133
|
+
})
|
|
134
|
+
})
|
|
135
|
+
|
|
136
|
+
describe("composeVideoPromptText — multi-pick doctrine", () => {
|
|
137
|
+
it("BLENDS two moods into ONE clause (not a per-id loop)", () => {
|
|
138
|
+
const blended = buildMoodHints({ mood: ["happy", "serene"] }, "full")
|
|
139
|
+
expect(blended).toHaveLength(1)
|
|
140
|
+
expect(composeVideoPromptText("a knight", { mood: ["happy", "serene"] })).toBe(
|
|
141
|
+
`a knight. ${blended[0]}`,
|
|
142
|
+
)
|
|
143
|
+
})
|
|
144
|
+
|
|
145
|
+
it("BLENDS two aesthetics into ONE clause", () => {
|
|
146
|
+
const blended = buildAestheticHints(["y2k", "cottagecore"], "full")
|
|
147
|
+
expect(blended.length).toBeGreaterThan(0)
|
|
148
|
+
expect(
|
|
149
|
+
composeVideoPromptText("a knight", { aesthetic: ["y2k", "cottagecore"] }),
|
|
150
|
+
).toBe(`a knight. ${blended}`)
|
|
151
|
+
})
|
|
152
|
+
|
|
153
|
+
it("slices an over-cap array to the dimension's maxPicks (atmosphere = 2)", () => {
|
|
154
|
+
const out = composeVideoPromptText("a knight", {
|
|
155
|
+
atmosphere: ["clear", "cloudy", "overcast"],
|
|
156
|
+
})!
|
|
157
|
+
const kept = buildAtmosphereHints(["clear", "cloudy"], "full")
|
|
158
|
+
expect(kept).toHaveLength(2)
|
|
159
|
+
expect(out).toBe(`a knight. ${kept.join(". ")}`)
|
|
160
|
+
expect(out).not.toContain(buildAtmosphereHints("overcast", "full")[0])
|
|
161
|
+
})
|
|
162
|
+
|
|
163
|
+
it("accepts an ARRAY on a single-pick key and keeps the first id", () => {
|
|
164
|
+
// The legacy `V2LookPicker` shape: a single-pick dimension that stored an
|
|
165
|
+
// array. Must degrade to one hint, never throw and never drop the key.
|
|
166
|
+
const out = composeVideoPromptText("a knight", { style: [STYLE, "anime"] })
|
|
167
|
+
expect(out).toBe(`a knight. ${getStylePromptHint(STYLE)}`)
|
|
168
|
+
})
|
|
169
|
+
})
|
|
170
|
+
|
|
171
|
+
describe("composeVideoPromptText — tolerance", () => {
|
|
172
|
+
it("skips an unknown id and leaves the prompt verbatim (no dangling '. ')", () => {
|
|
173
|
+
expect(composeVideoPromptText("a knight", { style: NO_SUCH_ID })).toBe("a knight")
|
|
174
|
+
})
|
|
175
|
+
|
|
176
|
+
it("skips an IMAGE-ONLY dimension sent to a video run", () => {
|
|
177
|
+
// `photographer` is accepted on the wire (surface is a render concern, not
|
|
178
|
+
// a wire concern) and simply contributes nothing here.
|
|
179
|
+
expect(buildPhotographerHints(PHOTOGRAPHER, "full").length).toBeGreaterThan(0)
|
|
180
|
+
expect(composeVideoPromptText("a knight", { photographer: PHOTOGRAPHER })).toBe(
|
|
181
|
+
"a knight",
|
|
182
|
+
)
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
it("skips an unknown wire key entirely", () => {
|
|
186
|
+
expect(
|
|
187
|
+
composeVideoPromptText("a knight", { __not_a_dimension__: "x" } as never),
|
|
188
|
+
).toBe("a knight")
|
|
189
|
+
})
|
|
190
|
+
})
|
|
191
|
+
|
|
192
|
+
describe("composeVideoPromptText — an empty or absent body", () => {
|
|
193
|
+
it("returns the hints alone for an empty prompt (never a leading '. ')", () => {
|
|
194
|
+
expect(composeVideoPromptText("", { style: STYLE })).toBe(getStylePromptHint(STYLE))
|
|
195
|
+
})
|
|
196
|
+
|
|
197
|
+
it("returns the hints alone for an ABSENT prompt", () => {
|
|
198
|
+
expect(composeVideoPromptText(undefined, { style: STYLE })).toBe(
|
|
199
|
+
getStylePromptHint(STYLE),
|
|
200
|
+
)
|
|
201
|
+
})
|
|
202
|
+
})
|
|
203
|
+
|
|
204
|
+
describe("composeVideoPromptText — the dedupe invariant", () => {
|
|
205
|
+
// The five legacy keys address a WHOLE catalog, so they are not aliases of
|
|
206
|
+
// their canonical counterparts. Overlap is resolved by exact-clause dedupe,
|
|
207
|
+
// which suppresses a repeated clause without suppressing a different id.
|
|
208
|
+
it("emits ONE clause when a legacy and a canonical key carry the SAME id", () => {
|
|
209
|
+
expect(
|
|
210
|
+
composeVideoPromptText("a knight", { framingId: SHOT_SIZE, shotSize: SHOT_SIZE }),
|
|
211
|
+
).toBe(`a knight. ${getFramingPromptHint(SHOT_SIZE)}`)
|
|
212
|
+
})
|
|
213
|
+
|
|
214
|
+
it("emits BOTH clauses for two DIFFERENT ids of one catalog", () => {
|
|
215
|
+
// The case an alias table would have wrongly collapsed: `lightingId` is
|
|
216
|
+
// whole-catalog, so a time-of-day pick beside a lighting-style pick is two
|
|
217
|
+
// legitimate selections.
|
|
218
|
+
const out = composeVideoPromptText("a knight", {
|
|
219
|
+
lightingStyle: LIGHTING_STYLE,
|
|
220
|
+
lightingId: TIME_OF_DAY,
|
|
221
|
+
})
|
|
222
|
+
expect(out).toBe(
|
|
223
|
+
`a knight. ${getLightingPromptHint(LIGHTING_STYLE)}. ${getLightingPromptHint(TIME_OF_DAY)}`,
|
|
224
|
+
)
|
|
225
|
+
})
|
|
226
|
+
})
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* ORDER TOTALITY — every video-surface dimension, folded in one call.
|
|
230
|
+
*
|
|
231
|
+
* The fixture is keyed in `directionFieldsForSurface("video")` order and pinned
|
|
232
|
+
* against it, so adding, removing or reordering a video row fails HERE as well
|
|
233
|
+
* as in the registry test. Each id was chosen to render a clause distinct from
|
|
234
|
+
* every other row's, so the dedupe pass cannot mask a mis-ordering.
|
|
235
|
+
*/
|
|
236
|
+
const EVERY_VIDEO_DIMENSION: Record<string, string> = {
|
|
237
|
+
cameraMotion: "static",
|
|
238
|
+
shotSize: "extreme-wide-shot",
|
|
239
|
+
angle: "eye-level",
|
|
240
|
+
coverage: "single",
|
|
241
|
+
composition: "rule-of-thirds",
|
|
242
|
+
vantage: "front-on",
|
|
243
|
+
pose: "standing-upright",
|
|
244
|
+
compositionEffect: "bursting-through-frame",
|
|
245
|
+
cameraFormat: "35mm-film",
|
|
246
|
+
lens: "ultra-wide-14mm",
|
|
247
|
+
timeOfDay: "dawn",
|
|
248
|
+
lightingStyle: "three-point",
|
|
249
|
+
lightingDirection: "front",
|
|
250
|
+
lightingRatio: "ratio-1-1",
|
|
251
|
+
colorTemperature: "temp-2700k",
|
|
252
|
+
colorLook: "warm",
|
|
253
|
+
atmosphere: "clear",
|
|
254
|
+
style: "3d-render",
|
|
255
|
+
mood: "happy",
|
|
256
|
+
aesthetic: "y2k",
|
|
257
|
+
setting: "coffee-shop",
|
|
258
|
+
era: "1920s-flapper",
|
|
259
|
+
backdrop: "white-seamless",
|
|
260
|
+
actionFx: "earthquake-tremor",
|
|
261
|
+
temporalSpeed: "real-time",
|
|
262
|
+
temporalFreeze: "full-freeze",
|
|
263
|
+
temporalDirection: "forward",
|
|
264
|
+
temporalShutter: "long-exposure",
|
|
265
|
+
transition: "none",
|
|
266
|
+
loopSubject: "aurora",
|
|
267
|
+
framingId: "wide-shot",
|
|
268
|
+
framingAngleId: "medium-wide-shot",
|
|
269
|
+
lightingId: "sunrise",
|
|
270
|
+
lensId: "wide-24mm",
|
|
271
|
+
cameraFormatId: "16mm-film",
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
describe("composeVideoPromptText — order totality over every video dimension", () => {
|
|
275
|
+
it("covers exactly the video surface, in table order", () => {
|
|
276
|
+
expect(Object.keys(EVERY_VIDEO_DIMENSION)).toEqual(
|
|
277
|
+
directionFieldsForSurface("video").map((f) => f.key),
|
|
278
|
+
)
|
|
279
|
+
})
|
|
280
|
+
|
|
281
|
+
it("resolves every fixture id to a real clause", () => {
|
|
282
|
+
for (const [key, id] of Object.entries(EVERY_VIDEO_DIMENSION)) {
|
|
283
|
+
expect(composeVideoPromptText("", { [key]: id }), `${key}=${id}`).not.toBe("")
|
|
284
|
+
}
|
|
285
|
+
})
|
|
286
|
+
|
|
287
|
+
it("folds all 35 dimensions in registry order, one clause each", () => {
|
|
288
|
+
// Per-dimension renders, composed in isolation through the same public
|
|
289
|
+
// entry point, then concatenated in table order: the whole fold must equal
|
|
290
|
+
// exactly that. Any reorder, drop or duplicate shows up as a diff.
|
|
291
|
+
const expected = Object.entries(EVERY_VIDEO_DIMENSION).map(
|
|
292
|
+
([key, id]) => composeVideoPromptText("", { [key]: id })!,
|
|
293
|
+
)
|
|
294
|
+
expect(new Set(expected).size, "fixture ids must render distinct clauses").toBe(
|
|
295
|
+
expected.length,
|
|
296
|
+
)
|
|
297
|
+
expect(composeVideoPromptText("a knight", EVERY_VIDEO_DIMENSION)).toBe(
|
|
298
|
+
["a knight", ...expected].join(". "),
|
|
299
|
+
)
|
|
300
|
+
})
|
|
301
|
+
})
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
import { describe, it, expect } from "vitest"
|
|
2
|
+
import {
|
|
3
|
+
getRegisteredPickerCatalogs,
|
|
4
|
+
type PickerCatalog,
|
|
5
|
+
type PickerOption,
|
|
6
|
+
} from "../picker-catalogs.js"
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* THE GUARD THAT MAKES "FOLD BEFORE THE REFERENCE RESOLVER" SAFE.
|
|
10
|
+
*
|
|
11
|
+
* `composeVideoPromptText` folds catalog text into the prompt BODY *before*
|
|
12
|
+
* `resolveVideoReferenceCore` runs (the look/motion description has to be
|
|
13
|
+
* inside the body the resolver frames, not appended after the identity
|
|
14
|
+
* directives). The consequence is that catalog text is then scanned by three
|
|
15
|
+
* passes that treat certain substrings as INPUT GRAMMAR:
|
|
16
|
+
*
|
|
17
|
+
* - the `{image:N}` / `{video:N}` / `{audio:N}` reference-slot expander,
|
|
18
|
+
* - the `{ref:<id>}` id-addressed reference-token pass,
|
|
19
|
+
* - the `@slug:N` character/named-image mention pass.
|
|
20
|
+
*
|
|
21
|
+
* A catalog whose text happened to contain one of those shapes would be
|
|
22
|
+
* rewritten by a pass that was never meant to see it — and, worse, could change
|
|
23
|
+
* the ASSEMBLED REFERENCE COUNT, which is exactly the quantity MiniMax-H3
|
|
24
|
+
* credit prediction reserves against. That is the one theoretical coupling
|
|
25
|
+
* between this text-only fold and pricing, and those three patterns are what
|
|
26
|
+
* close it.
|
|
27
|
+
*
|
|
28
|
+
* A fourth pattern is scanned for HYGIENE, not pricing: the resolver's OUTPUT
|
|
29
|
+
* binding form `@image_N` / `@video_N` / `@audio_N`. Nothing re-parses that
|
|
30
|
+
* shape, so it cannot move the assembled reference count — but a catalog
|
|
31
|
+
* emitting one would ship a binding directive to the model that binds to
|
|
32
|
+
* nothing.
|
|
33
|
+
*
|
|
34
|
+
* Scope is deliberately TOTAL rather than "the direction dimensions": it runs
|
|
35
|
+
* over every registered picker catalog (pack-composed, so a deployment's own
|
|
36
|
+
* pack is covered too) and over EVERY string a fold can inject — the full
|
|
37
|
+
* hint, the resolved compact term, AND the label. Labels are not decoration
|
|
38
|
+
* here: the multi-pick blend renderers weave them straight into the clause
|
|
39
|
+
* (`buildMoodHint`'s "with a {label} and {label} expression", `buildAestheticHints`'
|
|
40
|
+
* "{label} + {label} aesthetic blend", `buildPhotographerHints`' "blended visual
|
|
41
|
+
* language of {label} and {label}"), and mood + aesthetic are both
|
|
42
|
+
* `surface: "both"` with `maxPicks: 2`. Totality also has to survive promotion:
|
|
43
|
+
* a dimension can join the direction channel at any time, and the guard must
|
|
44
|
+
* already hold when it does.
|
|
45
|
+
*
|
|
46
|
+
* A failure here is a CATALOG fix (reword the entry), never a fold-site change.
|
|
47
|
+
*/
|
|
48
|
+
|
|
49
|
+
/** The reference-slot grammar (`video-reference-resolver.ts`'s `REFERENCE_TOKEN_RE`, `i`-flagged there). */
|
|
50
|
+
const SLOT_TOKEN = /\{(?:image|video|audio):\d+/i
|
|
51
|
+
/** The id-addressed reference token (`ref-id-tokens.ts`'s `HAS_REF_ID_TOKEN_RE`, `i`-flagged there). */
|
|
52
|
+
const REF_ID_TOKEN = /\{ref:/i
|
|
53
|
+
/**
|
|
54
|
+
* A character / named-image mention (`@kira:1`), boundary-guarded like the pass.
|
|
55
|
+
*
|
|
56
|
+
* `i`-flagged even though `findCharacterMentionTokens` is NOT: `buildMoodHint`
|
|
57
|
+
* folds `label.toLowerCase()`, so an upper-case mention-shaped label reaches the
|
|
58
|
+
* prompt lower-cased — i.e. as live grammar. Scanning case-insensitively is
|
|
59
|
+
* strictly the safe side; do not "correct" this to match the pass.
|
|
60
|
+
*/
|
|
61
|
+
const MENTION = /(?:^|[^A-Za-z0-9])@[a-z][a-z0-9-]*:\d+/i
|
|
62
|
+
/** The resolver's OUTPUT binding form (`video-reference-resolver.ts` emits `@${kind}_${n}`). */
|
|
63
|
+
const BINDING_FORM = /@(?:image|video|audio)_\d+/i
|
|
64
|
+
|
|
65
|
+
const FORBIDDEN: ReadonlyArray<{ name: string; re: RegExp }> = [
|
|
66
|
+
{ name: "reference slot token ({image:N} / {video:N} / {audio:N})", re: SLOT_TOKEN },
|
|
67
|
+
{ name: "id-addressed reference token ({ref:…})", re: REF_ID_TOKEN },
|
|
68
|
+
{ name: "character/named-image mention (@slug:N)", re: MENTION },
|
|
69
|
+
{ name: "reference binding form (@image_N / @video_N / @audio_N)", re: BINDING_FORM },
|
|
70
|
+
]
|
|
71
|
+
|
|
72
|
+
/** Every option of a catalog, single-dim and multi-dim alike. */
|
|
73
|
+
function allOptions(catalog: PickerCatalog): ReadonlyArray<PickerOption> {
|
|
74
|
+
return [
|
|
75
|
+
...(catalog.options ?? []),
|
|
76
|
+
...(catalog.dimensions ?? []).flatMap((d) => d.options),
|
|
77
|
+
]
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
describe("direction hint token safety", () => {
|
|
81
|
+
const catalogs = getRegisteredPickerCatalogs()
|
|
82
|
+
|
|
83
|
+
it("has catalogs to check (the guard must not pass vacuously)", () => {
|
|
84
|
+
expect(catalogs.length).toBeGreaterThan(0)
|
|
85
|
+
expect(catalogs.reduce((n, c) => n + allOptions(c).length, 0)).toBeGreaterThan(1000)
|
|
86
|
+
})
|
|
87
|
+
|
|
88
|
+
it("emits no reference-grammar token from any catalog hint, term or label", () => {
|
|
89
|
+
const offenders: string[] = []
|
|
90
|
+
for (const catalog of catalogs) {
|
|
91
|
+
for (const option of allOptions(catalog)) {
|
|
92
|
+
// `term` is already RESOLVED by the projection (`resolveTerm`), so this
|
|
93
|
+
// checks exactly the string a compact-mode fold would inject. `label`
|
|
94
|
+
// is injected verbatim by the multi-pick blend renderers (see the
|
|
95
|
+
// header). `description` is deliberately NOT scanned — no render path
|
|
96
|
+
// folds it into the prompt.
|
|
97
|
+
for (const [field, text] of [
|
|
98
|
+
["promptHint", option.promptHint],
|
|
99
|
+
["term", option.term],
|
|
100
|
+
["label", option.label],
|
|
101
|
+
] as const) {
|
|
102
|
+
if (!text) continue
|
|
103
|
+
for (const { name, re } of FORBIDDEN) {
|
|
104
|
+
if (re.test(text)) {
|
|
105
|
+
offenders.push(`${catalog.nodeType} • ${option.id} • ${field}: ${name}`)
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
expect(offenders).toEqual([])
|
|
112
|
+
})
|
|
113
|
+
})
|