@nodaro/prompts 1.12.0 → 1.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +303 -187
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +233 -28
- package/dist/index.d.ts +233 -28
- package/dist/index.js +290 -188
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/__tests__/assemble-image-input-cap.test.ts +37 -13
- package/src/__tests__/assemble-image-input.test.ts +100 -19
- package/src/__tests__/assemble-video-input-cap.test.ts +101 -15
- package/src/__tests__/assemble-video-input.test.ts +167 -33
- package/src/__tests__/direction-hint-token-safety.test.ts +21 -0
- package/src/__tests__/prompt-style-section.test.ts +345 -0
- package/src/__tests__/style-section-boundary.test.ts +179 -0
- package/src/__tests__/subject-fold.test.ts +32 -13
- package/src/assemble-image-input.ts +51 -26
- package/src/assemble-video-input.ts +54 -34
- package/src/direction-registry.ts +108 -37
- package/src/hint-shedding.ts +23 -4
- package/src/index.ts +1 -0
- package/src/prompt-builder.ts +84 -27
- package/src/prompt-hint-join.ts +9 -0
- package/src/prompt-style-section.ts +256 -0
- package/src/video-reference-resolver.ts +5 -2
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
import { describe, it, expect } from "vitest"
|
|
2
|
+
import {
|
|
3
|
+
STYLE_SECTION_HEADER,
|
|
4
|
+
asBodyClauses,
|
|
5
|
+
composeSectionedPrompt,
|
|
6
|
+
endsInsideStyleSection,
|
|
7
|
+
insertBeforeStyleSection,
|
|
8
|
+
partitionStyleClauses,
|
|
9
|
+
renderStyleSection,
|
|
10
|
+
sectionedClauseCosts,
|
|
11
|
+
splitStyleSection,
|
|
12
|
+
styleSectionFromClauses,
|
|
13
|
+
} from "../prompt-style-section.js"
|
|
14
|
+
import {
|
|
15
|
+
DIRECTION_FIELDS,
|
|
16
|
+
FILM_STYLE_KEYS,
|
|
17
|
+
IMAGE_HINT_MODE_DEFAULT,
|
|
18
|
+
VIDEO_HINT_MODE_DEFAULT,
|
|
19
|
+
directionFieldsForSurface,
|
|
20
|
+
renderDirectionHints,
|
|
21
|
+
} from "../direction-registry.js"
|
|
22
|
+
import { joinPromptHints } from "../prompt-hint-join.js"
|
|
23
|
+
import { getStylePromptHint } from "../style.js"
|
|
24
|
+
import { getColorLookPromptHint } from "../color-look.js"
|
|
25
|
+
import { getEraPromptHint } from "../era.js"
|
|
26
|
+
import { getCameraFormatPromptHint } from "../camera-format.js"
|
|
27
|
+
import { getFramingPromptHint } from "../framing.js"
|
|
28
|
+
import { getLightingPromptHint } from "../lighting.js"
|
|
29
|
+
import { getCameraMotionTerm } from "../camera-motions.js"
|
|
30
|
+
import { getTransitionTerm } from "../transitions.js"
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* THE `[style]` SECTION CONTRACT, at the level it is defined: clauses in, one
|
|
34
|
+
* string out. The composers (`assembleImageInput`, `composeVideoPromptText`)
|
|
35
|
+
* are pinned against the same shape in their own suites; what lives here is the
|
|
36
|
+
* grammar itself — which clause lands on which line, and the exact bytes.
|
|
37
|
+
*
|
|
38
|
+
* The header is asserted as a LITERAL everywhere below, never through
|
|
39
|
+
* `STYLE_SECTION_HEADER`, so a typo in the constant fails here instead of
|
|
40
|
+
* silently redefining the contract.
|
|
41
|
+
*/
|
|
42
|
+
|
|
43
|
+
// Real catalog ids: every `get*PromptHint` returns "" on a miss, so a made-up
|
|
44
|
+
// id would make most of these assertions vacuously pass.
|
|
45
|
+
const STYLE = "anime" // look • film
|
|
46
|
+
const COLOR_LOOK = "teal-orange" // look • film
|
|
47
|
+
const ERA = "1920s-flapper" // look • film
|
|
48
|
+
const CAMERA_FORMAT = "16mm-film" // look • film
|
|
49
|
+
const SHOT_SIZE = "wide-shot" // look • scene
|
|
50
|
+
const TIME_OF_DAY = "golden-hour" // look • scene
|
|
51
|
+
const CAMERA_MOTION = "handheld" // motion • body
|
|
52
|
+
const TRANSITION = "cross-dissolve" // motion • body
|
|
53
|
+
|
|
54
|
+
const IMAGE = { surface: "image", mode: IMAGE_HINT_MODE_DEFAULT } as const
|
|
55
|
+
const VIDEO = { surface: "video", mode: VIDEO_HINT_MODE_DEFAULT } as const
|
|
56
|
+
|
|
57
|
+
describe("the section header", () => {
|
|
58
|
+
it("is exactly `[style]:`, lowercase", () => {
|
|
59
|
+
expect(STYLE_SECTION_HEADER).toBe("[style]:")
|
|
60
|
+
})
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
describe("FILM_STYLE_KEYS — the film/scene split lives in the registry", () => {
|
|
64
|
+
it("names the five film rows, in table order", () => {
|
|
65
|
+
expect(FILM_STYLE_KEYS).toEqual([
|
|
66
|
+
"cameraFormat",
|
|
67
|
+
"colorLook",
|
|
68
|
+
"style",
|
|
69
|
+
"era",
|
|
70
|
+
"cameraFormatId",
|
|
71
|
+
])
|
|
72
|
+
})
|
|
73
|
+
|
|
74
|
+
it("derives from the table's own `styleGroup` column (no second list)", () => {
|
|
75
|
+
expect(FILM_STYLE_KEYS).toEqual(
|
|
76
|
+
DIRECTION_FIELDS.filter((f) => "styleGroup" in f && f.styleGroup === "film").map(
|
|
77
|
+
(f) => f.key,
|
|
78
|
+
),
|
|
79
|
+
)
|
|
80
|
+
})
|
|
81
|
+
|
|
82
|
+
it("marks only LOOK rows as film (a motion row could never reach the section)", () => {
|
|
83
|
+
for (const spec of DIRECTION_FIELDS) {
|
|
84
|
+
if ("styleGroup" in spec && spec.styleGroup === "film") {
|
|
85
|
+
expect(spec.family, spec.key).toBe("look")
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
})
|
|
89
|
+
})
|
|
90
|
+
|
|
91
|
+
describe("partitionStyleClauses — which slot a clause lands in", () => {
|
|
92
|
+
it("sends the MOTION family to the body and the LOOK family to the section", () => {
|
|
93
|
+
const slots = new Map(
|
|
94
|
+
partitionStyleClauses(
|
|
95
|
+
{ cameraMotion: CAMERA_MOTION, transition: TRANSITION, style: STYLE, shotSize: SHOT_SIZE },
|
|
96
|
+
VIDEO,
|
|
97
|
+
).map((c) => [c.text, c.slot]),
|
|
98
|
+
)
|
|
99
|
+
expect(slots.get(getCameraMotionTerm(CAMERA_MOTION))).toBe("body")
|
|
100
|
+
expect(slots.get(getTransitionTerm(TRANSITION))).toBe("body")
|
|
101
|
+
expect(slots.get(getStylePromptHint(STYLE))).toBe("film")
|
|
102
|
+
expect(slots.get(getFramingPromptHint(SHOT_SIZE))).toBe("scene")
|
|
103
|
+
})
|
|
104
|
+
|
|
105
|
+
it("leaves the image surface with no body clause at all (no motion row folds there)", () => {
|
|
106
|
+
// The surface filter and the family split are deliberately aligned: every
|
|
107
|
+
// image-surface direction row is `look`, so an image `[style]` section
|
|
108
|
+
// carries the WHOLE direction fold and the body carries none of it.
|
|
109
|
+
expect(directionFieldsForSurface("image").every((f) => f.family === "look")).toBe(true)
|
|
110
|
+
const everyImageKey = Object.fromEntries(
|
|
111
|
+
directionFieldsForSurface("image").map((f) => [f.key, ""]),
|
|
112
|
+
)
|
|
113
|
+
expect(
|
|
114
|
+
partitionStyleClauses({ ...everyImageKey, style: STYLE, shotSize: SHOT_SIZE }, IMAGE).every(
|
|
115
|
+
(c) => c.slot !== "body",
|
|
116
|
+
),
|
|
117
|
+
).toBe(true)
|
|
118
|
+
})
|
|
119
|
+
|
|
120
|
+
it("keeps registry table order inside each slot", () => {
|
|
121
|
+
const direction = { style: STYLE, colorLook: COLOR_LOOK, shotSize: SHOT_SIZE, timeOfDay: TIME_OF_DAY }
|
|
122
|
+
expect(partitionStyleClauses(direction, IMAGE).map((c) => c.text)).toEqual(
|
|
123
|
+
renderDirectionHints(direction, IMAGE),
|
|
124
|
+
)
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
it("returns nothing for an absent, empty or unresolvable direction", () => {
|
|
128
|
+
expect(partitionStyleClauses(undefined, IMAGE)).toEqual([])
|
|
129
|
+
expect(partitionStyleClauses({}, IMAGE)).toEqual([])
|
|
130
|
+
expect(partitionStyleClauses({ style: "__no_such_id__" }, IMAGE)).toEqual([])
|
|
131
|
+
})
|
|
132
|
+
})
|
|
133
|
+
|
|
134
|
+
describe("renderStyleSection — the two lines", () => {
|
|
135
|
+
it("puts the film line first and the scene line second, each `. `-joined", () => {
|
|
136
|
+
expect(
|
|
137
|
+
renderStyleSection(
|
|
138
|
+
{
|
|
139
|
+
cameraFormat: CAMERA_FORMAT,
|
|
140
|
+
colorLook: COLOR_LOOK,
|
|
141
|
+
style: STYLE,
|
|
142
|
+
era: ERA,
|
|
143
|
+
shotSize: SHOT_SIZE,
|
|
144
|
+
timeOfDay: TIME_OF_DAY,
|
|
145
|
+
},
|
|
146
|
+
IMAGE,
|
|
147
|
+
),
|
|
148
|
+
).toBe(
|
|
149
|
+
"[style]:\n" +
|
|
150
|
+
[
|
|
151
|
+
getCameraFormatPromptHint(CAMERA_FORMAT),
|
|
152
|
+
getColorLookPromptHint(COLOR_LOOK),
|
|
153
|
+
getStylePromptHint(STYLE),
|
|
154
|
+
getEraPromptHint(ERA),
|
|
155
|
+
].join(". ") +
|
|
156
|
+
"\n" +
|
|
157
|
+
[getFramingPromptHint(SHOT_SIZE), getLightingPromptHint(TIME_OF_DAY)].join(". "),
|
|
158
|
+
)
|
|
159
|
+
})
|
|
160
|
+
|
|
161
|
+
it("omits the film line entirely when no film dimension is selected", () => {
|
|
162
|
+
expect(renderStyleSection({ shotSize: SHOT_SIZE }, IMAGE)).toBe(
|
|
163
|
+
`[style]:\n${getFramingPromptHint(SHOT_SIZE)}`,
|
|
164
|
+
)
|
|
165
|
+
})
|
|
166
|
+
|
|
167
|
+
it("omits the scene line entirely when no other look dimension is selected", () => {
|
|
168
|
+
expect(renderStyleSection({ style: STYLE }, IMAGE)).toBe(
|
|
169
|
+
`[style]:\n${getStylePromptHint(STYLE)}`,
|
|
170
|
+
)
|
|
171
|
+
})
|
|
172
|
+
|
|
173
|
+
it("renders NOTHING when the fold carries no look clause", () => {
|
|
174
|
+
expect(renderStyleSection(undefined, VIDEO)).toBe("")
|
|
175
|
+
expect(renderStyleSection({}, VIDEO)).toBe("")
|
|
176
|
+
expect(renderStyleSection({ cameraMotion: CAMERA_MOTION, transition: TRANSITION }, VIDEO)).toBe("")
|
|
177
|
+
})
|
|
178
|
+
|
|
179
|
+
it("never indents a line and never ends with a newline", () => {
|
|
180
|
+
// The video reference resolver collapses 2+ HORIZONTAL spaces unanchored,
|
|
181
|
+
// so an indented section line would come back flattened — the section is
|
|
182
|
+
// written flush-left instead of relying on the collapse leaving it alone.
|
|
183
|
+
const section = renderStyleSection(
|
|
184
|
+
{ style: STYLE, colorLook: COLOR_LOOK, shotSize: SHOT_SIZE },
|
|
185
|
+
IMAGE,
|
|
186
|
+
)
|
|
187
|
+
for (const line of section.split("\n")) expect(line).toBe(line.trimStart())
|
|
188
|
+
expect(section.endsWith("\n")).toBe(false)
|
|
189
|
+
expect(section).not.toMatch(/[^\S\r\n]{2,}/)
|
|
190
|
+
})
|
|
191
|
+
|
|
192
|
+
it("agrees with the clause-level renderer the composers use", () => {
|
|
193
|
+
const direction = { style: STYLE, shotSize: SHOT_SIZE }
|
|
194
|
+
expect(renderStyleSection(direction, VIDEO)).toBe(
|
|
195
|
+
styleSectionFromClauses(partitionStyleClauses(direction, VIDEO)),
|
|
196
|
+
)
|
|
197
|
+
})
|
|
198
|
+
})
|
|
199
|
+
|
|
200
|
+
describe("composeSectionedPrompt — body, gap, section", () => {
|
|
201
|
+
const FILM = getStylePromptHint(STYLE)
|
|
202
|
+
const SCENE = getFramingPromptHint(SHOT_SIZE)
|
|
203
|
+
const clauses = [
|
|
204
|
+
{ text: "a knight rides", slot: "body" },
|
|
205
|
+
{ text: FILM, slot: "film" },
|
|
206
|
+
{ text: SCENE, slot: "scene" },
|
|
207
|
+
] as const
|
|
208
|
+
|
|
209
|
+
it("joins body clauses with `. ` and hangs the section off a blank line", () => {
|
|
210
|
+
expect(composeSectionedPrompt("at dusk", clauses, "")).toBe(
|
|
211
|
+
`at dusk. a knight rides\n\n[style]:\n${FILM}\n${SCENE}`,
|
|
212
|
+
)
|
|
213
|
+
})
|
|
214
|
+
|
|
215
|
+
it("keeps the structured fragment last IN THE BODY, ahead of the section", () => {
|
|
216
|
+
expect(composeSectionedPrompt("at dusk", clauses, "Subject: a knight.")).toBe(
|
|
217
|
+
`at dusk. a knight rides. Subject: a knight.\n\n[style]:\n${FILM}\n${SCENE}`,
|
|
218
|
+
)
|
|
219
|
+
})
|
|
220
|
+
|
|
221
|
+
it("emits NO header and no extra newline when nothing reaches the section", () => {
|
|
222
|
+
// Byte-identical to the plain hint join — this is what keeps every
|
|
223
|
+
// look-free caller (and every fully-shed one) exactly where it was.
|
|
224
|
+
const bodyOnly = [{ text: "a knight rides", slot: "body" }] as const
|
|
225
|
+
expect(composeSectionedPrompt("at dusk", bodyOnly, "")).toBe(
|
|
226
|
+
joinPromptHints("at dusk", ["a knight rides"]),
|
|
227
|
+
)
|
|
228
|
+
expect(composeSectionedPrompt("at dusk", bodyOnly, "")).not.toContain("[style]")
|
|
229
|
+
})
|
|
230
|
+
|
|
231
|
+
it("returns the prompt VERBATIM AND UNTRIMMED with no clause and no fragment", () => {
|
|
232
|
+
expect(composeSectionedPrompt(" a knight \n", [], "")).toBe(" a knight \n")
|
|
233
|
+
expect(composeSectionedPrompt(undefined, [], "")).toBeUndefined()
|
|
234
|
+
})
|
|
235
|
+
|
|
236
|
+
it("TRIMS the prompt when the section is the only thing folded", () => {
|
|
237
|
+
// The section counts as "something folded", so the body is trimmed exactly
|
|
238
|
+
// as the hint-join branch trims it — otherwise the blank line would inherit
|
|
239
|
+
// the prompt's trailing whitespace.
|
|
240
|
+
expect(composeSectionedPrompt(" a knight \n", [{ text: FILM, slot: "film" }], "")).toBe(
|
|
241
|
+
`a knight\n\n[style]:\n${FILM}`,
|
|
242
|
+
)
|
|
243
|
+
})
|
|
244
|
+
|
|
245
|
+
it("drops the gap for a blank or absent prompt (never a leading newline)", () => {
|
|
246
|
+
const only = [{ text: FILM, slot: "film" }] as const
|
|
247
|
+
expect(composeSectionedPrompt("", only, "")).toBe(`[style]:\n${FILM}`)
|
|
248
|
+
expect(composeSectionedPrompt(" ", only, "")).toBe(`[style]:\n${FILM}`)
|
|
249
|
+
expect(composeSectionedPrompt(undefined, only, "")).toBe(`[style]:\n${FILM}`)
|
|
250
|
+
})
|
|
251
|
+
|
|
252
|
+
it("never ends the composed prompt with a newline", () => {
|
|
253
|
+
for (const prompt of ["a knight", "", undefined]) {
|
|
254
|
+
expect(composeSectionedPrompt(prompt, clauses, "Subject: a knight.")!.endsWith("\n")).toBe(
|
|
255
|
+
false,
|
|
256
|
+
)
|
|
257
|
+
}
|
|
258
|
+
})
|
|
259
|
+
|
|
260
|
+
it("marks every subject clause as body", () => {
|
|
261
|
+
expect(asBodyClauses(["a woman in her 30s", "wearing a red coat"])).toEqual([
|
|
262
|
+
{ text: "a woman in her 30s", slot: "body" },
|
|
263
|
+
{ text: "wearing a red coat", slot: "body" },
|
|
264
|
+
])
|
|
265
|
+
})
|
|
266
|
+
})
|
|
267
|
+
|
|
268
|
+
describe("sectionedClauseCosts — what each clause really costs", () => {
|
|
269
|
+
const clauses = [
|
|
270
|
+
{ text: "a knight rides", slot: "body" },
|
|
271
|
+
{ text: getStylePromptHint(STYLE), slot: "film" },
|
|
272
|
+
{ text: getFramingPromptHint(SHOT_SIZE), slot: "scene" },
|
|
273
|
+
] as const
|
|
274
|
+
|
|
275
|
+
it("is the exact composed-length delta of each clause, tail-first", () => {
|
|
276
|
+
const costs = sectionedClauseCosts("at dusk", clauses, "")
|
|
277
|
+
expect(costs).toHaveLength(clauses.length)
|
|
278
|
+
for (let kept = 0; kept < clauses.length; kept++) {
|
|
279
|
+
const below = composeSectionedPrompt("at dusk", clauses.slice(0, kept), "")?.length ?? 0
|
|
280
|
+
const at = composeSectionedPrompt("at dusk", clauses.slice(0, kept + 1), "")?.length ?? 0
|
|
281
|
+
expect(costs[kept]).toBe(at - below)
|
|
282
|
+
}
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
it("charges the LAST surviving look clause for the header it keeps alive", () => {
|
|
286
|
+
// "\n\n[style]:\n" is 11 characters that only come back when the section
|
|
287
|
+
// disappears entirely — so the first look clause carries them, and a shed
|
|
288
|
+
// that drops it reclaims more than the clause's own text.
|
|
289
|
+
const costs = sectionedClauseCosts("at dusk", clauses, "")
|
|
290
|
+
expect(costs[1]).toBe(getStylePromptHint(STYLE).length + "\n\n[style]:\n".length)
|
|
291
|
+
// The second look clause only brings its own line separator.
|
|
292
|
+
expect(costs[2]).toBe(getFramingPromptHint(SHOT_SIZE).length + "\n".length)
|
|
293
|
+
// A body clause brings the ". " it was joined with.
|
|
294
|
+
expect(costs[0]).toBe("a knight rides".length + ". ".length)
|
|
295
|
+
})
|
|
296
|
+
})
|
|
297
|
+
|
|
298
|
+
describe("the section boundary — what a later assembler may append", () => {
|
|
299
|
+
const FILM = getStylePromptHint(STYLE)
|
|
300
|
+
const clauses = [{ text: FILM, slot: "film" }] as const
|
|
301
|
+
const composed = composeSectionedPrompt("a knight", clauses, "")!
|
|
302
|
+
const bodyless = composeSectionedPrompt("", clauses, "")!
|
|
303
|
+
|
|
304
|
+
it("splits a composed prompt into its body and its section", () => {
|
|
305
|
+
expect(splitStyleSection(composed)).toEqual({ body: "a knight", section: `[style]:\n${FILM}` })
|
|
306
|
+
})
|
|
307
|
+
|
|
308
|
+
it("splits the body-less form, where the section IS the prompt", () => {
|
|
309
|
+
expect(splitStyleSection(bodyless)).toEqual({ body: "", section: `[style]:\n${FILM}` })
|
|
310
|
+
})
|
|
311
|
+
|
|
312
|
+
it("reports no section for a prompt that carries none", () => {
|
|
313
|
+
expect(splitStyleSection("a knight")).toEqual({ body: "a knight", section: "" })
|
|
314
|
+
})
|
|
315
|
+
|
|
316
|
+
it("inserts body lines AHEAD of the section, keeping the look clauses last", () => {
|
|
317
|
+
expect(insertBeforeStyleSection(composed, ["the person from reference image A"])).toBe(
|
|
318
|
+
`a knight\nthe person from reference image A\n\n[style]:\n${FILM}`,
|
|
319
|
+
)
|
|
320
|
+
expect(insertBeforeStyleSection(bodyless, ["the person from reference image A"])).toBe(
|
|
321
|
+
`the person from reference image A\n\n[style]:\n${FILM}`,
|
|
322
|
+
)
|
|
323
|
+
})
|
|
324
|
+
|
|
325
|
+
it("is the plain `\\n` join with no section — the byte-parity path", () => {
|
|
326
|
+
// What every appender emitted before the section existed, including the
|
|
327
|
+
// leading newline an empty prompt produces. Anything else would move bytes
|
|
328
|
+
// on the look-free runs, which are most of them.
|
|
329
|
+
expect(insertBeforeStyleSection("a knight", ["a", "b"])).toBe("a knight\na\nb")
|
|
330
|
+
expect(insertBeforeStyleSection("", ["a"])).toBe("\na")
|
|
331
|
+
})
|
|
332
|
+
|
|
333
|
+
it("is a no-op with no lines to add", () => {
|
|
334
|
+
expect(insertBeforeStyleSection(composed, [])).toBe(composed)
|
|
335
|
+
})
|
|
336
|
+
|
|
337
|
+
it("knows when a prompt ends INSIDE the section", () => {
|
|
338
|
+
expect(endsInsideStyleSection(composed)).toBe(true)
|
|
339
|
+
expect(endsInsideStyleSection(bodyless)).toBe(true)
|
|
340
|
+
// A blank line closes the header's scope, and the next appender sees it.
|
|
341
|
+
expect(endsInsideStyleSection(`${composed}\n\nStyle: cinematic`)).toBe(false)
|
|
342
|
+
expect(endsInsideStyleSection("a knight")).toBe(false)
|
|
343
|
+
expect(endsInsideStyleSection("")).toBe(false)
|
|
344
|
+
})
|
|
345
|
+
})
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
import { describe, it, expect } from "vitest"
|
|
2
|
+
import { applyVideoNegativePrompt, getMaxImagePromptChars } from "@nodaro/shared"
|
|
3
|
+
import type { CharacterDef, ConnectedReference } from "@nodaro/shared"
|
|
4
|
+
import { buildImagePrompt } from "../prompt-builder.js"
|
|
5
|
+
import { assembleImageInput } from "../assemble-image-input.js"
|
|
6
|
+
import { composeVideoPromptText } from "../assemble-video-input.js"
|
|
7
|
+
import { resolveVideoReferenceCore } from "../video-reference-resolver.js"
|
|
8
|
+
import { getStylePromptHint } from "../style.js"
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* WHERE THE `[style]` SECTION ENDS. The section has no terminator and every
|
|
12
|
+
* assembler downstream of the composer appends to the string it returns — the
|
|
13
|
+
* reference resolvers' role phrases and element directives, the legacy
|
|
14
|
+
* character descriptions, `Style:` and `Avoid:`. Line-joined onto a prompt that
|
|
15
|
+
* ENDS with the section, each of them reads as one more look clause under the
|
|
16
|
+
* header, which is exactly the confusion the section exists to remove.
|
|
17
|
+
*
|
|
18
|
+
* Two shapes keep them out, one per kind of text:
|
|
19
|
+
* - BODY content — a reference binding, an element directive, a character
|
|
20
|
+
* description — is SPLICED in ahead of the section, where the prose it
|
|
21
|
+
* belongs with already is. The look tail stays last, which is where it was
|
|
22
|
+
* measured to be free.
|
|
23
|
+
* - The self-labeling control lines (`Style:` / `Avoid:`) stay at the end and
|
|
24
|
+
* close the header's scope with a blank line instead.
|
|
25
|
+
*
|
|
26
|
+
* Assertions read the header as a LITERAL and the clause wording through the
|
|
27
|
+
* catalogs, so this suite pins the BOUNDARY, not the catalog copy.
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
/** The `\n\n`-delimited block the `[style]:` header opens — every line the
|
|
31
|
+
* model reads as a look clause. `""` when the prompt carries no section. */
|
|
32
|
+
const styleBlockOf = (prompt: string): string =>
|
|
33
|
+
prompt.split("\n\n").find((b) => b.startsWith("[style]:")) ?? ""
|
|
34
|
+
|
|
35
|
+
const LIBRARY: ConnectedReference = {
|
|
36
|
+
id: "l", defaultName: "Old Library", source: "wired-location",
|
|
37
|
+
url: "https://cdn/library.png", locationSlug: "old-library",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
const KIRA: ConnectedReference = {
|
|
41
|
+
id: "kira", defaultName: "Kira", source: "wired-character",
|
|
42
|
+
url: "https://cdn/kira.png", characterSlug: "kira",
|
|
43
|
+
characterCanonicalDescription: "a young woman with copper hair",
|
|
44
|
+
variantSlug: undefined, variantDescription: null, variantDisplayName: "canonical",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
const KIRA_DEF: CharacterDef = {
|
|
48
|
+
id: "kira", name: "Kira", type: "description", category: "character",
|
|
49
|
+
description: "auburn hair, hazel eyes",
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Look on both lines: `style` is a film clause, `shotSize` a scene clause. */
|
|
53
|
+
const DIRECTION = { style: "anime", shotSize: "wide-shot" } as const
|
|
54
|
+
|
|
55
|
+
/** What the inline `style` text renders as in the `Style:` control line. */
|
|
56
|
+
const CINEMATIC = getStylePromptHint("cinematic")
|
|
57
|
+
|
|
58
|
+
/** The trailing role phrase a canonical (unmentioned) wired location renders. */
|
|
59
|
+
const LOCATION_PHRASE = "the location from reference image A"
|
|
60
|
+
/** The video twin, bound to the resolver's `@image_N` shape. */
|
|
61
|
+
const CHARACTER_PHRASE = "the person from @image_1"
|
|
62
|
+
|
|
63
|
+
describe("image assembly — nothing lands under the `[style]` header", () => {
|
|
64
|
+
const hybrid = assembleImageInput({
|
|
65
|
+
userPrompt: "a man walks",
|
|
66
|
+
provider: "nano-banana",
|
|
67
|
+
direction: DIRECTION,
|
|
68
|
+
connectedReferences: [LIBRARY],
|
|
69
|
+
referenceFormat: "hybrid",
|
|
70
|
+
style: "cinematic",
|
|
71
|
+
negativePrompt: "blurry",
|
|
72
|
+
}).prompt
|
|
73
|
+
|
|
74
|
+
it("splices the reference binding into the body, ahead of the section", () => {
|
|
75
|
+
// Non-vacuity: the section really is there, carrying both look lines.
|
|
76
|
+
expect(styleBlockOf(hybrid)).toContain("[style]:\n")
|
|
77
|
+
expect(styleBlockOf(hybrid)).toContain("wide shot")
|
|
78
|
+
|
|
79
|
+
expect(styleBlockOf(hybrid)).not.toContain(LOCATION_PHRASE)
|
|
80
|
+
expect(hybrid.indexOf(LOCATION_PHRASE)).toBeGreaterThan(-1)
|
|
81
|
+
expect(hybrid.indexOf(LOCATION_PHRASE)).toBeLessThan(hybrid.indexOf("[style]:"))
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
it("closes the section with a blank line before `Style:` / `Avoid:`", () => {
|
|
85
|
+
expect(styleBlockOf(hybrid)).not.toContain("Style:")
|
|
86
|
+
expect(styleBlockOf(hybrid)).not.toContain("Avoid:")
|
|
87
|
+
// The control lines stay together at the very end, one block of their own:
|
|
88
|
+
// the blank line is `Style:`'s to add, and `Avoid:` sees a closed section.
|
|
89
|
+
expect(hybrid.endsWith(`\n\nStyle: ${CINEMATIC}\nAvoid: blurry`)).toBe(true)
|
|
90
|
+
})
|
|
91
|
+
|
|
92
|
+
it("splices the LEGACY character description into the body too", () => {
|
|
93
|
+
const legacy = assembleImageInput({
|
|
94
|
+
userPrompt: "a man walks",
|
|
95
|
+
provider: "nano-banana",
|
|
96
|
+
direction: DIRECTION,
|
|
97
|
+
characterDefs: [KIRA_DEF],
|
|
98
|
+
}).prompt
|
|
99
|
+
const desc = "Include character 'Kira': auburn hair, hazel eyes."
|
|
100
|
+
expect(legacy).toContain(desc)
|
|
101
|
+
expect(styleBlockOf(legacy)).not.toContain(desc)
|
|
102
|
+
expect(legacy.indexOf(desc)).toBeLessThan(legacy.indexOf("[style]:"))
|
|
103
|
+
})
|
|
104
|
+
|
|
105
|
+
it("honors a user-overridden wrapper template while keeping the section last", () => {
|
|
106
|
+
const legacy = assembleImageInput({
|
|
107
|
+
userPrompt: "a man walks",
|
|
108
|
+
provider: "nano-banana",
|
|
109
|
+
direction: DIRECTION,
|
|
110
|
+
characterDefs: [KIRA_DEF],
|
|
111
|
+
userTemplates: { "generate-image-wrapper": "{assetDescriptions} // {userPrompt}" },
|
|
112
|
+
}).prompt
|
|
113
|
+
expect(legacy.startsWith("Include character 'Kira': auburn hair, hazel eyes. // a man walks")).toBe(true)
|
|
114
|
+
expect(styleBlockOf(legacy)).not.toContain("Kira")
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
it("re-derives the separator after the cap's tail cut, within the reservation", () => {
|
|
118
|
+
// The control lines are reserved BEFORE the cut and rendered AFTER it, and
|
|
119
|
+
// the cut moves the boundary they read. Both directions, one cap:
|
|
120
|
+
const cap = getMaxImagePromptChars("seedream")
|
|
121
|
+
const cut = (prompt: string) =>
|
|
122
|
+
buildImagePrompt({ prompt, provider: "seedream", style: "cinematic", negativePrompt: "blurry" }).prompt
|
|
123
|
+
const tail = `\nStyle: ${CINEMATIC}\nAvoid: blurry`
|
|
124
|
+
|
|
125
|
+
// (a) a short section, cut away entirely → the separator narrows to `\n`.
|
|
126
|
+
const shortSection = cut(`${"a man walks. ".repeat(400)}\n\n[style]:\nanime style`)
|
|
127
|
+
expect(shortSection.length).toBeLessThanOrEqual(cap)
|
|
128
|
+
expect(shortSection.endsWith(`...${tail}`)).toBe(true)
|
|
129
|
+
|
|
130
|
+
// (b) a section longer than the reservation, so the cut lands INSIDE it →
|
|
131
|
+
// the separator stays a blank line, exactly what was reserved.
|
|
132
|
+
const look = "anime style, ".repeat(24)
|
|
133
|
+
const longSection = cut(`${"x".repeat(cap - look.length - 60)}\n\n[style]:\n${look}`)
|
|
134
|
+
expect(longSection.length).toBeLessThanOrEqual(cap)
|
|
135
|
+
expect(longSection.endsWith(`...\n${tail}`)).toBe(true)
|
|
136
|
+
})
|
|
137
|
+
|
|
138
|
+
it("keeps the no-section shapes byte-identical (the append fallback)", () => {
|
|
139
|
+
const noSection = assembleImageInput({
|
|
140
|
+
userPrompt: "a man walks",
|
|
141
|
+
provider: "nano-banana",
|
|
142
|
+
connectedReferences: [LIBRARY],
|
|
143
|
+
referenceFormat: "hybrid",
|
|
144
|
+
style: "cinematic",
|
|
145
|
+
negativePrompt: "blurry",
|
|
146
|
+
}).prompt
|
|
147
|
+
expect(noSection).toBe(
|
|
148
|
+
`A man walks\n${LOCATION_PHRASE}\nStyle: ${CINEMATIC}\nAvoid: blurry`,
|
|
149
|
+
)
|
|
150
|
+
})
|
|
151
|
+
})
|
|
152
|
+
|
|
153
|
+
describe("video assembly — nothing lands under the `[style]` header", () => {
|
|
154
|
+
const body = composeVideoPromptText("a knight rides", DIRECTION)
|
|
155
|
+
const framed = resolveVideoReferenceCore({
|
|
156
|
+
prompt: body,
|
|
157
|
+
wiredCharRefs: [KIRA],
|
|
158
|
+
hybridRoles: true,
|
|
159
|
+
}).prompt!
|
|
160
|
+
|
|
161
|
+
it("splices the resolver's role phrase ahead of the section, which ends the prompt", () => {
|
|
162
|
+
expect(styleBlockOf(framed)).toContain("[style]:\n")
|
|
163
|
+
expect(styleBlockOf(framed)).not.toContain(CHARACTER_PHRASE)
|
|
164
|
+
expect(framed.indexOf(CHARACTER_PHRASE)).toBeLessThan(framed.indexOf("[style]:"))
|
|
165
|
+
// The look tail really is the tail: nothing follows the section.
|
|
166
|
+
expect(framed.endsWith(styleBlockOf(framed))).toBe(true)
|
|
167
|
+
})
|
|
168
|
+
|
|
169
|
+
it("closes the section with a blank line before the folded `Avoid:`", () => {
|
|
170
|
+
const withNeg = applyVideoNegativePrompt(framed, "blurry, watermark", "seedance-2").prompt!
|
|
171
|
+
expect(styleBlockOf(withNeg)).not.toContain("Avoid:")
|
|
172
|
+
expect(withNeg.endsWith("\n\nAvoid: blurry, watermark")).toBe(true)
|
|
173
|
+
})
|
|
174
|
+
|
|
175
|
+
it("keeps the sectionless `Avoid:` join byte-identical", () => {
|
|
176
|
+
const plain = applyVideoNegativePrompt("a knight rides", "blurry", "seedance-2").prompt
|
|
177
|
+
expect(plain).toBe("a knight rides\nAvoid: blurry")
|
|
178
|
+
})
|
|
179
|
+
})
|
|
@@ -15,6 +15,7 @@ import { composeVideoPromptText } from "../assemble-video-input.js"
|
|
|
15
15
|
import { buildImagePrompt } from "../prompt-builder.js"
|
|
16
16
|
import { joinPromptHints, PROMPT_HINT_SEPARATOR } from "../prompt-hint-join.js"
|
|
17
17
|
import { renderDirectionHints, IMAGE_HINT_MODE_DEFAULT, VIDEO_HINT_MODE_DEFAULT } from "../direction-registry.js"
|
|
18
|
+
import { renderStyleSection } from "../prompt-style-section.js"
|
|
18
19
|
import {
|
|
19
20
|
renderSubjectHints,
|
|
20
21
|
SUBJECT_IMAGE_HINT_MODE_DEFAULT,
|
|
@@ -95,12 +96,15 @@ describe("the no-subject oracle — the fold is dark until a caller opts in", ()
|
|
|
95
96
|
})
|
|
96
97
|
|
|
97
98
|
it("matches an independent recomputation of the direction-only fold", () => {
|
|
99
|
+
// Every image-surface direction row is `look`, so a direction-only fold
|
|
100
|
+
// adds NOTHING to the body: the prompt is trimmed (something folded) and
|
|
101
|
+
// the whole fold reads in the section.
|
|
98
102
|
const userPrompt = "a knight on a hill"
|
|
99
103
|
const expected = buildImagePrompt({
|
|
100
|
-
prompt:
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
)
|
|
104
|
+
prompt: `${userPrompt}\n\n${renderStyleSection(DIRECTION, {
|
|
105
|
+
surface: "image",
|
|
106
|
+
mode: IMAGE_HINT_MODE_DEFAULT,
|
|
107
|
+
})}`,
|
|
104
108
|
provider: "nano-banana",
|
|
105
109
|
})
|
|
106
110
|
expect(assembleImageInput({ userPrompt, provider: "nano-banana", direction: DIRECTION })).toEqual(
|
|
@@ -121,7 +125,7 @@ describe("the no-subject oracle — the fold is dark until a caller opts in", ()
|
|
|
121
125
|
})
|
|
122
126
|
|
|
123
127
|
describe("the subject fold — position and content", () => {
|
|
124
|
-
it("lands the subject clauses
|
|
128
|
+
it("lands the subject clauses in the BODY, ahead of the direction section", () => {
|
|
125
129
|
const userPrompt = "on the seawall"
|
|
126
130
|
const subjectHints = renderSubjectHints(SUBJECT, {
|
|
127
131
|
surface: "image",
|
|
@@ -137,9 +141,14 @@ describe("the subject fold — position and content", () => {
|
|
|
137
141
|
subject: SUBJECT,
|
|
138
142
|
direction: DIRECTION,
|
|
139
143
|
})
|
|
144
|
+
// The subject is the noun phrase the look modifies, so it stays inline with
|
|
145
|
+
// the prose; the direction fold lifts out behind it.
|
|
140
146
|
expect(result.prompt).toBe(
|
|
141
147
|
buildImagePrompt({
|
|
142
|
-
prompt: joinPromptHints(userPrompt,
|
|
148
|
+
prompt: `${joinPromptHints(userPrompt, subjectHints)}\n\n${renderStyleSection(DIRECTION, {
|
|
149
|
+
surface: "image",
|
|
150
|
+
mode: IMAGE_HINT_MODE_DEFAULT,
|
|
151
|
+
})}`,
|
|
143
152
|
provider: "nano-banana",
|
|
144
153
|
}).prompt,
|
|
145
154
|
)
|
|
@@ -161,16 +170,19 @@ describe("the subject fold — position and content", () => {
|
|
|
161
170
|
expect(sentences[1]).toContain(", ")
|
|
162
171
|
})
|
|
163
172
|
|
|
164
|
-
it("folds the subject on the video surface, compact,
|
|
173
|
+
it("folds the subject on the video surface, compact, in the body", () => {
|
|
165
174
|
const composed = composeVideoPromptText("she walks", DIRECTION, undefined, { subject: SUBJECT })
|
|
166
175
|
expect(composed).toBe(
|
|
167
|
-
joinPromptHints(
|
|
168
|
-
|
|
176
|
+
`${joinPromptHints(
|
|
177
|
+
"she walks",
|
|
178
|
+
renderSubjectHints(SUBJECT, {
|
|
169
179
|
surface: "video",
|
|
170
180
|
mode: SUBJECT_VIDEO_HINT_MODE_DEFAULT,
|
|
171
181
|
}),
|
|
172
|
-
|
|
173
|
-
|
|
182
|
+
)}\n\n${renderStyleSection(DIRECTION, {
|
|
183
|
+
surface: "video",
|
|
184
|
+
mode: VIDEO_HINT_MODE_DEFAULT,
|
|
185
|
+
})}`,
|
|
174
186
|
)
|
|
175
187
|
})
|
|
176
188
|
|
|
@@ -212,9 +224,16 @@ describe("the subject fold — under the provider cap", () => {
|
|
|
212
224
|
surface: "image",
|
|
213
225
|
mode: IMAGE_HINT_MODE_DEFAULT,
|
|
214
226
|
})
|
|
215
|
-
// Non-vacuity: the un-shed fold really does overflow.
|
|
227
|
+
// Non-vacuity: the un-shed fold really does overflow. Measured through a
|
|
228
|
+
// high-cap provider so the oracle is the assembler's own composition, not a
|
|
229
|
+
// hand-rebuilt approximation of it.
|
|
216
230
|
expect(
|
|
217
|
-
|
|
231
|
+
assembleImageInput({
|
|
232
|
+
userPrompt: PROSE,
|
|
233
|
+
provider: "nano-banana-pro",
|
|
234
|
+
subject: SUBJECT,
|
|
235
|
+
direction: BIG_DIRECTION,
|
|
236
|
+
}).prompt.length,
|
|
218
237
|
).toBeGreaterThan(SEEDREAM_CAP)
|
|
219
238
|
|
|
220
239
|
const result = assembleImageInput({
|