@nodaro/prompts 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +140 -5
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +214 -1
- package/dist/index.d.ts +214 -1
- package/dist/index.js +134 -6
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
- package/src/__tests__/assemble-suno-input.test.ts +20 -0
- package/src/__tests__/location-convergence-image.test.ts +6 -2
- package/src/__tests__/reference-rules.test.ts +240 -0
- package/src/__tests__/video-reference-features.test.ts +1 -0
- package/src/assemble-suno-input.ts +8 -0
- package/src/factory-presets/generate-image.ts +48 -0
- package/src/factory-snippets/catalog.ts +19 -1
- package/src/index.ts +1 -0
- package/src/prompt-wizard-categories.ts +4 -0
- package/src/provider-prompt-doctrine.ts +52 -2
- package/src/reference-rules.ts +226 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nodaro/prompts",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"description": "Nodaro's prompt-engineering layer — person/picker catalogs with prompt hints, identity-lock clauses, entity prompt builders, brand presets, and prompt/reference assembly shared by the Nodaro platform and SDK.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "FSL-1.1-Apache-2.0",
|
|
@@ -179,6 +179,26 @@ describe("assembleSunoInput — persona spread", () => {
|
|
|
179
179
|
})
|
|
180
180
|
})
|
|
181
181
|
|
|
182
|
+
describe("assembleSunoInput — duration pass-through", () => {
|
|
183
|
+
it("carries data.duration onto the result verbatim (the provider client gates the send)", () => {
|
|
184
|
+
const r = assembleSunoInput({
|
|
185
|
+
node: node({ customMode: true, style: "pop", model: "V5_5", duration: 120 }),
|
|
186
|
+
graph: emptyGraph,
|
|
187
|
+
userPrompt: "song",
|
|
188
|
+
})
|
|
189
|
+
expect(r.duration).toBe(120)
|
|
190
|
+
})
|
|
191
|
+
|
|
192
|
+
it("no data.duration → undefined", () => {
|
|
193
|
+
const r = assembleSunoInput({
|
|
194
|
+
node: node({ model: "V5_5" }),
|
|
195
|
+
graph: emptyGraph,
|
|
196
|
+
userPrompt: "song",
|
|
197
|
+
})
|
|
198
|
+
expect(r.duration).toBeUndefined()
|
|
199
|
+
})
|
|
200
|
+
})
|
|
201
|
+
|
|
182
202
|
describe("assembleSunoInput — || undefined normalization (divergence E)", () => {
|
|
183
203
|
it("empty model/style/title/negativeStyle normalize to undefined", () => {
|
|
184
204
|
const r = assembleSunoInput({
|
|
@@ -10,12 +10,16 @@ const library: ConnectedReference = {
|
|
|
10
10
|
}
|
|
11
11
|
|
|
12
12
|
describe("location reference converges onto the image hybrid form", () => {
|
|
13
|
-
|
|
13
|
+
// The DEFAULT role for a wired location became "location" (2026-08-05) — a
|
|
14
|
+
// place, not a backdrop to paste. This assertion previously pinned "the
|
|
15
|
+
// background from …", the wording measured to produce cut-out composites.
|
|
16
|
+
// The explicit `:background` token below is unaffected and still renders it.
|
|
17
|
+
it("wired location (canonical) → 'the location from reference image A', no legacy block", () => {
|
|
14
18
|
const out = buildImagePrompt({
|
|
15
19
|
prompt: "a detective at her desk", connectedReferences: [library],
|
|
16
20
|
provider: "nano-banana-pro", referenceFormat: "hybrid",
|
|
17
21
|
})
|
|
18
|
-
expect(out.prompt).toContain("the
|
|
22
|
+
expect(out.prompt).toContain("the location from reference image A")
|
|
19
23
|
expect(out.prompt).not.toContain("Use these locations:")
|
|
20
24
|
expect(out.referenceImageUrls).toContain("https://cdn/library.png")
|
|
21
25
|
})
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
import { describe, it, expect } from "vitest"
|
|
2
|
+
import {
|
|
3
|
+
REFERENCE_RULES,
|
|
4
|
+
REFERENCE_RULES_MULTI_PERSON,
|
|
5
|
+
SCENE_FRAME_RULE,
|
|
6
|
+
FILM_STILL_PREFIX,
|
|
7
|
+
CINEMATIC_LOOK_TAIL,
|
|
8
|
+
referenceRulesBlock,
|
|
9
|
+
} from "../reference-rules.js"
|
|
10
|
+
import { FACTORY_SNIPPETS } from "../factory-snippets/catalog.js"
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* These pin the OUTCOME of a measurement, not a preference — 36 draws on
|
|
14
|
+
* gpt-image-2 against a four-reference brief with wardrobe swapped between two
|
|
15
|
+
* people (see reference-rules.ts for the arms and the counts). Changing any of
|
|
16
|
+
* these strings without re-running that comparison is how the two wordings
|
|
17
|
+
* drifted apart in the first place.
|
|
18
|
+
*/
|
|
19
|
+
describe("REFERENCE_RULES is the short block — the default follows the population", () => {
|
|
20
|
+
it("keeps the default-deny, the likeness rule and the compose clause", () => {
|
|
21
|
+
expect(REFERENCE_RULES).toContain("Do not use anything from reference images unless specified explicitly.")
|
|
22
|
+
expect(REFERENCE_RULES).toContain("All elements taken from reference images must preserve likeness.")
|
|
23
|
+
expect(REFERENCE_RULES).toContain("Compose them naturally into a single image.")
|
|
24
|
+
})
|
|
25
|
+
|
|
26
|
+
it("carries NO face clauses — they are dead weight on a product or a landscape", () => {
|
|
27
|
+
// The default lands on every brief. Tal's volume of real jobs says the
|
|
28
|
+
// short block wins in general; the controlled brief that favoured the face
|
|
29
|
+
// clauses had two faces swapping a garment, which is what MULTI_PERSON is
|
|
30
|
+
// for. Both results stand; this is the one that has to be safe everywhere.
|
|
31
|
+
expect(REFERENCE_RULES).not.toContain("face structure")
|
|
32
|
+
expect(REFERENCE_RULES).not.toContain("blend faces")
|
|
33
|
+
})
|
|
34
|
+
|
|
35
|
+
it("MULTI_PERSON adds exactly the two face clauses and nothing else", () => {
|
|
36
|
+
expect(REFERENCE_RULES_MULTI_PERSON).toContain("Do not alter face structure.")
|
|
37
|
+
expect(REFERENCE_RULES_MULTI_PERSON).toContain("Do not blend faces.")
|
|
38
|
+
expect(REFERENCE_RULES_MULTI_PERSON).toContain("Compose them naturally into a single image.")
|
|
39
|
+
})
|
|
40
|
+
|
|
41
|
+
it("neither block carries the performance clause — a composite has no scene to perform", () => {
|
|
42
|
+
// That clause belongs to gvp's scene lane, where a subject acts a beat.
|
|
43
|
+
// Here it replaced the compose clause and scored 1/4 against 4/4.
|
|
44
|
+
for (const block of [REFERENCE_RULES, REFERENCE_RULES_MULTI_PERSON]) {
|
|
45
|
+
expect(block).not.toContain("Expression, gaze and pose follow the scene")
|
|
46
|
+
}
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
it("does not claim a medium, a genre or a mood", () => {
|
|
50
|
+
// Every arm that described what the picture IS cost reference fidelity —
|
|
51
|
+
// "scene start frame of a video" lost the lead's identity in 3 of 3.
|
|
52
|
+
for (const banned of ["film", "video", "cinematic", "candid", "photo"]) {
|
|
53
|
+
expect(REFERENCE_RULES.toLowerCase()).not.toContain(banned)
|
|
54
|
+
expect(REFERENCE_RULES_MULTI_PERSON.toLowerCase()).not.toContain(banned)
|
|
55
|
+
}
|
|
56
|
+
})
|
|
57
|
+
})
|
|
58
|
+
|
|
59
|
+
describe("SCENE_FRAME_RULE is the short negative, and stays short", () => {
|
|
60
|
+
it("is exactly the sentence that measured free", () => {
|
|
61
|
+
expect(SCENE_FRAME_RULE).toBe("Nobody looks at the camera.")
|
|
62
|
+
})
|
|
63
|
+
|
|
64
|
+
it("constrains the eyeline and nothing else", () => {
|
|
65
|
+
// The longer phrasings ("a candid moment, unposed, nobody aware of the
|
|
66
|
+
// camera") fixed the gaze and lost a face. Length IS the failure mode.
|
|
67
|
+
expect(SCENE_FRAME_RULE.split(" ")).toHaveLength(5)
|
|
68
|
+
for (const banned of ["film", "cinematic", "candid", "unposed", "movie"]) {
|
|
69
|
+
expect(SCENE_FRAME_RULE.toLowerCase()).not.toContain(banned)
|
|
70
|
+
}
|
|
71
|
+
})
|
|
72
|
+
})
|
|
73
|
+
|
|
74
|
+
describe("referenceRulesBlock resolves the two toggles independently", () => {
|
|
75
|
+
it("defaults to the rules alone — the eyeline is a creative choice, not a rule", () => {
|
|
76
|
+
expect(referenceRulesBlock()).toBe(REFERENCE_RULES)
|
|
77
|
+
expect(referenceRulesBlock({})).toBe(REFERENCE_RULES)
|
|
78
|
+
expect(referenceRulesBlock()).not.toContain(SCENE_FRAME_RULE)
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
it("adds the eyeline rule only when asked", () => {
|
|
82
|
+
const both = referenceRulesBlock({ sceneFrame: true })
|
|
83
|
+
expect(both).toContain(REFERENCE_RULES)
|
|
84
|
+
expect(both).toContain(SCENE_FRAME_RULE)
|
|
85
|
+
expect(both.indexOf(REFERENCE_RULES)).toBeLessThan(both.indexOf(SCENE_FRAME_RULE))
|
|
86
|
+
})
|
|
87
|
+
|
|
88
|
+
it("returns EMPTY when everything is off, so a caller can prepend blind", () => {
|
|
89
|
+
expect(referenceRulesBlock({ referenceRules: false })).toBe("")
|
|
90
|
+
expect(referenceRulesBlock({ referenceRules: false, sceneFrame: false })).toBe("")
|
|
91
|
+
})
|
|
92
|
+
|
|
93
|
+
it("can emit the eyeline rule alone", () => {
|
|
94
|
+
expect(referenceRulesBlock({ referenceRules: false, sceneFrame: true })).toBe(SCENE_FRAME_RULE)
|
|
95
|
+
})
|
|
96
|
+
})
|
|
97
|
+
|
|
98
|
+
describe("the snippet catalog and the injected default cannot drift", () => {
|
|
99
|
+
const byId = (id: string) => FACTORY_SNIPPETS.find((s) => s.id === id)
|
|
100
|
+
|
|
101
|
+
it("Reference Lock IS the default constant, not a hand-kept twin", () => {
|
|
102
|
+
// The bug this closes: the catalog carried its own wording, written without
|
|
103
|
+
// knowledge of gvp's, and scored 0/4 where the merged block scored 4/4.
|
|
104
|
+
expect(byId("reference-lock")?.text).toBe(REFERENCE_RULES)
|
|
105
|
+
})
|
|
106
|
+
|
|
107
|
+
it("Scene Frame IS the constant, and is a SEPARATE entry", () => {
|
|
108
|
+
expect(byId("scene-frame")?.text).toBe(SCENE_FRAME_RULE)
|
|
109
|
+
expect(byId("reference-lock")?.text).not.toContain(SCENE_FRAME_RULE)
|
|
110
|
+
})
|
|
111
|
+
|
|
112
|
+
it("both sit in the Reference locks category, for images", () => {
|
|
113
|
+
for (const id of ["reference-lock", "scene-frame"]) {
|
|
114
|
+
expect(byId(id)?.category).toBe("Reference locks")
|
|
115
|
+
expect(byId(id)?.target).toBe("prompt")
|
|
116
|
+
}
|
|
117
|
+
})
|
|
118
|
+
})
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* END TO END through the assembler, because the block is only worth anything
|
|
122
|
+
* if it arrives AHEAD of the lettered scene it talks about. "Reference image A"
|
|
123
|
+
* is a hybrid phrase — rules prepended to a legacy assembly would be telling
|
|
124
|
+
* the model to obey bindings that were never lettered.
|
|
125
|
+
*/
|
|
126
|
+
describe("the block reaches the prompt ahead of the scene it governs", () => {
|
|
127
|
+
const ref = (id: string, url: string) =>
|
|
128
|
+
({ id, defaultName: id, source: "manual" as const, url })
|
|
129
|
+
|
|
130
|
+
it("prepends the measured wording, then the lettered scene", async () => {
|
|
131
|
+
const { buildImagePrompt } = await import("../prompt-builder.js")
|
|
132
|
+
const { prompt } = buildImagePrompt({
|
|
133
|
+
provider: "gpt-image-2",
|
|
134
|
+
prompt: "{image:1:person} wears {image:2:clothes}.",
|
|
135
|
+
connectedReferences: [ref("a", "https://r2/a.png"), ref("b", "https://r2/b.png")],
|
|
136
|
+
referenceFormat: "hybrid",
|
|
137
|
+
referenceLockSnippet: referenceRulesBlock({ sceneFrame: true }),
|
|
138
|
+
})
|
|
139
|
+
expect(prompt.startsWith(REFERENCE_RULES)).toBe(true)
|
|
140
|
+
expect(prompt).toContain(SCENE_FRAME_RULE)
|
|
141
|
+
// The rules come FIRST; the bindings they govern come after.
|
|
142
|
+
expect(prompt.indexOf(SCENE_FRAME_RULE)).toBeLessThan(prompt.indexOf("reference image A"))
|
|
143
|
+
// Line-initial is capitalized by the hybrid scene builder.
|
|
144
|
+
expect(prompt).toContain("The person from reference image A wears the clothes from reference image B.")
|
|
145
|
+
})
|
|
146
|
+
|
|
147
|
+
it("leaves the prompt untouched when the caller turned both off", async () => {
|
|
148
|
+
const { buildImagePrompt } = await import("../prompt-builder.js")
|
|
149
|
+
const snippet = referenceRulesBlock({ referenceRules: false })
|
|
150
|
+
const { prompt } = buildImagePrompt({
|
|
151
|
+
provider: "gpt-image-2",
|
|
152
|
+
prompt: "{image:1:person} sits.",
|
|
153
|
+
connectedReferences: [ref("a", "https://r2/a.png")],
|
|
154
|
+
referenceFormat: "hybrid",
|
|
155
|
+
...(snippet ? { referenceLockSnippet: snippet } : {}),
|
|
156
|
+
})
|
|
157
|
+
expect(prompt).not.toContain("Do not take anything")
|
|
158
|
+
expect(prompt).not.toContain("Nobody looks at the camera")
|
|
159
|
+
})
|
|
160
|
+
})
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* THE ONE RULE ALL OF TONIGHT'S ARMS TURNED OUT TO BE: position decides whether
|
|
165
|
+
* an instruction helps or fights the reference bindings.
|
|
166
|
+
*
|
|
167
|
+
* rules FIRST · framing as a PREFIX · look LAST · never a claim in the middle
|
|
168
|
+
*
|
|
169
|
+
* "This image is a scene start frame of a video." is the middle-claim shape and
|
|
170
|
+
* it cost the lead's identity in 3 of 3 draws. "Film still of" is the prefix
|
|
171
|
+
* shape and costs nothing. The look tail is the last-position shape and costs
|
|
172
|
+
* nothing. gvp found the last one independently: "THE MEDIUM GOES LAST".
|
|
173
|
+
*/
|
|
174
|
+
describe("the framing prefix and the look tail keep their shapes", () => {
|
|
175
|
+
it("the framing snippet is a PREFIX, not a sentence", () => {
|
|
176
|
+
// A fragment, so it swallows the scene that follows. A full stop here would
|
|
177
|
+
// make it a standalone claim — the shape that lost the references.
|
|
178
|
+
expect(FILM_STILL_PREFIX).toBe("Film still of")
|
|
179
|
+
expect(FILM_STILL_PREFIX.endsWith(".")).toBe(false)
|
|
180
|
+
})
|
|
181
|
+
|
|
182
|
+
it("the look tail names stock, lens, light and palette", () => {
|
|
183
|
+
for (const part of ["16mm", "lenses", "light", "palette"]) {
|
|
184
|
+
expect(CINEMATIC_LOOK_TAIL.toLowerCase()).toContain(part)
|
|
185
|
+
}
|
|
186
|
+
// It is a tail — no trailing full stop, nothing after it to argue with.
|
|
187
|
+
expect(CINEMATIC_LOOK_TAIL.endsWith(".")).toBe(false)
|
|
188
|
+
})
|
|
189
|
+
})
|
|
190
|
+
|
|
191
|
+
describe("multiPerson swaps in the face clauses without touching anything else", () => {
|
|
192
|
+
it("is off by default", () => {
|
|
193
|
+
expect(referenceRulesBlock()).toBe(REFERENCE_RULES)
|
|
194
|
+
expect(referenceRulesBlock()).not.toContain("blend faces")
|
|
195
|
+
})
|
|
196
|
+
|
|
197
|
+
it("opts into the measured composition block", () => {
|
|
198
|
+
expect(referenceRulesBlock({ multiPerson: true })).toBe(REFERENCE_RULES_MULTI_PERSON)
|
|
199
|
+
})
|
|
200
|
+
|
|
201
|
+
it("still composes with the eyeline rule", () => {
|
|
202
|
+
const both = referenceRulesBlock({ multiPerson: true, sceneFrame: true })
|
|
203
|
+
expect(both).toContain(REFERENCE_RULES_MULTI_PERSON)
|
|
204
|
+
expect(both).toContain(SCENE_FRAME_RULE)
|
|
205
|
+
})
|
|
206
|
+
|
|
207
|
+
it("stays silent when the rules are off, whatever multiPerson says", () => {
|
|
208
|
+
expect(referenceRulesBlock({ referenceRules: false, multiPerson: true })).toBe("")
|
|
209
|
+
})
|
|
210
|
+
})
|
|
211
|
+
|
|
212
|
+
describe("filmStillPrefix leads with the shot size", () => {
|
|
213
|
+
it("puts the framing first, then the fixed tail", async () => {
|
|
214
|
+
const { filmStillPrefix } = await import("../reference-rules.js")
|
|
215
|
+
expect(filmStillPrefix("Medium wide")).toBe("Medium wide film still of")
|
|
216
|
+
expect(filmStillPrefix("Extreme wide")).toBe("Extreme wide film still of")
|
|
217
|
+
})
|
|
218
|
+
|
|
219
|
+
it("falls back to the bare prefix when no shot is given", async () => {
|
|
220
|
+
const { filmStillPrefix } = await import("../reference-rules.js")
|
|
221
|
+
expect(filmStillPrefix()).toBe(FILM_STILL_PREFIX)
|
|
222
|
+
expect(filmStillPrefix(" ")).toBe(FILM_STILL_PREFIX)
|
|
223
|
+
})
|
|
224
|
+
|
|
225
|
+
it("claims no genre — a UGC clip and a documentary are film stills too", async () => {
|
|
226
|
+
const { filmStillPrefix } = await import("../reference-rules.js")
|
|
227
|
+
// "Cinematic" imposes a register. It is also the exact category of word
|
|
228
|
+
// every measured arm punished, so it does not belong in a default.
|
|
229
|
+
for (const shot of [undefined, "Medium wide"]) {
|
|
230
|
+
expect(filmStillPrefix(shot).toLowerCase()).not.toContain("cinematic")
|
|
231
|
+
}
|
|
232
|
+
})
|
|
233
|
+
|
|
234
|
+
it("never ends in a full stop — it must swallow the scene, not stand alone", async () => {
|
|
235
|
+
const { filmStillPrefix } = await import("../reference-rules.js")
|
|
236
|
+
for (const shot of [undefined, "Extreme wide", "Close-up"]) {
|
|
237
|
+
expect(filmStillPrefix(shot).endsWith(".")).toBe(false)
|
|
238
|
+
}
|
|
239
|
+
})
|
|
240
|
+
})
|
|
@@ -84,6 +84,13 @@ export interface AssembleSunoResult {
|
|
|
84
84
|
customMode: boolean
|
|
85
85
|
instrumental: boolean
|
|
86
86
|
model?: string
|
|
87
|
+
/**
|
|
88
|
+
* Requested song length in seconds (KIE: 10–360). Passed through from
|
|
89
|
+
* `data.duration` unconditionally — the provider client is the single
|
|
90
|
+
* gate that only sends it when customMode && model V5_5 (KIE ignores it
|
|
91
|
+
* elsewhere), so the assembler stays a faithful field carrier.
|
|
92
|
+
*/
|
|
93
|
+
duration?: number
|
|
87
94
|
personaId?: string
|
|
88
95
|
personaModel?: "voice_persona" | "style_persona"
|
|
89
96
|
}
|
|
@@ -140,6 +147,7 @@ export function assembleSunoInput(input: AssembleSunoInput): AssembleSunoResult
|
|
|
140
147
|
styleWeight: data.styleWeight as number | undefined,
|
|
141
148
|
weirdnessConstraint: data.weirdnessConstraint as number | undefined,
|
|
142
149
|
audioWeight: data.audioWeight as number | undefined,
|
|
150
|
+
duration: data.duration as number | undefined,
|
|
143
151
|
customMode,
|
|
144
152
|
instrumental: (data.instrumental as boolean | undefined) ?? false,
|
|
145
153
|
...(input.persona ?? {}),
|
|
@@ -231,6 +231,54 @@ export const GENERATE_IMAGE_PRESETS: readonly FactoryPreset[] = [
|
|
|
231
231
|
"leftover callout labels, visible text labels, text boxes, leader lines, garbled text, restyled scene, recolored unrelated elements, redrawn image, cartoon, illustration, watermark, blurry",
|
|
232
232
|
},
|
|
233
233
|
},
|
|
234
|
+
// ── Face Privacy (connect the photo → {image:1}) ─────────────────────────
|
|
235
|
+
// gpt-image-2 + aspectRatio "auto" so the output keeps the input photo's
|
|
236
|
+
// shape. NOTE: KIE's gpt-image-2 spec forces resolution 1K when aspect is
|
|
237
|
+
// "auto" (the config panel's fail-safe snaps any other value back), so
|
|
238
|
+
// these pin 1K explicitly rather than shipping a self-contradicting 2K.
|
|
239
|
+
{
|
|
240
|
+
id: "generate-image/faceless-3d-head",
|
|
241
|
+
name: "Faceless · 3D Blank Head",
|
|
242
|
+
description: "Persons' faces → smooth generic white 3D head shape; everything else untouched.",
|
|
243
|
+
group: "Face Privacy",
|
|
244
|
+
data: {
|
|
245
|
+
provider: "gpt-image-2",
|
|
246
|
+
aspectRatio: "auto",
|
|
247
|
+
resolution: "1K",
|
|
248
|
+
prompt:
|
|
249
|
+
"make the persons in {image:1} faceless by making it 3d white generic face shape, keep everything else the same",
|
|
250
|
+
negativePrompt:
|
|
251
|
+
"visible facial features, eyes, nose, mouth, skin texture on the face, changing the background, altering clothing, hair or body, restyling the scene, relighting, changing pose or framing, cartoon look, distorted anatomy, watermark, blurry",
|
|
252
|
+
},
|
|
253
|
+
},
|
|
254
|
+
{
|
|
255
|
+
id: "generate-image/remove-faces",
|
|
256
|
+
name: "Remove Faces",
|
|
257
|
+
description: "Erase persons' faces entirely; everything else untouched.",
|
|
258
|
+
group: "Face Privacy",
|
|
259
|
+
data: {
|
|
260
|
+
provider: "gpt-image-2",
|
|
261
|
+
aspectRatio: "auto",
|
|
262
|
+
resolution: "1K",
|
|
263
|
+
prompt: "remove persons faces from {image:1}, keep everything else the same",
|
|
264
|
+
negativePrompt:
|
|
265
|
+
"visible facial features, eyes, nose, mouth, leftover face fragments, changing the background, altering clothing, hair or body, restyling the scene, relighting, changing pose or framing, horror or gore look, distorted anatomy, watermark, blurry",
|
|
266
|
+
},
|
|
267
|
+
},
|
|
268
|
+
{
|
|
269
|
+
id: "generate-image/transparent-faces",
|
|
270
|
+
name: "Transparent Faces",
|
|
271
|
+
description: "Persons' faces become transparent; everything else untouched.",
|
|
272
|
+
group: "Face Privacy",
|
|
273
|
+
data: {
|
|
274
|
+
provider: "gpt-image-2",
|
|
275
|
+
aspectRatio: "auto",
|
|
276
|
+
resolution: "1K",
|
|
277
|
+
prompt: "make faces transparent in {image:1}, keep everything else the same",
|
|
278
|
+
negativePrompt:
|
|
279
|
+
"opaque faces, visible facial features, eyes, nose, mouth, changing the background, altering clothing, hair or body, restyling the scene, relighting, changing pose or framing, distorted anatomy, watermark, blurry",
|
|
280
|
+
},
|
|
281
|
+
},
|
|
234
282
|
// ── Photography & Cinematic ──────────────────────────────────────────────
|
|
235
283
|
{
|
|
236
284
|
id: "generate-image/cinematic-portrait",
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
/** Factory snippet catalog (v1: image + video; audio/text follow later).
|
|
2
2
|
* Order within a category = menu order = pill quick-cycle order. */
|
|
3
3
|
import type { FactorySnippet } from "./types.js"
|
|
4
|
+
import { REFERENCE_RULES, SCENE_FRAME_RULE, FILM_STILL_PREFIX, CINEMATIC_LOOK_TAIL } from "../reference-rules.js"
|
|
4
5
|
|
|
5
6
|
const B = ["image", "video"] as const
|
|
6
7
|
const I = ["image"] as const
|
|
@@ -15,7 +16,24 @@ export const FACTORY_SNIPPETS: readonly FactorySnippet[] = [
|
|
|
15
16
|
{ id: "no-beautify", name: "No Beautify", description: "Stop the model 'improving' a face", text: "preserve natural skin texture, age lines, and asymmetries; do not beautify, smooth, slim, or rejuvenate the face", target: "prompt", media: I, category: "Identity & Consistency" },
|
|
16
17
|
|
|
17
18
|
// ── Reference locks (prompt) — insert at the START of a reference prompt ──
|
|
18
|
-
|
|
19
|
+
// Text comes from the shared constant, not a hand-written twin. This entry used
|
|
20
|
+
// to carry its own wording — no face rules, likeness phrased as a passive —
|
|
21
|
+
// which scored 0/4 on moving a garment between references where the merged
|
|
22
|
+
// block scored 4/4 (see reference-rules.ts). A snippet is copied into the
|
|
23
|
+
// prompt at INSERT time, so changing it here only affects future insertions.
|
|
24
|
+
{ id: "reference-lock", name: "Reference Lock", description: "Default-deny + preserve likeness + compose into one image (full scenes)", text: REFERENCE_RULES, target: "prompt", media: I, category: "Reference locks" },
|
|
25
|
+
// The eyeline suppressor, kept OUT of Reference Lock on purpose: a portrait
|
|
26
|
+
// or a piece to camera wants the eyeline. 4/4 on a brief where the lock alone
|
|
27
|
+
// was 0/4, at no measured cost to identity or wardrobe.
|
|
28
|
+
{ id: "scene-frame", name: "Scene Frame", description: "Reads as a frame from a film, not a posed photo — nobody faces the lens", text: SCENE_FRAME_RULE, target: "prompt", media: I, category: "Reference locks" },
|
|
29
|
+
// A PREFIX, not a sentence — it swallows the scene that follows it, which is
|
|
30
|
+
// exactly why it costs nothing. The same idea written as a standalone claim
|
|
31
|
+
// ("This image is a scene start frame of a video.") lost the lead's identity
|
|
32
|
+
// in 3 of 3 draws. Goes at the very TOP, above the scene.
|
|
33
|
+
{ id: "film-still-of", name: "Film Still Of", description: "Put at the very top, before the scene — lead with the shot size, e.g. \"Medium wide…\"", text: `Medium wide ${FILM_STILL_PREFIX.charAt(0).toLowerCase()}${FILM_STILL_PREFIX.slice(1)}`, target: "prompt", media: I, category: "Reference locks" },
|
|
34
|
+
// Goes at the very END. An example to edit — a different film wants a
|
|
35
|
+
// different stock; what generalises is that the look comes LAST.
|
|
36
|
+
{ id: "cinematic-look", name: "Cinematic Look (16mm)", description: "Append at the END — film stock, lens, light and palette", text: CINEMATIC_LOOK_TAIL, target: "prompt", media: I, category: "Reference locks" },
|
|
19
37
|
{ id: "reference-extract", name: "Reference Extract", description: "Isolate only the specified elements — no scene, no figure", text: "Take only what is specified from the reference images. Do not take anything else.", target: "prompt", media: I, category: "Reference locks" },
|
|
20
38
|
{ id: "ghost-mannequin", name: "Ghost Mannequin", description: "Show worn garments without a wearer (product shot)", text: "Ghost-mannequin product shot: the garment in its natural worn shape, with no person, body, face, or mannequin visible.", target: "prompt", media: I, category: "Reference locks" },
|
|
21
39
|
|
package/src/index.ts
CHANGED
|
@@ -242,6 +242,8 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
|
|
|
242
242
|
"seedance-2": "Seedance 2.0 — multimodal refs (9 images / 3 videos / 3 audio), native multi-track audio, multi-shot storytelling, 4-15s",
|
|
243
243
|
"seedance-2-fast": "Seedance 2.0 Fast — same multimodal + audio capabilities, cheaper and quicker",
|
|
244
244
|
"seedance-2-mini": "Seedance 2.0 Mini — same multimodal + audio capabilities, budget tier, 480p/720p, 4-15s",
|
|
245
|
+
"seedance-2-5": "Seedance 2.5 — up to 30s in ONE shot (no stitching), wider multimodal refs (30 images / 10 videos / 10 audio), native audio, 480p/720p",
|
|
246
|
+
"minimax-h3": "MiniMax Hailuo 3 — premium multimodal refs (9 images / 3 videos / 3 audio), always-on audio, 2K or 768P, 4-15s per-second pricing",
|
|
245
247
|
"wan": "Versatile, good for animations and transformations",
|
|
246
248
|
"wan-turbo": "Faster Wan generation",
|
|
247
249
|
"hailuo-standard": "Standard quality, cost-effective",
|
|
@@ -268,6 +270,8 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
|
|
|
268
270
|
"seedance-2": "Seedance 2.0 — start/end frame + multimodal refs, native audio, 4-15s",
|
|
269
271
|
"seedance-2-fast": "Seedance 2.0 Fast — same capabilities, cheaper and quicker",
|
|
270
272
|
"seedance-2-mini": "Seedance 2.0 Mini — same capabilities, budget tier, 480p/720p",
|
|
273
|
+
"seedance-2-5": "Seedance 2.5 — start/end frame + wide multimodal refs, native audio, up to 30s, 480p/720p",
|
|
274
|
+
"minimax-h3": "MiniMax Hailuo 3 — first/last frame + multimodal refs, always-on audio, 2K or 768P, 4-15s",
|
|
271
275
|
"hailuo-2.3-pro": "Premium Hailuo animation",
|
|
272
276
|
"hailuo-2.3": "Standard Hailuo animation",
|
|
273
277
|
"hailuo-standard": "Cost-effective animation",
|
|
@@ -19,6 +19,12 @@
|
|
|
19
19
|
* - KIE API docs https://docs.kie.ai/market/kling/kling-3-0
|
|
20
20
|
* - Live KIE-path dialogue probe 2026-07-16 (scripted lines spoken verbatim,
|
|
21
21
|
* lip-synced, on both kling-2.6 and kling-3.0 with sound=true)
|
|
22
|
+
* MiniMax Hailuo 3:
|
|
23
|
+
* - KIE API docs https://docs.kie.ai/market/minimax-h3 (text-to-video /
|
|
24
|
+
* image-to-video / reference-to-video — the API contract is the doctrine
|
|
25
|
+
* source: limits, modes, aspect rules, pricing dimensions). No live probe
|
|
26
|
+
* yet — structure guidance mirrors the platform's ordinal-reference
|
|
27
|
+
* conventions rather than vendor style claims.
|
|
22
28
|
*/
|
|
23
29
|
export interface ProviderPromptDoctrine {
|
|
24
30
|
/** MODEL_CATALOG ids this doctrine covers. */
|
|
@@ -32,14 +38,15 @@ export interface ProviderPromptDoctrine {
|
|
|
32
38
|
}
|
|
33
39
|
|
|
34
40
|
const SEEDANCE_2_DOCTRINE: ProviderPromptDoctrine = {
|
|
35
|
-
providers: ["seedance-2", "seedance-2-fast", "seedance-2-mini"],
|
|
36
|
-
heading: "Seedance 2
|
|
41
|
+
providers: ["seedance-2", "seedance-2-fast", "seedance-2-mini", "seedance-2-5"],
|
|
42
|
+
heading: "Seedance 2 (seedance-2, seedance-2-fast, seedance-2-mini, seedance-2-5)",
|
|
37
43
|
tips: [
|
|
38
44
|
"Storyboard complex videos as 'Shot 1: … Shot 2: …' WITHOUT timestamps — timed shots like '(0-3s)' are officially unstable and can break generation.",
|
|
39
45
|
"One camera movement per shot; describe actions per body part with degree ('slowly raises a hand'); express emotion as physical detail, never abstract words.",
|
|
40
46
|
"Native multi-track audio — cue it inline: (background music), <sound effects>, and quoted dialogue.",
|
|
41
47
|
"References go by ordinal (@Image 1, Video 2) in attachment order; earlier = higher priority. Identity = ONE headshot + ONE full-body (multi-view sheets cause ID drift). 4-5 assets total beats maxing the 9/3/3 caps.",
|
|
42
48
|
"No negative-prompt parameter — put constraints in the prompt: 'keep it subtitle-free, do not generate a watermark, do not generate a logo'.",
|
|
49
|
+
"seedance-2-5 only: one shot runs to 30s (the 2.0 SKUs stop at 15s), so storyboard a whole beat instead of planning a stitch. Ref caps are wider (30/10/10), but 4-5 assets still gives the best identity fidelity.",
|
|
43
50
|
],
|
|
44
51
|
doctrine: `Prompt structure (front-load what matters most):
|
|
45
52
|
precise subject → action details → scene/environment → lighting & color tone → camera movement → visual style → image quality → constraints.
|
|
@@ -51,6 +58,12 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
51
58
|
- Prefer slow, gentle, continuous movements over high-burst action (sprints, big jumps, violent rolls morph). Describe actions per body part with quantified degree: "slowly raises a hand", "pushes hard off the ground". Chain actions with inertia: "uses the momentum of the turn to naturally raise an arm".
|
|
52
59
|
- Express emotion as externalized physical detail, never abstract words: not "very sad" but "lowering the head, shoulders trembling slightly, eyes reddening, fingers clutching the corner of clothing".
|
|
53
60
|
|
|
61
|
+
**Generation differences (seedance-2-5 vs the 2.0 SKUs)**
|
|
62
|
+
- A single 2.5 shot runs to 30s, where every 2.0 SKU stops at 15s. Plan a complete 4-6 shot beat inside ONE generation instead of splitting it into two clips and stitching — no seam to hide, and continuity holds because it never leaves the model.
|
|
63
|
+
- 2.5 also takes far more reference material (30 images / 10 videos / 10 audio vs 9/3/3). Treat that as room for COVERAGE — more distinct characters, locations and props in one shot — not as licence to pile refs onto one identity. The "ONE headshot + ONE full-body, 4-5 assets total" rule above still produces the best likeness on 2.5.
|
|
64
|
+
- 2.5 renders at 480p/720p only: there is no 1080p or 4K tier, so route a job that needs one to seedance-2 (which has both) or upscale afterwards.
|
|
65
|
+
- With a start frame, 2.5 always derives the output aspect from that frame — an explicit aspect ratio is rejected outright, so compose the frame at the ratio you want.
|
|
66
|
+
|
|
54
67
|
**References (when reference media is attached)**
|
|
55
68
|
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync.)
|
|
56
69
|
- Define each subject once, then reuse the label consistently: 'Define the woman in the red dress in Image 1 as the courier' … 'the courier opens the door'. In multi-character scenes bind every character to its image ("the man from Image 1 hands the box to the woman from Image 2") and append: "do not generate duplicate copies of the same character".
|
|
@@ -111,9 +124,46 @@ const KLING_AUDIO_DOCTRINE: ProviderPromptDoctrine = {
|
|
|
111
124
|
- Durations: 2.6 = 5/10s; 3.0/omni = 3-15s. A spoken line needs roughly 1s per 2-3 words — don't script more dialogue than the clip can hold.`,
|
|
112
125
|
}
|
|
113
126
|
|
|
127
|
+
const MINIMAX_H3_DOCTRINE: ProviderPromptDoctrine = {
|
|
128
|
+
providers: ["minimax-h3"],
|
|
129
|
+
heading: "MiniMax Hailuo 3 (minimax-h3)",
|
|
130
|
+
tips: [
|
|
131
|
+
"Natural-language prompts up to 7000 chars. Front-load what matters: subject → action → scene/environment → lighting → camera move → style. One camera movement per shot.",
|
|
132
|
+
"Output 2K (default) or 768P (cheaper tier). Aspect: pure text-to-video needs a concrete ratio (21:9/16:9/4:3/1:1/3:4/9:16); reference runs default to adaptive (match the input).",
|
|
133
|
+
"References go by ordinal in attachment order (@Image 1, Video 1) — earlier = higher priority. Caps: 9 images, 3 videos (2-15s each, ≤15s total), 3 audio clips (≤15s total).",
|
|
134
|
+
"Reference audio drives speech/lip-sync but never rides alone — pair it with an image or video reference. Audio input is free; the first 5 input images are free, extras bill per image.",
|
|
135
|
+
"No negative-prompt parameter — put constraints in the prompt text: 'keep it subtitle-free, do not generate a watermark, do not generate a logo'.",
|
|
136
|
+
],
|
|
137
|
+
doctrine: `Prompt structure (front-load what matters most):
|
|
138
|
+
precise subject → action details → scene/environment → lighting & color tone → camera movement → visual style → image quality → constraints. Prompts are natural language, 1-7000 characters, across all three modes.
|
|
139
|
+
|
|
140
|
+
**Modes (picked automatically from the wired inputs)**
|
|
141
|
+
- First frame and/or last frame connected, nothing else → exact frame mode (image-to-video): the output opens on the first frame and/or closes on the last. The clip's aspect is inferred from the frame — there is no aspect parameter in this mode.
|
|
142
|
+
- ANY reference connected (image, video, or audio) → reference mode (reference-to-video): frames ride along as reference images with a prompt directive binding them to the opening/closing position. Aspect defaults to adaptive (matches the input); a concrete ratio can be forced.
|
|
143
|
+
- Nothing visual connected → text-to-video. A concrete aspect ratio is required (21:9 / 16:9 / 4:3 / 1:1 / 3:4 / 9:16 — no adaptive); Nodaro renders 16:9 unless one is picked.
|
|
144
|
+
|
|
145
|
+
**References (when reference media is attached)**
|
|
146
|
+
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync.)
|
|
147
|
+
- Caps: 9 reference images; 3 reference videos, each 2-15s and ≤15s combined; 3 reference audio clips, ≤15s combined. Reference audio cannot be used alone — it must accompany an image or video reference.
|
|
148
|
+
- Define each subject once, then reuse the label consistently ("the woman from @Image 1 … the woman opens the door"). A focused set of 4-5 assets beats maxing every cap.
|
|
149
|
+
- Billing note: generated seconds AND reference-video input seconds bill at the same per-second rate; the first 5 input images are free and each extra image adds a small surcharge; audio input is free.
|
|
150
|
+
|
|
151
|
+
**Audio**
|
|
152
|
+
- Audio is always generated — there is no on/off toggle. With reference audio attached, the model syncs speech to the supplied track (the platform's lip-sync surface routes image + voice line through this mode automatically).
|
|
153
|
+
- Quoted dialogue in the prompt gives the model the line to perform; describe the voice in words when no reference audio is supplied.
|
|
154
|
+
|
|
155
|
+
**Duration & pacing**
|
|
156
|
+
- 4-15 seconds, integer, default 6. Per-second pricing — a 15s clip costs ~3.7× a 4s clip, so pick the shortest duration that serves the shot.
|
|
157
|
+
- One camera movement type per shot; chain actions with physical, quantified detail ("slowly raises a hand", "pushes hard off the ground") rather than abstract emotion words.
|
|
158
|
+
|
|
159
|
+
**Constraints**
|
|
160
|
+
- There is NO negative-prompt parameter — all constraints belong in the prompt text itself: "keep it subtitle-free, do not generate a watermark, do not generate a logo, stable picture".`,
|
|
161
|
+
}
|
|
162
|
+
|
|
114
163
|
export const PROVIDER_PROMPT_DOCTRINES: readonly ProviderPromptDoctrine[] = [
|
|
115
164
|
SEEDANCE_2_DOCTRINE,
|
|
116
165
|
KLING_AUDIO_DOCTRINE,
|
|
166
|
+
MINIMAX_H3_DOCTRINE,
|
|
117
167
|
]
|
|
118
168
|
|
|
119
169
|
const DOCTRINE_BY_PROVIDER: ReadonlyMap<string, ProviderPromptDoctrine> = new Map(
|