@nodaro/prompts 1.7.0 → 1.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +83 -9
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +128 -4
- package/dist/index.d.ts +128 -4
- package/dist/index.js +81 -11
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
- package/src/__tests__/gemini-omni-inputs.test.ts +68 -0
- package/src/__tests__/seedance-2-inputs.test.ts +48 -0
- package/src/__tests__/veo-i2v-inputs.test.ts +58 -0
- package/src/gemini-omni-inputs.ts +73 -0
- package/src/index.ts +3 -0
- package/src/picker-wiring.ts +1 -1
- package/src/prompt-wizard-categories.ts +3 -2
- package/src/provider-prompt-doctrine.ts +2 -3
- package/src/resolve-prompt.ts +12 -1
- package/src/seedance-2-inputs.ts +10 -3
- package/src/style-presets.ts +1 -1
- package/src/surround-fill.ts +67 -0
- package/src/veo-i2v-inputs.ts +68 -0
- package/src/video-reference-resolver.ts +12 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nodaro/prompts",
|
|
3
|
-
"version": "1.7.
|
|
3
|
+
"version": "1.7.3",
|
|
4
4
|
"description": "Nodaro's prompt-engineering layer — person/picker catalogs with prompt hints, identity-lock clauses, entity prompt builders, brand presets, and prompt/reference assembly shared by the Nodaro platform and SDK.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "FSL-1.1-Apache-2.0",
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import { describe, it, expect } from "vitest"
|
|
2
|
+
|
|
3
|
+
import { resolveGeminiOmniI2vInputs } from "../gemini-omni-inputs.js"
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Gemini Omni i2v input resolution — the flat `image_urls` sibling of the
|
|
7
|
+
* seedance-2 resolver. The stakes: an unbound image list reads as loose
|
|
8
|
+
* context to a multimodal model (field finding 2026-08-14 — identity refs
|
|
9
|
+
* rode every keyframes call and the cast still drifted), and an unbudgeted
|
|
10
|
+
* list trips KIE's 7-input hard reject.
|
|
11
|
+
*/
|
|
12
|
+
describe("resolveGeminiOmniI2vInputs", () => {
|
|
13
|
+
const FIRST = "https://r2/anchor.png"
|
|
14
|
+
const refs = (n: number) => Array.from({ length: n }, (_, i) => `https://r2/ref-${i + 1}.png`)
|
|
15
|
+
|
|
16
|
+
it("binds the roles: image 1 is the opening frame, the rest are identities — not frames", () => {
|
|
17
|
+
const r = resolveGeminiOmniI2vInputs({ prompt: "a walk on the beach", firstFrameUrl: FIRST, refImageUrls: refs(3) })
|
|
18
|
+
expect(r.imageUrls).toEqual([FIRST, ...refs(3)])
|
|
19
|
+
expect(r.promptSuffix).toBe(
|
|
20
|
+
"Use @image_1 as the opening (first) frame of the video. " +
|
|
21
|
+
"@image_2 through @image_4 are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.",
|
|
22
|
+
)
|
|
23
|
+
expect(r.droppedRefImages).toBe(0)
|
|
24
|
+
})
|
|
25
|
+
|
|
26
|
+
it("a single reference gets the singular sentence", () => {
|
|
27
|
+
const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(1) })
|
|
28
|
+
expect(r.promptSuffix).toContain("@image_2 is an identity reference")
|
|
29
|
+
expect(r.promptSuffix).not.toContain("through")
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
it("no references ⇒ byte-identical plain i2v: single image, no suffix", () => {
|
|
33
|
+
const r = resolveGeminiOmniI2vInputs({ prompt: "p", firstFrameUrl: FIRST })
|
|
34
|
+
expect(r).toEqual({ imageUrls: [FIRST], promptSuffix: "", droppedRefImages: 0 })
|
|
35
|
+
})
|
|
36
|
+
|
|
37
|
+
it("drops TRAILING references to fit the 7-input quota — the start frame is never the one that goes", () => {
|
|
38
|
+
const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(9) })
|
|
39
|
+
expect(r.imageUrls).toHaveLength(7)
|
|
40
|
+
expect(r.imageUrls[0]).toBe(FIRST)
|
|
41
|
+
expect(r.imageUrls.at(-1)).toBe("https://r2/ref-6.png")
|
|
42
|
+
expect(r.droppedRefImages).toBe(3)
|
|
43
|
+
// The binding names exactly the kept span.
|
|
44
|
+
expect(r.promptSuffix).toContain("@image_2 through @image_7")
|
|
45
|
+
})
|
|
46
|
+
|
|
47
|
+
it("a connected source video eats two slots (images + 2×videos ≤ 7)", () => {
|
|
48
|
+
const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(9), videoConnected: true })
|
|
49
|
+
expect(r.imageUrls).toHaveLength(5)
|
|
50
|
+
expect(r.droppedRefImages).toBe(5)
|
|
51
|
+
})
|
|
52
|
+
|
|
53
|
+
it("suppresses the opening-frame sentence when the prompt already binds it, keeping the identity sentence", () => {
|
|
54
|
+
const r = resolveGeminiOmniI2vInputs({
|
|
55
|
+
prompt: "use @image_1 as the first frame, it is the last keyframe of @video_1",
|
|
56
|
+
firstFrameUrl: FIRST,
|
|
57
|
+
refImageUrls: refs(2),
|
|
58
|
+
})
|
|
59
|
+
expect(r.promptSuffix).not.toContain("opening (first) frame")
|
|
60
|
+
expect(r.promptSuffix).toContain("identity references")
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
it("skips empty/undefined reference entries without burning slots", () => {
|
|
64
|
+
const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: [undefined, "", ...refs(2)] })
|
|
65
|
+
expect(r.imageUrls).toEqual([FIRST, ...refs(2)])
|
|
66
|
+
expect(r.droppedRefImages).toBe(0)
|
|
67
|
+
})
|
|
68
|
+
})
|
|
@@ -193,3 +193,51 @@ describe("promptBindsFirstFrame suffix suppression (overlap colon-position findi
|
|
|
193
193
|
expect(both.promptSuffix).toContain("closing (last) frame") // pair sentence kept — last frame has no in-prompt binding
|
|
194
194
|
})
|
|
195
195
|
})
|
|
196
|
+
|
|
197
|
+
// ---------------------------------------------------------------------------
|
|
198
|
+
// Per-provider limits (2026-08-15): Seedance 2.5 carries the same three input
|
|
199
|
+
// kinds with much wider caps (30 / 10 / 10). The resolver takes the limits as
|
|
200
|
+
// an argument — defaulting to the 2.0 caps so every existing caller is
|
|
201
|
+
// byte-identical — and the adapter passes the provider's own.
|
|
202
|
+
// ---------------------------------------------------------------------------
|
|
203
|
+
|
|
204
|
+
describe("resolveSeedance2Inputs — per-provider limits", () => {
|
|
205
|
+
const WIDE = { images: 30, videos: 10, audio: 10 }
|
|
206
|
+
const refs = (n: number) => Array.from({ length: n }, (_, i) => `https://r2/ref-${i + 1}.png`)
|
|
207
|
+
|
|
208
|
+
it("keeps 12 reference images + both frames under the 2.5 caps (the 2.0 default would drop 5)", () => {
|
|
209
|
+
const r = resolveSeedance2Inputs({
|
|
210
|
+
firstFrameUrl: "https://r2/first.png",
|
|
211
|
+
lastFrameUrl: "https://r2/last.png",
|
|
212
|
+
refImageUrls: refs(12),
|
|
213
|
+
limits: WIDE,
|
|
214
|
+
})
|
|
215
|
+
expect(r.mode).toBe("reference")
|
|
216
|
+
expect(r.referenceImageUrls).toHaveLength(14)
|
|
217
|
+
expect(r.droppedRefImages).toBe(0)
|
|
218
|
+
// Frames still ride LAST, ordinals bound to their true positions.
|
|
219
|
+
expect(r.promptSuffix).toContain("@image_13")
|
|
220
|
+
expect(r.promptSuffix).toContain("@image_14")
|
|
221
|
+
})
|
|
222
|
+
|
|
223
|
+
it("videos and audio slice to the provided caps", () => {
|
|
224
|
+
const r = resolveSeedance2Inputs({
|
|
225
|
+
refImageUrls: refs(1),
|
|
226
|
+
refVideoUrls: Array.from({ length: 12 }, (_, i) => `https://r2/v${i}.mp4`),
|
|
227
|
+
refAudioUrls: Array.from({ length: 12 }, (_, i) => `https://r2/a${i}.mp3`),
|
|
228
|
+
limits: WIDE,
|
|
229
|
+
})
|
|
230
|
+
expect(r.referenceVideoUrls).toHaveLength(10)
|
|
231
|
+
expect(r.referenceAudioUrls).toHaveLength(10)
|
|
232
|
+
})
|
|
233
|
+
|
|
234
|
+
it("omitting limits keeps the 2.0 caps byte-identical (9-slot drop-trailing)", () => {
|
|
235
|
+
const r = resolveSeedance2Inputs({
|
|
236
|
+
firstFrameUrl: "https://r2/first.png",
|
|
237
|
+
lastFrameUrl: "https://r2/last.png",
|
|
238
|
+
refImageUrls: refs(12),
|
|
239
|
+
})
|
|
240
|
+
expect(r.referenceImageUrls).toHaveLength(9)
|
|
241
|
+
expect(r.droppedRefImages).toBe(5)
|
|
242
|
+
})
|
|
243
|
+
})
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { describe, it, expect } from "vitest"
|
|
2
|
+
|
|
3
|
+
import { resolveVeoI2vInputs } from "../veo-i2v-inputs.js"
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* VEO 3.x i2v input resolution. VEO's API makes frame conditioning and
|
|
7
|
+
* reference ingredients mutually exclusive (one imageUrls array, ≤3, whose
|
|
8
|
+
* meaning flips with generationType) — so an anchored call that must carry
|
|
9
|
+
* identity references moves to REFERENCE_2_VIDEO with the anchor in seat 1.
|
|
10
|
+
* References win the seats (the 2026-08-14 standing rule: refs are a must,
|
|
11
|
+
* frames additional): the end anchor is dropped in reference mode.
|
|
12
|
+
*/
|
|
13
|
+
describe("resolveVeoI2vInputs", () => {
|
|
14
|
+
const FIRST = "https://r2/anchor.png"
|
|
15
|
+
const refs = (n: number) => Array.from({ length: n }, (_, i) => `https://r2/ref-${i + 1}.png`)
|
|
16
|
+
|
|
17
|
+
it("no references ⇒ plain frame mode, byte-identical: frames kept, no generationType, no suffix", () => {
|
|
18
|
+
const r = resolveVeoI2vInputs({ prompt: "p", firstFrameUrl: FIRST, endFrameUrl: "https://r2/end.png" })
|
|
19
|
+
expect(r).toEqual({
|
|
20
|
+
imageUrls: [FIRST, "https://r2/end.png"],
|
|
21
|
+
promptSuffix: "",
|
|
22
|
+
droppedRefImages: 0,
|
|
23
|
+
droppedEndFrame: false,
|
|
24
|
+
})
|
|
25
|
+
})
|
|
26
|
+
|
|
27
|
+
it("references flip the call to REFERENCE_2_VIDEO with the anchor in seat 1, capped at 3", () => {
|
|
28
|
+
const r = resolveVeoI2vInputs({ prompt: "p", firstFrameUrl: FIRST, refImageUrls: refs(4) })
|
|
29
|
+
expect(r.generationType).toBe("REFERENCE_2_VIDEO")
|
|
30
|
+
expect(r.imageUrls).toEqual([FIRST, "https://r2/ref-1.png", "https://r2/ref-2.png"])
|
|
31
|
+
expect(r.droppedRefImages).toBe(2)
|
|
32
|
+
expect(r.promptSuffix).toBe(
|
|
33
|
+
"Use @image_1 as the opening (first) frame of the video. " +
|
|
34
|
+
"@image_2 through @image_3 are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.",
|
|
35
|
+
)
|
|
36
|
+
})
|
|
37
|
+
|
|
38
|
+
it("the end anchor is DROPPED in reference mode — references win the seats", () => {
|
|
39
|
+
const r = resolveVeoI2vInputs({ firstFrameUrl: FIRST, endFrameUrl: "https://r2/end.png", refImageUrls: refs(2) })
|
|
40
|
+
expect(r.imageUrls).toEqual([FIRST, "https://r2/ref-1.png", "https://r2/ref-2.png"])
|
|
41
|
+
expect(r.droppedEndFrame).toBe(true)
|
|
42
|
+
})
|
|
43
|
+
|
|
44
|
+
it("a single kept reference gets the singular sentence", () => {
|
|
45
|
+
const r = resolveVeoI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(1) })
|
|
46
|
+
expect(r.promptSuffix).toContain("@image_2 is an identity reference")
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
it("suppresses the opening-frame sentence when the prompt already binds it", () => {
|
|
50
|
+
const r = resolveVeoI2vInputs({
|
|
51
|
+
prompt: "use @image_1 as the first frame, it is the last keyframe of @video_1",
|
|
52
|
+
firstFrameUrl: FIRST,
|
|
53
|
+
refImageUrls: refs(1),
|
|
54
|
+
})
|
|
55
|
+
expect(r.promptSuffix).not.toContain("opening (first) frame")
|
|
56
|
+
expect(r.promptSuffix).toContain("identity reference")
|
|
57
|
+
})
|
|
58
|
+
})
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
import { VIDEO_REF_LIMITS_BY_PROVIDER } from "@nodaro/shared"
|
|
2
|
+
|
|
3
|
+
import { promptBindsFirstFrame } from "./seedance-2-inputs.js"
|
|
4
|
+
import { identityRefsSentence, REF_BINDING } from "./video-reference-resolver.js"
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Gemini Omni Video i2v input resolution — the sibling of
|
|
8
|
+
* `resolveSeedance2Inputs` for a model whose multimodal channel is ONE flat
|
|
9
|
+
* `image_urls` list.
|
|
10
|
+
*
|
|
11
|
+
* WHY BINDING IS LOAD-BEARING: Gemini Omni receives the start frame and the
|
|
12
|
+
* identity references in the same array, with nothing in the payload marking
|
|
13
|
+
* which is which. A multimodal model treats unbound images as loose context —
|
|
14
|
+
* field finding (recast keyframes run, 2026-08-14): the identity references
|
|
15
|
+
* rode every call and the cast still drifted part to part, because the prompt
|
|
16
|
+
* never said the images WERE identities to keep. So the resolver names the
|
|
17
|
+
* roles in a prompt suffix, through the same `REF_BINDING` swap-point every
|
|
18
|
+
* other video binding uses: image 1 is the opening frame; the rest are
|
|
19
|
+
* identity references, explicitly not frames.
|
|
20
|
+
*
|
|
21
|
+
* BUDGETED, NEVER REJECTED, for the list this resolver assembles: KIE's quota
|
|
22
|
+
* is `images + 2×videos ≤ 7`, and `runGeminiOmni` hard-rejects overflow. That
|
|
23
|
+
* reject is right for a caller-assembled list (the user's own images should
|
|
24
|
+
* not silently thin out) and wrong for THIS merge, where the overflow is our
|
|
25
|
+
* own construction — so trailing references are dropped to fit, the start
|
|
26
|
+
* frame always kept, mirroring `resolveSeedance2Inputs`' drop-trailing
|
|
27
|
+
* convention, and the drop count is reported for the caller to log.
|
|
28
|
+
*
|
|
29
|
+
* BYTE-IDENTICAL when there is nothing to bind: no references ⇒ no suffix and
|
|
30
|
+
* a single-image list — exactly what every plain gemini-omni i2v call has
|
|
31
|
+
* always sent.
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
export interface GeminiOmniI2vInputsArgs {
|
|
35
|
+
/** The composed prompt, used only to detect an existing first-frame binding. */
|
|
36
|
+
prompt?: string
|
|
37
|
+
/** The start frame — always kept, always first in the list. */
|
|
38
|
+
firstFrameUrl: string
|
|
39
|
+
/** Identity references, in priority order (trailing ones drop first). */
|
|
40
|
+
refImageUrls?: Array<string | undefined>
|
|
41
|
+
/** A connected source video occupies 2 of the 7 input slots (KIE quota). */
|
|
42
|
+
videoConnected?: boolean
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface GeminiOmniI2vInputsResult {
|
|
46
|
+
/** `[firstFrameUrl, ...keptRefs]` — the `image_urls` payload, quota-fitted. */
|
|
47
|
+
imageUrls: string[]
|
|
48
|
+
/** The role-binding sentences; empty when no reference survived the budget. */
|
|
49
|
+
promptSuffix: string
|
|
50
|
+
/** References dropped to fit the quota — surface in a log, never silently. */
|
|
51
|
+
droppedRefImages: number
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** The catalog-declared cap (7) — read from the shared limits map so the
|
|
55
|
+
* wire-contract number has one home; the literal is only the safety net. */
|
|
56
|
+
const GEMINI_OMNI_INPUT_SLOTS = VIDEO_REF_LIMITS_BY_PROVIDER["gemini-omni-video"]?.images ?? 7
|
|
57
|
+
|
|
58
|
+
export function resolveGeminiOmniI2vInputs(args: GeminiOmniI2vInputsArgs): GeminiOmniI2vInputsResult {
|
|
59
|
+
const refs = (args.refImageUrls ?? []).filter((u): u is string => typeof u === "string" && u.length > 0)
|
|
60
|
+
const slots = GEMINI_OMNI_INPUT_SLOTS - (args.videoConnected ? 2 : 0)
|
|
61
|
+
const refSlots = Math.max(0, slots - 1)
|
|
62
|
+
const kept = refs.slice(0, refSlots)
|
|
63
|
+
const droppedRefImages = refs.length - kept.length
|
|
64
|
+
const imageUrls = [args.firstFrameUrl, ...kept]
|
|
65
|
+
if (kept.length === 0) return { imageUrls, promptSuffix: "", droppedRefImages }
|
|
66
|
+
|
|
67
|
+
// The opening-frame sentence is suppressed when the prompt already binds it
|
|
68
|
+
// at its own (working) position — same field-finding rule as seedance-2: a
|
|
69
|
+
// duplicate directive at the end dilutes the one that works.
|
|
70
|
+
const frameSentence = promptBindsFirstFrame(args.prompt) ? "" : REF_BINDING.frame(1, "opening")
|
|
71
|
+
const promptSuffix = [frameSentence, identityRefsSentence(2, kept.length + 1)].filter(Boolean).join(" ")
|
|
72
|
+
return { imageUrls, promptSuffix, droppedRefImages }
|
|
73
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -20,6 +20,8 @@ export * from "./sound-aggregator.js"
|
|
|
20
20
|
export * from "./assemble-suno-input.js"
|
|
21
21
|
export * from "./assemble-image-input.js"
|
|
22
22
|
export * from "./seedance-2-inputs.js"
|
|
23
|
+
export * from "./gemini-omni-inputs.js"
|
|
24
|
+
export * from "./veo-i2v-inputs.js"
|
|
23
25
|
export * from "./person.js"
|
|
24
26
|
export * from "./picker-catalogs.js"
|
|
25
27
|
export * from "./picker-analyzer-registry.js"
|
|
@@ -66,3 +68,4 @@ export * from "./style-presets.js"
|
|
|
66
68
|
export * from "./object-asset-presets.js"
|
|
67
69
|
export * from "./factory-snippets/index.js"
|
|
68
70
|
export * from "./picker-wiring.js"
|
|
71
|
+
export * from "./surround-fill.js"
|
package/src/picker-wiring.ts
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
* Pure data, no React. Extracted from the app's parameter-picker-registry so
|
|
7
7
|
* THREE consumers share one definition and cannot drift:
|
|
8
8
|
* 1. The app's community fallback registry (chip pickers, no rich previews).
|
|
9
|
-
* 2. `@
|
|
9
|
+
* 2. `@nodaro/picker-ui`'s registry (attaches preview/Picker renderers).
|
|
10
10
|
* 3. Nodaro Cine's builder panels.
|
|
11
11
|
*
|
|
12
12
|
* Renderers (preview components, multi-dim Picker components) deliberately do
|
|
@@ -166,6 +166,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
|
|
|
166
166
|
"gpt-image": "Creative concepts, illustration, variable quality tiers",
|
|
167
167
|
"gpt-image-2": "Latest GPT Image — sharp text, photorealism, 1K/2K/4K resolution",
|
|
168
168
|
"grok": "General purpose, good text understanding",
|
|
169
|
+
"grok-2": "Grok Imagine 2 — expressive, high-contrast, stylized output",
|
|
169
170
|
"imagen4": "Google's latest, strong photorealism and text rendering",
|
|
170
171
|
"imagen4-fast": "Faster Imagen 4 variant",
|
|
171
172
|
"imagen4-ultra": "Highest quality Imagen 4",
|
|
@@ -242,7 +243,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
|
|
|
242
243
|
"seedance-2": "Seedance 2.0 — multimodal refs (9 images / 3 videos / 3 audio), native multi-track audio, multi-shot storytelling, 4-15s",
|
|
243
244
|
"seedance-2-fast": "Seedance 2.0 Fast — same multimodal + audio capabilities, cheaper and quicker",
|
|
244
245
|
"seedance-2-mini": "Seedance 2.0 Mini — same multimodal + audio capabilities, budget tier, 480p/720p, 4-15s",
|
|
245
|
-
"seedance-2-5": "Seedance 2.5 — up to 30s in ONE shot (no stitching), wider multimodal refs (30 images / 10 videos / 10 audio), native audio, 480p/720p",
|
|
246
|
+
"seedance-2-5": "Seedance 2.5 — up to 30s in ONE shot (no stitching), wider multimodal refs (30 images / 10 videos / 10 audio), native audio, 480p/720p/1080p",
|
|
246
247
|
"minimax-h3": "MiniMax Hailuo 3 — premium multimodal refs (9 images / 3 videos / 3 audio), always-on audio, 2K or 768P, 4-15s per-second pricing",
|
|
247
248
|
"wan": "Versatile, good for animations and transformations",
|
|
248
249
|
"wan-turbo": "Faster Wan generation",
|
|
@@ -270,7 +271,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
|
|
|
270
271
|
"seedance-2": "Seedance 2.0 — start/end frame + multimodal refs, native audio, 4-15s",
|
|
271
272
|
"seedance-2-fast": "Seedance 2.0 Fast — same capabilities, cheaper and quicker",
|
|
272
273
|
"seedance-2-mini": "Seedance 2.0 Mini — same capabilities, budget tier, 480p/720p",
|
|
273
|
-
"seedance-2-5": "Seedance 2.5 — start/end frame + wide multimodal refs, native audio, up to 30s, 480p/720p",
|
|
274
|
+
"seedance-2-5": "Seedance 2.5 — start/end frame + wide multimodal refs, native audio, up to 30s, 480p/720p/1080p",
|
|
274
275
|
"minimax-h3": "MiniMax Hailuo 3 — first/last frame + multimodal refs, always-on audio, 2K or 768P, 4-15s",
|
|
275
276
|
"hailuo-2.3-pro": "Premium Hailuo animation",
|
|
276
277
|
"hailuo-2.3": "Standard Hailuo animation",
|
|
@@ -62,7 +62,7 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
62
62
|
**Generation differences (seedance-2-5 vs the 2.0 SKUs)**
|
|
63
63
|
- A single 2.5 shot runs to 30s, where every 2.0 SKU stops at 15s. Plan a complete 4-6 shot beat inside ONE generation instead of splitting it into two clips and stitching — no seam to hide, and continuity holds because it never leaves the model.
|
|
64
64
|
- 2.5 also takes far more reference material (30 images / 10 videos / 10 audio vs 9/3/3). Treat that as room for COVERAGE — more distinct characters, locations and props in one shot — not as licence to pile refs onto one identity. The "ONE headshot + ONE full-body, 4-5 assets total" rule above still produces the best likeness on 2.5.
|
|
65
|
-
- 2.5 renders at 480p/720p
|
|
65
|
+
- 2.5 renders at 480p/720p/1080p (1080p since 2026-08-17): there is no 4K tier, so route a job that needs 4K to seedance-2 (which has it) or upscale afterwards.
|
|
66
66
|
- With a start frame, 2.5 always derives the output aspect from that frame — an explicit aspect ratio is rejected outright, so compose the frame at the ratio you want.
|
|
67
67
|
|
|
68
68
|
**References (when reference media is attached)**
|
|
@@ -87,8 +87,7 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
87
87
|
- More than 4 referenced people gets unstable: group people into composite images of ≤4 first (image generation), then reference those composites.
|
|
88
88
|
- Repeated extension degrades quality: prefer high-definition reference assets and avoid stacking many continuations.
|
|
89
89
|
|
|
90
|
-
**Auto-path formula (community-sourced enrichment
|
|
91
|
-
higgsfield.ai 4K breakdown; captured 2026-08-09)**
|
|
90
|
+
**Auto-path formula (community-sourced enrichment; captured 2026-08-09)**
|
|
92
91
|
- Six steps IN ORDER, 60-100 words total (longer measurably degrades): Subject → Action → Environment → Camera → Style → Constraints.
|
|
93
92
|
- ONE primary camera instruction per shot. Compound moves chain with "then": "camera slow tracking then subtle rise" — never two competing verbs. The 8 reliable camera types: push-in, pull-out, pan, tracking, orbit/arc, aerial, handheld, locked-off.
|
|
94
93
|
- SEPARATE camera movement from subject movement — the single biggest quality lever: "The dancer spins slowly. Camera holds fixed framing." — never "spinning camera around a dancing person".
|
package/src/resolve-prompt.ts
CHANGED
|
@@ -94,7 +94,18 @@ export function computeNodePrompt(
|
|
|
94
94
|
let typed: ReadonlyArray<string | undefined>
|
|
95
95
|
if (nodeType === "text-to-speech") {
|
|
96
96
|
// data.text is a phantom field on TTS; only directText (gated) is real.
|
|
97
|
-
|
|
97
|
+
//
|
|
98
|
+
// The gate is a PREFERENCE, not a lock: when textSource is "connected" we
|
|
99
|
+
// still fall back to typed text if nothing is wired. Writers flip the gate
|
|
100
|
+
// (PromptFieldSpec.promptGate), but data reaches nodes from places no
|
|
101
|
+
// writer touches — workflows saved before that fix, JSON imports, MCP
|
|
102
|
+
// writes, templates — and there the text sat visible in the node while the
|
|
103
|
+
// run failed with "no text found" (founder, 2026-08-14). Coerce rather
|
|
104
|
+
// than reject, same principle as normalizeModelInput.
|
|
105
|
+
typed =
|
|
106
|
+
data.textSource === "direct" || !present(wired)
|
|
107
|
+
? [data.directText as string | undefined]
|
|
108
|
+
: []
|
|
98
109
|
} else {
|
|
99
110
|
const fields = NODE_PROMPT_CANDIDATE_FIELDS[nodeType] ?? ["prompt"]
|
|
100
111
|
typed = fields.map((f) => data[f] as string | undefined)
|
package/src/seedance-2-inputs.ts
CHANGED
|
@@ -12,6 +12,12 @@ export interface Seedance2InputsArgs {
|
|
|
12
12
|
refImageUrls?: readonly string[]
|
|
13
13
|
refVideoUrls?: readonly string[]
|
|
14
14
|
refAudioUrls?: readonly string[]
|
|
15
|
+
/** Per-provider input caps (2026-08-15). The Seedance 2.x GENERATIONS share
|
|
16
|
+
* this resolver's whole mode logic but not their caps — 2.5 takes the same
|
|
17
|
+
* three kinds at 30/10/10 where 2.0 stops at 9/3/3. Defaults to the 2.0
|
|
18
|
+
* caps so every existing caller is byte-identical; the adapter passes the
|
|
19
|
+
* provider's own entry from VIDEO_REF_LIMITS_BY_PROVIDER. */
|
|
20
|
+
limits?: { images: number; videos: number; audio: number }
|
|
15
21
|
}
|
|
16
22
|
|
|
17
23
|
export interface Seedance2InputsResult {
|
|
@@ -54,11 +60,12 @@ export function promptBindsFirstFrame(prompt: string | undefined): boolean {
|
|
|
54
60
|
}
|
|
55
61
|
|
|
56
62
|
export function resolveSeedance2Inputs(args: Seedance2InputsArgs): Seedance2InputsResult {
|
|
63
|
+
const limits = args.limits ?? SEEDANCE_2_REF_LIMITS
|
|
57
64
|
const firstFrameUrl = clean(args.firstFrameUrl)
|
|
58
65
|
const lastFrameUrl = clean(args.lastFrameUrl)
|
|
59
66
|
const refImages = cleanList(args.refImageUrls)
|
|
60
|
-
const refVideos = cleanList(args.refVideoUrls).slice(0,
|
|
61
|
-
const refAudios = cleanList(args.refAudioUrls).slice(0,
|
|
67
|
+
const refVideos = cleanList(args.refVideoUrls).slice(0, limits.videos)
|
|
68
|
+
const refAudios = cleanList(args.refAudioUrls).slice(0, limits.audio)
|
|
62
69
|
|
|
63
70
|
const hasAnyReference = refImages.length > 0 || refVideos.length > 0 || refAudios.length > 0
|
|
64
71
|
|
|
@@ -81,7 +88,7 @@ export function resolveSeedance2Inputs(args: Seedance2InputsArgs): Seedance2Inpu
|
|
|
81
88
|
// if the 9-image cap is exceeded. Frames are appended AFTER the kept user
|
|
82
89
|
// images so existing user @Image ordinals are preserved.
|
|
83
90
|
const frameCount = (firstFrameUrl ? 1 : 0) + (lastFrameUrl ? 1 : 0)
|
|
84
|
-
const userImageSlots = Math.max(0,
|
|
91
|
+
const userImageSlots = Math.max(0, limits.images - frameCount)
|
|
85
92
|
const keptUserImages = refImages.slice(0, userImageSlots)
|
|
86
93
|
const droppedRefImages = refImages.length - keptUserImages.length
|
|
87
94
|
|
package/src/style-presets.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import type { StyleDirectives } from "@nodaro/shared"
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
* Style Gallery presets
|
|
4
|
+
* Style Gallery presets.
|
|
5
5
|
*
|
|
6
6
|
* Each preset is a named "look" the user picks at Start. Picking one sets the
|
|
7
7
|
* pipeline's `style_directives`, which the Showrunner folds into the plan's
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Surround continuation — the fill prompt.
|
|
3
|
+
*
|
|
4
|
+
* Prompt engineering, so it lives here and not in `@nodaro/shared`: that
|
|
5
|
+
* package is published to npm under Apache-2.0, where every release is an
|
|
6
|
+
* irrevocable grant. This package is never published (`"private": true`).
|
|
7
|
+
* `@nodaro/shared` keeps only the wire contract — the direction enum and the
|
|
8
|
+
* carried-fraction defaults the route Zod schema and the SDK input type need.
|
|
9
|
+
*/
|
|
10
|
+
import type { SurroundDirection } from "@nodaro/shared"
|
|
11
|
+
|
|
12
|
+
/** Which edge of the NEW frame holds the carried pixels vs the painted region. */
|
|
13
|
+
const EDGE: Record<SurroundDirection, { carried: string; painted: string }> = {
|
|
14
|
+
right: { carried: "left", painted: "right" },
|
|
15
|
+
left: { carried: "right", painted: "left" },
|
|
16
|
+
up: { carried: "bottom", painted: "top" },
|
|
17
|
+
down: { carried: "top", painted: "bottom" },
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** What a tilt must actually render (NOT a continuation of the landscape). */
|
|
21
|
+
const TILT_SUBJECT: Record<"up" | "down", { word: string; subject: string; where: string }> = {
|
|
22
|
+
up: {
|
|
23
|
+
word: "up",
|
|
24
|
+
subject: "the open sky directly overhead — sky, clouds, or (for an interior) the canopy or ceiling",
|
|
25
|
+
where: "overhead",
|
|
26
|
+
},
|
|
27
|
+
down: {
|
|
28
|
+
word: "down",
|
|
29
|
+
subject: "the ground directly below — terrain, floor, or water surface",
|
|
30
|
+
where: "below",
|
|
31
|
+
},
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Build the fill prompt the model receives alongside the half-carry composite.
|
|
36
|
+
*
|
|
37
|
+
* `userPrompt` (an optional scene hint from the caller) is woven in front. PAN
|
|
38
|
+
* directions get the seamless-continuation prompt (with the anti-golden-hour
|
|
39
|
+
* negative that fights the documented warm-regrade drift). TILT directions get a
|
|
40
|
+
* subject-forcing prompt — render the sky / ground overhead / below, explicitly
|
|
41
|
+
* NOT a mirrored landscape — which is what stops the vertical echo.
|
|
42
|
+
*/
|
|
43
|
+
export function buildSurroundFillPrompt(direction: SurroundDirection, userPrompt?: string): string {
|
|
44
|
+
const scene = userPrompt && userPrompt.trim() ? `${userPrompt.trim()}. ` : ""
|
|
45
|
+
const { carried, painted } = EDGE[direction]
|
|
46
|
+
|
|
47
|
+
if (direction === "up" || direction === "down") {
|
|
48
|
+
const t = TILT_SUBJECT[direction]
|
|
49
|
+
return (
|
|
50
|
+
`${scene}` +
|
|
51
|
+
`This is a camera tilted straight ${t.word} from the same scene. The ${carried} strip holds real, finished pixels from the edge of the horizon view; the ${painted} region is flat gray and MUST be painted as ${t.subject}. ` +
|
|
52
|
+
`Render what is genuinely ${t.where} — do NOT repeat, mirror, or continue the landscape, and do NOT draw a horizon line or distant scenery in the painted region. ` +
|
|
53
|
+
`CRITICAL: keep the ${carried} strip unchanged and match the scene's EXACT lighting, time of day, white balance, and color grade — the same light as the ${carried} strip; no golden hour, no sunset, no warm relight, no cinematic regrade. ` +
|
|
54
|
+
`Blend smoothly into the ${carried} strip with no visible seam. No people, no text, no labels, no watermarks.`
|
|
55
|
+
)
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// pan (right / left)
|
|
59
|
+
return (
|
|
60
|
+
`${scene}` +
|
|
61
|
+
`This is a partial frame: the ${carried} portion contains real, finished pixels and the ${painted} portion is flat gray that MUST be painted in. ` +
|
|
62
|
+
`Paint ONLY the ${painted} gray region as a natural, seamless continuation of the ${carried} portion — same scene, same perspective, continuing the horizon, geometry, and content across the boundary with no break. ` +
|
|
63
|
+
`Keep the ${carried} portion completely unchanged. ` +
|
|
64
|
+
`CRITICAL: do NOT change the lighting, exposure, white balance, or time of day. Match the ${carried} portion's EXACT light, color temperature, and contrast across the whole frame — if it is flat overcast daylight, keep flat overcast daylight. No golden hour, no sunset, no warm relight, no cinematic regrade. ` +
|
|
65
|
+
`The seam between the ${carried} and ${painted} portions must be invisible. No people, no text, no labels, no watermarks.`
|
|
66
|
+
)
|
|
67
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import { promptBindsFirstFrame } from "./seedance-2-inputs.js"
|
|
2
|
+
import { identityRefsSentence, REF_BINDING } from "./video-reference-resolver.js"
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* VEO 3.x i2v input resolution — the mutually-exclusive sibling of
|
|
6
|
+
* `resolveGeminiOmniI2vInputs`. VEO's API carries ONE `imageUrls` array
|
|
7
|
+
* (≤3) whose meaning flips with `generationType`: plain i2v reads it as
|
|
8
|
+
* [first(, last)] frames; REFERENCE_2_VIDEO reads every entry as a
|
|
9
|
+
* reference ingredient. Frames and identities cannot ride separate
|
|
10
|
+
* channels, so an anchored call that must carry identity references moves
|
|
11
|
+
* to REFERENCE_2_VIDEO with the anchor in seat 1, bound in prose as the
|
|
12
|
+
* opening frame (requested, not pixel-guaranteed — the accepted trade,
|
|
13
|
+
* same as seedance-2's reference mode).
|
|
14
|
+
*
|
|
15
|
+
* REFERENCES WIN THE SEATS (the 2026-08-14 standing rule: refs are a must,
|
|
16
|
+
* frames additional): the end anchor is dropped in reference mode rather
|
|
17
|
+
* than spending one of three seats on a closing guess. The caller logs it.
|
|
18
|
+
*
|
|
19
|
+
* BYTE-IDENTICAL with no references: plain frame mode, frames kept, no
|
|
20
|
+
* generationType, no suffix — exactly what every veo i2v call has always
|
|
21
|
+
* sent.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
export interface VeoI2vInputsArgs {
|
|
25
|
+
/** Used only to detect an existing first-frame binding (seedance rule). */
|
|
26
|
+
prompt?: string
|
|
27
|
+
firstFrameUrl: string
|
|
28
|
+
endFrameUrl?: string
|
|
29
|
+
refImageUrls?: Array<string | undefined>
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export interface VeoI2vInputsResult {
|
|
33
|
+
/** The `imageUrls` payload: frames in plain mode, [anchor, ...refs] in
|
|
34
|
+
* reference mode — never more than VEO's 3-ingredient cap. */
|
|
35
|
+
imageUrls: string[]
|
|
36
|
+
/** Present (REFERENCE_2_VIDEO) exactly when references ride. */
|
|
37
|
+
generationType?: "REFERENCE_2_VIDEO"
|
|
38
|
+
promptSuffix: string
|
|
39
|
+
droppedRefImages: number
|
|
40
|
+
/** True when an end anchor was surrendered to reference mode. */
|
|
41
|
+
droppedEndFrame: boolean
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** VEO's ingredient cap — the adapter's REFERENCE_2_VIDEO path has always
|
|
45
|
+
* sliced to 3 (kie/video.ts), mirrored in VIDEO_REF_LIMITS_BY_PROVIDER. */
|
|
46
|
+
const VEO_INGREDIENT_SLOTS = 3
|
|
47
|
+
|
|
48
|
+
export function resolveVeoI2vInputs(args: VeoI2vInputsArgs): VeoI2vInputsResult {
|
|
49
|
+
const refs = (args.refImageUrls ?? []).filter((u): u is string => typeof u === "string" && u.length > 0)
|
|
50
|
+
if (refs.length === 0) {
|
|
51
|
+
return {
|
|
52
|
+
imageUrls: args.endFrameUrl ? [args.firstFrameUrl, args.endFrameUrl] : [args.firstFrameUrl],
|
|
53
|
+
promptSuffix: "",
|
|
54
|
+
droppedRefImages: 0,
|
|
55
|
+
droppedEndFrame: false,
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
const kept = refs.slice(0, VEO_INGREDIENT_SLOTS - 1)
|
|
59
|
+
const droppedRefImages = refs.length - kept.length
|
|
60
|
+
const frameSentence = promptBindsFirstFrame(args.prompt) ? "" : REF_BINDING.frame(1, "opening")
|
|
61
|
+
return {
|
|
62
|
+
imageUrls: [args.firstFrameUrl, ...kept],
|
|
63
|
+
generationType: "REFERENCE_2_VIDEO",
|
|
64
|
+
promptSuffix: [frameSentence, identityRefsSentence(2, kept.length + 1)].filter(Boolean).join(" "),
|
|
65
|
+
droppedRefImages,
|
|
66
|
+
droppedEndFrame: Boolean(args.endFrameUrl),
|
|
67
|
+
}
|
|
68
|
+
}
|
|
@@ -50,6 +50,18 @@ import type { ConnectedReference } from "@nodaro/shared"
|
|
|
50
50
|
* the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
|
|
51
51
|
* are the ONLY emission sites for the binding surface string.
|
|
52
52
|
*/
|
|
53
|
+
/**
|
|
54
|
+
* The identity-reference binding sentence shared by the flat-image-list
|
|
55
|
+
* resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
|
|
56
|
+
* says the two things a multimodal model needs to hear — match exactly, and
|
|
57
|
+
* these are not frames. One spelling; both resolvers ride it.
|
|
58
|
+
*/
|
|
59
|
+
export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
|
|
60
|
+
return firstOrdinal === lastOrdinal
|
|
61
|
+
? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
|
|
62
|
+
: `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
|
|
63
|
+
}
|
|
64
|
+
|
|
53
65
|
export const REF_BINDING = {
|
|
54
66
|
image: (label: string, n: number) => `the ${label} from @image_${n}`,
|
|
55
67
|
video: (label: string, n: number) => `the ${label} from @video_${n}`,
|