@nodaro/prompts 1.7.0 → 1.7.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@nodaro/prompts",
3
- "version": "1.7.0",
3
+ "version": "1.7.3",
4
4
  "description": "Nodaro's prompt-engineering layer — person/picker catalogs with prompt hints, identity-lock clauses, entity prompt builders, brand presets, and prompt/reference assembly shared by the Nodaro platform and SDK.",
5
5
  "type": "module",
6
6
  "license": "FSL-1.1-Apache-2.0",
@@ -0,0 +1,68 @@
1
+ import { describe, it, expect } from "vitest"
2
+
3
+ import { resolveGeminiOmniI2vInputs } from "../gemini-omni-inputs.js"
4
+
5
+ /**
6
+ * Gemini Omni i2v input resolution — the flat `image_urls` sibling of the
7
+ * seedance-2 resolver. The stakes: an unbound image list reads as loose
8
+ * context to a multimodal model (field finding 2026-08-14 — identity refs
9
+ * rode every keyframes call and the cast still drifted), and an unbudgeted
10
+ * list trips KIE's 7-input hard reject.
11
+ */
12
+ describe("resolveGeminiOmniI2vInputs", () => {
13
+ const FIRST = "https://r2/anchor.png"
14
+ const refs = (n: number) => Array.from({ length: n }, (_, i) => `https://r2/ref-${i + 1}.png`)
15
+
16
+ it("binds the roles: image 1 is the opening frame, the rest are identities — not frames", () => {
17
+ const r = resolveGeminiOmniI2vInputs({ prompt: "a walk on the beach", firstFrameUrl: FIRST, refImageUrls: refs(3) })
18
+ expect(r.imageUrls).toEqual([FIRST, ...refs(3)])
19
+ expect(r.promptSuffix).toBe(
20
+ "Use @image_1 as the opening (first) frame of the video. " +
21
+ "@image_2 through @image_4 are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.",
22
+ )
23
+ expect(r.droppedRefImages).toBe(0)
24
+ })
25
+
26
+ it("a single reference gets the singular sentence", () => {
27
+ const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(1) })
28
+ expect(r.promptSuffix).toContain("@image_2 is an identity reference")
29
+ expect(r.promptSuffix).not.toContain("through")
30
+ })
31
+
32
+ it("no references ⇒ byte-identical plain i2v: single image, no suffix", () => {
33
+ const r = resolveGeminiOmniI2vInputs({ prompt: "p", firstFrameUrl: FIRST })
34
+ expect(r).toEqual({ imageUrls: [FIRST], promptSuffix: "", droppedRefImages: 0 })
35
+ })
36
+
37
+ it("drops TRAILING references to fit the 7-input quota — the start frame is never the one that goes", () => {
38
+ const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(9) })
39
+ expect(r.imageUrls).toHaveLength(7)
40
+ expect(r.imageUrls[0]).toBe(FIRST)
41
+ expect(r.imageUrls.at(-1)).toBe("https://r2/ref-6.png")
42
+ expect(r.droppedRefImages).toBe(3)
43
+ // The binding names exactly the kept span.
44
+ expect(r.promptSuffix).toContain("@image_2 through @image_7")
45
+ })
46
+
47
+ it("a connected source video eats two slots (images + 2×videos ≤ 7)", () => {
48
+ const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(9), videoConnected: true })
49
+ expect(r.imageUrls).toHaveLength(5)
50
+ expect(r.droppedRefImages).toBe(5)
51
+ })
52
+
53
+ it("suppresses the opening-frame sentence when the prompt already binds it, keeping the identity sentence", () => {
54
+ const r = resolveGeminiOmniI2vInputs({
55
+ prompt: "use @image_1 as the first frame, it is the last keyframe of @video_1",
56
+ firstFrameUrl: FIRST,
57
+ refImageUrls: refs(2),
58
+ })
59
+ expect(r.promptSuffix).not.toContain("opening (first) frame")
60
+ expect(r.promptSuffix).toContain("identity references")
61
+ })
62
+
63
+ it("skips empty/undefined reference entries without burning slots", () => {
64
+ const r = resolveGeminiOmniI2vInputs({ firstFrameUrl: FIRST, refImageUrls: [undefined, "", ...refs(2)] })
65
+ expect(r.imageUrls).toEqual([FIRST, ...refs(2)])
66
+ expect(r.droppedRefImages).toBe(0)
67
+ })
68
+ })
@@ -193,3 +193,51 @@ describe("promptBindsFirstFrame suffix suppression (overlap colon-position findi
193
193
  expect(both.promptSuffix).toContain("closing (last) frame") // pair sentence kept — last frame has no in-prompt binding
194
194
  })
195
195
  })
196
+
197
+ // ---------------------------------------------------------------------------
198
+ // Per-provider limits (2026-08-15): Seedance 2.5 carries the same three input
199
+ // kinds with much wider caps (30 / 10 / 10). The resolver takes the limits as
200
+ // an argument — defaulting to the 2.0 caps so every existing caller is
201
+ // byte-identical — and the adapter passes the provider's own.
202
+ // ---------------------------------------------------------------------------
203
+
204
+ describe("resolveSeedance2Inputs — per-provider limits", () => {
205
+ const WIDE = { images: 30, videos: 10, audio: 10 }
206
+ const refs = (n: number) => Array.from({ length: n }, (_, i) => `https://r2/ref-${i + 1}.png`)
207
+
208
+ it("keeps 12 reference images + both frames under the 2.5 caps (the 2.0 default would drop 5)", () => {
209
+ const r = resolveSeedance2Inputs({
210
+ firstFrameUrl: "https://r2/first.png",
211
+ lastFrameUrl: "https://r2/last.png",
212
+ refImageUrls: refs(12),
213
+ limits: WIDE,
214
+ })
215
+ expect(r.mode).toBe("reference")
216
+ expect(r.referenceImageUrls).toHaveLength(14)
217
+ expect(r.droppedRefImages).toBe(0)
218
+ // Frames still ride LAST, ordinals bound to their true positions.
219
+ expect(r.promptSuffix).toContain("@image_13")
220
+ expect(r.promptSuffix).toContain("@image_14")
221
+ })
222
+
223
+ it("videos and audio slice to the provided caps", () => {
224
+ const r = resolveSeedance2Inputs({
225
+ refImageUrls: refs(1),
226
+ refVideoUrls: Array.from({ length: 12 }, (_, i) => `https://r2/v${i}.mp4`),
227
+ refAudioUrls: Array.from({ length: 12 }, (_, i) => `https://r2/a${i}.mp3`),
228
+ limits: WIDE,
229
+ })
230
+ expect(r.referenceVideoUrls).toHaveLength(10)
231
+ expect(r.referenceAudioUrls).toHaveLength(10)
232
+ })
233
+
234
+ it("omitting limits keeps the 2.0 caps byte-identical (9-slot drop-trailing)", () => {
235
+ const r = resolveSeedance2Inputs({
236
+ firstFrameUrl: "https://r2/first.png",
237
+ lastFrameUrl: "https://r2/last.png",
238
+ refImageUrls: refs(12),
239
+ })
240
+ expect(r.referenceImageUrls).toHaveLength(9)
241
+ expect(r.droppedRefImages).toBe(5)
242
+ })
243
+ })
@@ -0,0 +1,58 @@
1
+ import { describe, it, expect } from "vitest"
2
+
3
+ import { resolveVeoI2vInputs } from "../veo-i2v-inputs.js"
4
+
5
+ /**
6
+ * VEO 3.x i2v input resolution. VEO's API makes frame conditioning and
7
+ * reference ingredients mutually exclusive (one imageUrls array, ≤3, whose
8
+ * meaning flips with generationType) — so an anchored call that must carry
9
+ * identity references moves to REFERENCE_2_VIDEO with the anchor in seat 1.
10
+ * References win the seats (the 2026-08-14 standing rule: refs are a must,
11
+ * frames additional): the end anchor is dropped in reference mode.
12
+ */
13
+ describe("resolveVeoI2vInputs", () => {
14
+ const FIRST = "https://r2/anchor.png"
15
+ const refs = (n: number) => Array.from({ length: n }, (_, i) => `https://r2/ref-${i + 1}.png`)
16
+
17
+ it("no references ⇒ plain frame mode, byte-identical: frames kept, no generationType, no suffix", () => {
18
+ const r = resolveVeoI2vInputs({ prompt: "p", firstFrameUrl: FIRST, endFrameUrl: "https://r2/end.png" })
19
+ expect(r).toEqual({
20
+ imageUrls: [FIRST, "https://r2/end.png"],
21
+ promptSuffix: "",
22
+ droppedRefImages: 0,
23
+ droppedEndFrame: false,
24
+ })
25
+ })
26
+
27
+ it("references flip the call to REFERENCE_2_VIDEO with the anchor in seat 1, capped at 3", () => {
28
+ const r = resolveVeoI2vInputs({ prompt: "p", firstFrameUrl: FIRST, refImageUrls: refs(4) })
29
+ expect(r.generationType).toBe("REFERENCE_2_VIDEO")
30
+ expect(r.imageUrls).toEqual([FIRST, "https://r2/ref-1.png", "https://r2/ref-2.png"])
31
+ expect(r.droppedRefImages).toBe(2)
32
+ expect(r.promptSuffix).toBe(
33
+ "Use @image_1 as the opening (first) frame of the video. " +
34
+ "@image_2 through @image_3 are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.",
35
+ )
36
+ })
37
+
38
+ it("the end anchor is DROPPED in reference mode — references win the seats", () => {
39
+ const r = resolveVeoI2vInputs({ firstFrameUrl: FIRST, endFrameUrl: "https://r2/end.png", refImageUrls: refs(2) })
40
+ expect(r.imageUrls).toEqual([FIRST, "https://r2/ref-1.png", "https://r2/ref-2.png"])
41
+ expect(r.droppedEndFrame).toBe(true)
42
+ })
43
+
44
+ it("a single kept reference gets the singular sentence", () => {
45
+ const r = resolveVeoI2vInputs({ firstFrameUrl: FIRST, refImageUrls: refs(1) })
46
+ expect(r.promptSuffix).toContain("@image_2 is an identity reference")
47
+ })
48
+
49
+ it("suppresses the opening-frame sentence when the prompt already binds it", () => {
50
+ const r = resolveVeoI2vInputs({
51
+ prompt: "use @image_1 as the first frame, it is the last keyframe of @video_1",
52
+ firstFrameUrl: FIRST,
53
+ refImageUrls: refs(1),
54
+ })
55
+ expect(r.promptSuffix).not.toContain("opening (first) frame")
56
+ expect(r.promptSuffix).toContain("identity reference")
57
+ })
58
+ })
@@ -0,0 +1,73 @@
1
+ import { VIDEO_REF_LIMITS_BY_PROVIDER } from "@nodaro/shared"
2
+
3
+ import { promptBindsFirstFrame } from "./seedance-2-inputs.js"
4
+ import { identityRefsSentence, REF_BINDING } from "./video-reference-resolver.js"
5
+
6
+ /**
7
+ * Gemini Omni Video i2v input resolution — the sibling of
8
+ * `resolveSeedance2Inputs` for a model whose multimodal channel is ONE flat
9
+ * `image_urls` list.
10
+ *
11
+ * WHY BINDING IS LOAD-BEARING: Gemini Omni receives the start frame and the
12
+ * identity references in the same array, with nothing in the payload marking
13
+ * which is which. A multimodal model treats unbound images as loose context —
14
+ * field finding (recast keyframes run, 2026-08-14): the identity references
15
+ * rode every call and the cast still drifted part to part, because the prompt
16
+ * never said the images WERE identities to keep. So the resolver names the
17
+ * roles in a prompt suffix, through the same `REF_BINDING` swap-point every
18
+ * other video binding uses: image 1 is the opening frame; the rest are
19
+ * identity references, explicitly not frames.
20
+ *
21
+ * BUDGETED, NEVER REJECTED, for the list this resolver assembles: KIE's quota
22
+ * is `images + 2×videos ≤ 7`, and `runGeminiOmni` hard-rejects overflow. That
23
+ * reject is right for a caller-assembled list (the user's own images should
24
+ * not silently thin out) and wrong for THIS merge, where the overflow is our
25
+ * own construction — so trailing references are dropped to fit, the start
26
+ * frame always kept, mirroring `resolveSeedance2Inputs`' drop-trailing
27
+ * convention, and the drop count is reported for the caller to log.
28
+ *
29
+ * BYTE-IDENTICAL when there is nothing to bind: no references ⇒ no suffix and
30
+ * a single-image list — exactly what every plain gemini-omni i2v call has
31
+ * always sent.
32
+ */
33
+
34
+ export interface GeminiOmniI2vInputsArgs {
35
+ /** The composed prompt, used only to detect an existing first-frame binding. */
36
+ prompt?: string
37
+ /** The start frame — always kept, always first in the list. */
38
+ firstFrameUrl: string
39
+ /** Identity references, in priority order (trailing ones drop first). */
40
+ refImageUrls?: Array<string | undefined>
41
+ /** A connected source video occupies 2 of the 7 input slots (KIE quota). */
42
+ videoConnected?: boolean
43
+ }
44
+
45
+ export interface GeminiOmniI2vInputsResult {
46
+ /** `[firstFrameUrl, ...keptRefs]` — the `image_urls` payload, quota-fitted. */
47
+ imageUrls: string[]
48
+ /** The role-binding sentences; empty when no reference survived the budget. */
49
+ promptSuffix: string
50
+ /** References dropped to fit the quota — surface in a log, never silently. */
51
+ droppedRefImages: number
52
+ }
53
+
54
+ /** The catalog-declared cap (7) — read from the shared limits map so the
55
+ * wire-contract number has one home; the literal is only the safety net. */
56
+ const GEMINI_OMNI_INPUT_SLOTS = VIDEO_REF_LIMITS_BY_PROVIDER["gemini-omni-video"]?.images ?? 7
57
+
58
+ export function resolveGeminiOmniI2vInputs(args: GeminiOmniI2vInputsArgs): GeminiOmniI2vInputsResult {
59
+ const refs = (args.refImageUrls ?? []).filter((u): u is string => typeof u === "string" && u.length > 0)
60
+ const slots = GEMINI_OMNI_INPUT_SLOTS - (args.videoConnected ? 2 : 0)
61
+ const refSlots = Math.max(0, slots - 1)
62
+ const kept = refs.slice(0, refSlots)
63
+ const droppedRefImages = refs.length - kept.length
64
+ const imageUrls = [args.firstFrameUrl, ...kept]
65
+ if (kept.length === 0) return { imageUrls, promptSuffix: "", droppedRefImages }
66
+
67
+ // The opening-frame sentence is suppressed when the prompt already binds it
68
+ // at its own (working) position — same field-finding rule as seedance-2: a
69
+ // duplicate directive at the end dilutes the one that works.
70
+ const frameSentence = promptBindsFirstFrame(args.prompt) ? "" : REF_BINDING.frame(1, "opening")
71
+ const promptSuffix = [frameSentence, identityRefsSentence(2, kept.length + 1)].filter(Boolean).join(" ")
72
+ return { imageUrls, promptSuffix, droppedRefImages }
73
+ }
package/src/index.ts CHANGED
@@ -20,6 +20,8 @@ export * from "./sound-aggregator.js"
20
20
  export * from "./assemble-suno-input.js"
21
21
  export * from "./assemble-image-input.js"
22
22
  export * from "./seedance-2-inputs.js"
23
+ export * from "./gemini-omni-inputs.js"
24
+ export * from "./veo-i2v-inputs.js"
23
25
  export * from "./person.js"
24
26
  export * from "./picker-catalogs.js"
25
27
  export * from "./picker-analyzer-registry.js"
@@ -66,3 +68,4 @@ export * from "./style-presets.js"
66
68
  export * from "./object-asset-presets.js"
67
69
  export * from "./factory-snippets/index.js"
68
70
  export * from "./picker-wiring.js"
71
+ export * from "./surround-fill.js"
@@ -6,7 +6,7 @@
6
6
  * Pure data, no React. Extracted from the app's parameter-picker-registry so
7
7
  * THREE consumers share one definition and cannot drift:
8
8
  * 1. The app's community fallback registry (chip pickers, no rich previews).
9
- * 2. `@nodaroai/picker-ui`'s rich registry (attaches preview/Picker renderers).
9
+ * 2. `@nodaro/picker-ui`'s registry (attaches preview/Picker renderers).
10
10
  * 3. Nodaro Cine's builder panels.
11
11
  *
12
12
  * Renderers (preview components, multi-dim Picker components) deliberately do
@@ -166,6 +166,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
166
166
  "gpt-image": "Creative concepts, illustration, variable quality tiers",
167
167
  "gpt-image-2": "Latest GPT Image — sharp text, photorealism, 1K/2K/4K resolution",
168
168
  "grok": "General purpose, good text understanding",
169
+ "grok-2": "Grok Imagine 2 — expressive, high-contrast, stylized output",
169
170
  "imagen4": "Google's latest, strong photorealism and text rendering",
170
171
  "imagen4-fast": "Faster Imagen 4 variant",
171
172
  "imagen4-ultra": "Highest quality Imagen 4",
@@ -242,7 +243,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
242
243
  "seedance-2": "Seedance 2.0 — multimodal refs (9 images / 3 videos / 3 audio), native multi-track audio, multi-shot storytelling, 4-15s",
243
244
  "seedance-2-fast": "Seedance 2.0 Fast — same multimodal + audio capabilities, cheaper and quicker",
244
245
  "seedance-2-mini": "Seedance 2.0 Mini — same multimodal + audio capabilities, budget tier, 480p/720p, 4-15s",
245
- "seedance-2-5": "Seedance 2.5 — up to 30s in ONE shot (no stitching), wider multimodal refs (30 images / 10 videos / 10 audio), native audio, 480p/720p",
246
+ "seedance-2-5": "Seedance 2.5 — up to 30s in ONE shot (no stitching), wider multimodal refs (30 images / 10 videos / 10 audio), native audio, 480p/720p/1080p",
246
247
  "minimax-h3": "MiniMax Hailuo 3 — premium multimodal refs (9 images / 3 videos / 3 audio), always-on audio, 2K or 768P, 4-15s per-second pricing",
247
248
  "wan": "Versatile, good for animations and transformations",
248
249
  "wan-turbo": "Faster Wan generation",
@@ -270,7 +271,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
270
271
  "seedance-2": "Seedance 2.0 — start/end frame + multimodal refs, native audio, 4-15s",
271
272
  "seedance-2-fast": "Seedance 2.0 Fast — same capabilities, cheaper and quicker",
272
273
  "seedance-2-mini": "Seedance 2.0 Mini — same capabilities, budget tier, 480p/720p",
273
- "seedance-2-5": "Seedance 2.5 — start/end frame + wide multimodal refs, native audio, up to 30s, 480p/720p",
274
+ "seedance-2-5": "Seedance 2.5 — start/end frame + wide multimodal refs, native audio, up to 30s, 480p/720p/1080p",
274
275
  "minimax-h3": "MiniMax Hailuo 3 — first/last frame + multimodal refs, always-on audio, 2K or 768P, 4-15s",
275
276
  "hailuo-2.3-pro": "Premium Hailuo animation",
276
277
  "hailuo-2.3": "Standard Hailuo animation",
@@ -62,7 +62,7 @@ precise subject → action details → scene/environment → lighting & color to
62
62
  **Generation differences (seedance-2-5 vs the 2.0 SKUs)**
63
63
  - A single 2.5 shot runs to 30s, where every 2.0 SKU stops at 15s. Plan a complete 4-6 shot beat inside ONE generation instead of splitting it into two clips and stitching — no seam to hide, and continuity holds because it never leaves the model.
64
64
  - 2.5 also takes far more reference material (30 images / 10 videos / 10 audio vs 9/3/3). Treat that as room for COVERAGE — more distinct characters, locations and props in one shot — not as licence to pile refs onto one identity. The "ONE headshot + ONE full-body, 4-5 assets total" rule above still produces the best likeness on 2.5.
65
- - 2.5 renders at 480p/720p only: there is no 1080p or 4K tier, so route a job that needs one to seedance-2 (which has both) or upscale afterwards.
65
+ - 2.5 renders at 480p/720p/1080p (1080p since 2026-08-17): there is no 4K tier, so route a job that needs 4K to seedance-2 (which has it) or upscale afterwards.
66
66
  - With a start frame, 2.5 always derives the output aspect from that frame — an explicit aspect ratio is rejected outright, so compose the frame at the ratio you want.
67
67
 
68
68
  **References (when reference media is attached)**
@@ -87,8 +87,7 @@ precise subject → action details → scene/environment → lighting & color to
87
87
  - More than 4 referenced people gets unstable: group people into composite images of ≤4 first (image generation), then reference those composites.
88
88
  - Repeated extension degrades quality: prefer high-definition reference assets and avoid stacking many continuations.
89
89
 
90
- **Auto-path formula (community-sourced enrichment — apiyi.com Seedance 2.0 prompt guide,
91
- higgsfield.ai 4K breakdown; captured 2026-08-09)**
90
+ **Auto-path formula (community-sourced enrichment; captured 2026-08-09)**
92
91
  - Six steps IN ORDER, 60-100 words total (longer measurably degrades): Subject → Action → Environment → Camera → Style → Constraints.
93
92
  - ONE primary camera instruction per shot. Compound moves chain with "then": "camera slow tracking then subtle rise" — never two competing verbs. The 8 reliable camera types: push-in, pull-out, pan, tracking, orbit/arc, aerial, handheld, locked-off.
94
93
  - SEPARATE camera movement from subject movement — the single biggest quality lever: "The dancer spins slowly. Camera holds fixed framing." — never "spinning camera around a dancing person".
@@ -94,7 +94,18 @@ export function computeNodePrompt(
94
94
  let typed: ReadonlyArray<string | undefined>
95
95
  if (nodeType === "text-to-speech") {
96
96
  // data.text is a phantom field on TTS; only directText (gated) is real.
97
- typed = data.textSource === "direct" ? [data.directText as string | undefined] : []
97
+ //
98
+ // The gate is a PREFERENCE, not a lock: when textSource is "connected" we
99
+ // still fall back to typed text if nothing is wired. Writers flip the gate
100
+ // (PromptFieldSpec.promptGate), but data reaches nodes from places no
101
+ // writer touches — workflows saved before that fix, JSON imports, MCP
102
+ // writes, templates — and there the text sat visible in the node while the
103
+ // run failed with "no text found" (founder, 2026-08-14). Coerce rather
104
+ // than reject, same principle as normalizeModelInput.
105
+ typed =
106
+ data.textSource === "direct" || !present(wired)
107
+ ? [data.directText as string | undefined]
108
+ : []
98
109
  } else {
99
110
  const fields = NODE_PROMPT_CANDIDATE_FIELDS[nodeType] ?? ["prompt"]
100
111
  typed = fields.map((f) => data[f] as string | undefined)
@@ -12,6 +12,12 @@ export interface Seedance2InputsArgs {
12
12
  refImageUrls?: readonly string[]
13
13
  refVideoUrls?: readonly string[]
14
14
  refAudioUrls?: readonly string[]
15
+ /** Per-provider input caps (2026-08-15). The Seedance 2.x GENERATIONS share
16
+ * this resolver's whole mode logic but not their caps — 2.5 takes the same
17
+ * three kinds at 30/10/10 where 2.0 stops at 9/3/3. Defaults to the 2.0
18
+ * caps so every existing caller is byte-identical; the adapter passes the
19
+ * provider's own entry from VIDEO_REF_LIMITS_BY_PROVIDER. */
20
+ limits?: { images: number; videos: number; audio: number }
15
21
  }
16
22
 
17
23
  export interface Seedance2InputsResult {
@@ -54,11 +60,12 @@ export function promptBindsFirstFrame(prompt: string | undefined): boolean {
54
60
  }
55
61
 
56
62
  export function resolveSeedance2Inputs(args: Seedance2InputsArgs): Seedance2InputsResult {
63
+ const limits = args.limits ?? SEEDANCE_2_REF_LIMITS
57
64
  const firstFrameUrl = clean(args.firstFrameUrl)
58
65
  const lastFrameUrl = clean(args.lastFrameUrl)
59
66
  const refImages = cleanList(args.refImageUrls)
60
- const refVideos = cleanList(args.refVideoUrls).slice(0, SEEDANCE_2_REF_LIMITS.videos)
61
- const refAudios = cleanList(args.refAudioUrls).slice(0, SEEDANCE_2_REF_LIMITS.audio)
67
+ const refVideos = cleanList(args.refVideoUrls).slice(0, limits.videos)
68
+ const refAudios = cleanList(args.refAudioUrls).slice(0, limits.audio)
62
69
 
63
70
  const hasAnyReference = refImages.length > 0 || refVideos.length > 0 || refAudios.length > 0
64
71
 
@@ -81,7 +88,7 @@ export function resolveSeedance2Inputs(args: Seedance2InputsArgs): Seedance2Inpu
81
88
  // if the 9-image cap is exceeded. Frames are appended AFTER the kept user
82
89
  // images so existing user @Image ordinals are preserved.
83
90
  const frameCount = (firstFrameUrl ? 1 : 0) + (lastFrameUrl ? 1 : 0)
84
- const userImageSlots = Math.max(0, SEEDANCE_2_REF_LIMITS.images - frameCount)
91
+ const userImageSlots = Math.max(0, limits.images - frameCount)
85
92
  const keptUserImages = refImages.slice(0, userImageSlots)
86
93
  const droppedRefImages = refImages.length - keptUserImages.length
87
94
 
@@ -1,7 +1,7 @@
1
1
  import type { StyleDirectives } from "@nodaro/shared"
2
2
 
3
3
  /**
4
- * Style Gallery presets (north-star §6 ①).
4
+ * Style Gallery presets.
5
5
  *
6
6
  * Each preset is a named "look" the user picks at Start. Picking one sets the
7
7
  * pipeline's `style_directives`, which the Showrunner folds into the plan's
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Surround continuation — the fill prompt.
3
+ *
4
+ * Prompt engineering, so it lives here and not in `@nodaro/shared`: that
5
+ * package is published to npm under Apache-2.0, where every release is an
6
+ * irrevocable grant. This package is never published (`"private": true`).
7
+ * `@nodaro/shared` keeps only the wire contract — the direction enum and the
8
+ * carried-fraction defaults the route Zod schema and the SDK input type need.
9
+ */
10
+ import type { SurroundDirection } from "@nodaro/shared"
11
+
12
+ /** Which edge of the NEW frame holds the carried pixels vs the painted region. */
13
+ const EDGE: Record<SurroundDirection, { carried: string; painted: string }> = {
14
+ right: { carried: "left", painted: "right" },
15
+ left: { carried: "right", painted: "left" },
16
+ up: { carried: "bottom", painted: "top" },
17
+ down: { carried: "top", painted: "bottom" },
18
+ }
19
+
20
+ /** What a tilt must actually render (NOT a continuation of the landscape). */
21
+ const TILT_SUBJECT: Record<"up" | "down", { word: string; subject: string; where: string }> = {
22
+ up: {
23
+ word: "up",
24
+ subject: "the open sky directly overhead — sky, clouds, or (for an interior) the canopy or ceiling",
25
+ where: "overhead",
26
+ },
27
+ down: {
28
+ word: "down",
29
+ subject: "the ground directly below — terrain, floor, or water surface",
30
+ where: "below",
31
+ },
32
+ }
33
+
34
+ /**
35
+ * Build the fill prompt the model receives alongside the half-carry composite.
36
+ *
37
+ * `userPrompt` (an optional scene hint from the caller) is woven in front. PAN
38
+ * directions get the seamless-continuation prompt (with the anti-golden-hour
39
+ * negative that fights the documented warm-regrade drift). TILT directions get a
40
+ * subject-forcing prompt — render the sky / ground overhead / below, explicitly
41
+ * NOT a mirrored landscape — which is what stops the vertical echo.
42
+ */
43
+ export function buildSurroundFillPrompt(direction: SurroundDirection, userPrompt?: string): string {
44
+ const scene = userPrompt && userPrompt.trim() ? `${userPrompt.trim()}. ` : ""
45
+ const { carried, painted } = EDGE[direction]
46
+
47
+ if (direction === "up" || direction === "down") {
48
+ const t = TILT_SUBJECT[direction]
49
+ return (
50
+ `${scene}` +
51
+ `This is a camera tilted straight ${t.word} from the same scene. The ${carried} strip holds real, finished pixels from the edge of the horizon view; the ${painted} region is flat gray and MUST be painted as ${t.subject}. ` +
52
+ `Render what is genuinely ${t.where} — do NOT repeat, mirror, or continue the landscape, and do NOT draw a horizon line or distant scenery in the painted region. ` +
53
+ `CRITICAL: keep the ${carried} strip unchanged and match the scene's EXACT lighting, time of day, white balance, and color grade — the same light as the ${carried} strip; no golden hour, no sunset, no warm relight, no cinematic regrade. ` +
54
+ `Blend smoothly into the ${carried} strip with no visible seam. No people, no text, no labels, no watermarks.`
55
+ )
56
+ }
57
+
58
+ // pan (right / left)
59
+ return (
60
+ `${scene}` +
61
+ `This is a partial frame: the ${carried} portion contains real, finished pixels and the ${painted} portion is flat gray that MUST be painted in. ` +
62
+ `Paint ONLY the ${painted} gray region as a natural, seamless continuation of the ${carried} portion — same scene, same perspective, continuing the horizon, geometry, and content across the boundary with no break. ` +
63
+ `Keep the ${carried} portion completely unchanged. ` +
64
+ `CRITICAL: do NOT change the lighting, exposure, white balance, or time of day. Match the ${carried} portion's EXACT light, color temperature, and contrast across the whole frame — if it is flat overcast daylight, keep flat overcast daylight. No golden hour, no sunset, no warm relight, no cinematic regrade. ` +
65
+ `The seam between the ${carried} and ${painted} portions must be invisible. No people, no text, no labels, no watermarks.`
66
+ )
67
+ }
@@ -0,0 +1,68 @@
1
+ import { promptBindsFirstFrame } from "./seedance-2-inputs.js"
2
+ import { identityRefsSentence, REF_BINDING } from "./video-reference-resolver.js"
3
+
4
+ /**
5
+ * VEO 3.x i2v input resolution — the mutually-exclusive sibling of
6
+ * `resolveGeminiOmniI2vInputs`. VEO's API carries ONE `imageUrls` array
7
+ * (≤3) whose meaning flips with `generationType`: plain i2v reads it as
8
+ * [first(, last)] frames; REFERENCE_2_VIDEO reads every entry as a
9
+ * reference ingredient. Frames and identities cannot ride separate
10
+ * channels, so an anchored call that must carry identity references moves
11
+ * to REFERENCE_2_VIDEO with the anchor in seat 1, bound in prose as the
12
+ * opening frame (requested, not pixel-guaranteed — the accepted trade,
13
+ * same as seedance-2's reference mode).
14
+ *
15
+ * REFERENCES WIN THE SEATS (the 2026-08-14 standing rule: refs are a must,
16
+ * frames additional): the end anchor is dropped in reference mode rather
17
+ * than spending one of three seats on a closing guess. The caller logs it.
18
+ *
19
+ * BYTE-IDENTICAL with no references: plain frame mode, frames kept, no
20
+ * generationType, no suffix — exactly what every veo i2v call has always
21
+ * sent.
22
+ */
23
+
24
+ export interface VeoI2vInputsArgs {
25
+ /** Used only to detect an existing first-frame binding (seedance rule). */
26
+ prompt?: string
27
+ firstFrameUrl: string
28
+ endFrameUrl?: string
29
+ refImageUrls?: Array<string | undefined>
30
+ }
31
+
32
+ export interface VeoI2vInputsResult {
33
+ /** The `imageUrls` payload: frames in plain mode, [anchor, ...refs] in
34
+ * reference mode — never more than VEO's 3-ingredient cap. */
35
+ imageUrls: string[]
36
+ /** Present (REFERENCE_2_VIDEO) exactly when references ride. */
37
+ generationType?: "REFERENCE_2_VIDEO"
38
+ promptSuffix: string
39
+ droppedRefImages: number
40
+ /** True when an end anchor was surrendered to reference mode. */
41
+ droppedEndFrame: boolean
42
+ }
43
+
44
+ /** VEO's ingredient cap — the adapter's REFERENCE_2_VIDEO path has always
45
+ * sliced to 3 (kie/video.ts), mirrored in VIDEO_REF_LIMITS_BY_PROVIDER. */
46
+ const VEO_INGREDIENT_SLOTS = 3
47
+
48
+ export function resolveVeoI2vInputs(args: VeoI2vInputsArgs): VeoI2vInputsResult {
49
+ const refs = (args.refImageUrls ?? []).filter((u): u is string => typeof u === "string" && u.length > 0)
50
+ if (refs.length === 0) {
51
+ return {
52
+ imageUrls: args.endFrameUrl ? [args.firstFrameUrl, args.endFrameUrl] : [args.firstFrameUrl],
53
+ promptSuffix: "",
54
+ droppedRefImages: 0,
55
+ droppedEndFrame: false,
56
+ }
57
+ }
58
+ const kept = refs.slice(0, VEO_INGREDIENT_SLOTS - 1)
59
+ const droppedRefImages = refs.length - kept.length
60
+ const frameSentence = promptBindsFirstFrame(args.prompt) ? "" : REF_BINDING.frame(1, "opening")
61
+ return {
62
+ imageUrls: [args.firstFrameUrl, ...kept],
63
+ generationType: "REFERENCE_2_VIDEO",
64
+ promptSuffix: [frameSentence, identityRefsSentence(2, kept.length + 1)].filter(Boolean).join(" "),
65
+ droppedRefImages,
66
+ droppedEndFrame: Boolean(args.endFrameUrl),
67
+ }
68
+ }
@@ -50,6 +50,18 @@ import type { ConnectedReference } from "@nodaro/shared"
50
50
  * the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
51
51
  * are the ONLY emission sites for the binding surface string.
52
52
  */
53
+ /**
54
+ * The identity-reference binding sentence shared by the flat-image-list
55
+ * resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
56
+ * says the two things a multimodal model needs to hear — match exactly, and
57
+ * these are not frames. One spelling; both resolvers ride it.
58
+ */
59
+ export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
60
+ return firstOrdinal === lastOrdinal
61
+ ? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
62
+ : `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
63
+ }
64
+
53
65
  export const REF_BINDING = {
54
66
  image: (label: string, n: number) => `the ${label} from @image_${n}`,
55
67
  video: (label: string, n: number) => `the ${label} from @video_${n}`,