@nodaro/prompts 1.9.0 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -37,9 +37,23 @@ export interface CharacterFx {
37
37
  readonly term?: string
38
38
  }
39
39
 
40
- export type CharacterFxPosition = "auto" | "start" | "middle" | "end" | "full"
41
- export type CharacterFxDuration = "auto" | "instant" | "short" | "medium" | "long"
42
- export type CharacterFxIntensity = "auto" | "subtle" | "natural" | "dynamic" | "crazy"
40
+ /**
41
+ * The three timing scales, each derived from the catalog that defines it (see
42
+ * `CHARACTER_FX_POSITIONS` and friends below).
43
+ *
44
+ * The direction matters. These used to be hand-written unions with the clause
45
+ * tables written out separately beside them, so the two could disagree: add a
46
+ * step to the union, forget the clause, and the composer indexed a missing key
47
+ * — pushing `undefined` into the parts list, which `join(", ")` renders as a
48
+ * dangling separator on a prompt that then ships to a provider with the user's
49
+ * chosen parameter silently dropped. Deriving the union FROM the catalog makes
50
+ * that unrepresentable: one array is the source of the type, the option list
51
+ * the API serves, and the clause table, so a new step reaches all three or
52
+ * none. The exact id sets are pinned by `character-fx-timing-catalogs.test.ts`.
53
+ */
54
+ export type CharacterFxPosition = (typeof CHARACTER_FX_POSITIONS)[number]["id"]
55
+ export type CharacterFxDuration = (typeof CHARACTER_FX_DURATIONS)[number]["id"]
56
+ export type CharacterFxIntensity = (typeof CHARACTER_FX_INTENSITIES)[number]["id"]
43
57
 
44
58
  export interface CharacterFxTiming {
45
59
  position?: CharacterFxPosition
@@ -266,27 +280,96 @@ export const CHARACTER_FX_IDS: ReadonlyArray<string> = CHARACTER_FX.map((c) => c
266
280
  // Graph-aware composer — target input handle + timing fields + multi-pick
267
281
  // ---------------------------------------------------------------------------
268
282
 
269
- const POSITION_CLAUSES: Record<Exclude<CharacterFxPosition, "auto">, string> = {
270
- start: "the effect occurs at the opening of the clip",
271
- middle: "the effect occurs in the middle of the clip",
272
- end: "the effect occurs at the end of the clip",
273
- full: "the effect persists for the entire clip",
283
+ /**
284
+ * The character-fx node's three timing parameters, as catalogs.
285
+ *
286
+ * Graded scales in the standard option shape, so a consumer that can only send
287
+ * ids (Studio, the SDK, MCP) can offer Position / Duration / Intensity without
288
+ * composing prompt text of its own. `auto` is the no-op head of each scale: an
289
+ * empty `promptHint`, so an unset parameter contributes nothing and the model
290
+ * is left to decide, exactly as before these were enumerable.
291
+ *
292
+ * These are NOT the transition node's scales, even though the ids match. The
293
+ * wording is deliberately different — a transition OCCURS and SPANS the clip,
294
+ * an effect MANIFESTS and PERSISTS — and the three intensity clauses coincide
295
+ * by accident, not by shared definition. Keep the two catalogs separate; do
296
+ * not fold one into the other.
297
+ *
298
+ * `POSITION_CLAUSES` / `DURATION_CLAUSES` / `INTENSITY_CLAUSES` below are
299
+ * DERIVED from these arrays, so the clause the composer injects and the hint
300
+ * the catalog advertises are the same string by construction and cannot drift.
301
+ */
302
+ export interface CharacterFxTimingOption {
303
+ readonly id: string
304
+ readonly label: string
305
+ readonly description: string
306
+ readonly promptHint: string
307
+ readonly term?: string
274
308
  }
275
309
 
276
- const DURATION_CLAUSES: Record<Exclude<CharacterFxDuration, "auto">, string> = {
277
- instant: "manifesting instantaneously",
278
- short: "manifesting over approximately 1 second",
279
- medium: "manifesting over approximately 2 seconds",
280
- long: "manifesting over approximately 3 seconds",
281
- }
310
+ export const CHARACTER_FX_POSITIONS = [
311
+ { id: "auto", label: "Auto", description: "Let the model place the effect", promptHint: "", term: "" },
312
+ { id: "start", label: "Start", description: "Occurs at the opening of the clip", promptHint: "the effect occurs at the opening of the clip", term: "at the opening of the clip" },
313
+ { id: "middle", label: "Middle", description: "Occurs in the middle of the clip", promptHint: "the effect occurs in the middle of the clip", term: "mid-clip" },
314
+ { id: "end", label: "End", description: "Occurs at the end of the clip", promptHint: "the effect occurs at the end of the clip", term: "at the end of the clip" },
315
+ { id: "full", label: "Full", description: "Persists for the entire clip", promptHint: "the effect persists for the entire clip", term: "persisting for the whole clip" },
316
+ ] as const satisfies ReadonlyArray<CharacterFxTimingOption>
317
+
318
+ export const CHARACTER_FX_DURATIONS = [
319
+ { id: "auto", label: "Auto", description: "Let the model time the effect", promptHint: "", term: "" },
320
+ { id: "instant", label: "Instant", description: "Manifests instantaneously", promptHint: "manifesting instantaneously", term: "manifesting instantly" },
321
+ { id: "short", label: "Short (~1s)", description: "Manifests over approximately 1 second", promptHint: "manifesting over approximately 1 second", term: "manifesting over about 1 second" },
322
+ { id: "medium", label: "Medium (~2s)", description: "Manifests over approximately 2 seconds", promptHint: "manifesting over approximately 2 seconds", term: "manifesting over about 2 seconds" },
323
+ { id: "long", label: "Long (~3s)", description: "Manifests over approximately 3 seconds", promptHint: "manifesting over approximately 3 seconds", term: "manifesting over about 3 seconds" },
324
+ ] as const satisfies ReadonlyArray<CharacterFxTimingOption>
282
325
 
283
- const INTENSITY_CLAUSES: Record<Exclude<CharacterFxIntensity, "auto">, string> = {
284
- subtle: "with subtle restrained energy and minimal flourish",
285
- natural: "with natural unhurried timing",
286
- dynamic: "with dynamic energy and assertive flourish",
287
- crazy: "with extreme exaggerated energy, wild flourishes, and dramatic distortion",
326
+ export const CHARACTER_FX_INTENSITIES = [
327
+ { id: "auto", label: "Auto", description: "Let the model judge the effect's energy", promptHint: "", term: "" },
328
+ { id: "subtle", label: "Subtle", description: "Restrained, minimal flourish", promptHint: "with subtle restrained energy and minimal flourish", term: "subtly" },
329
+ { id: "natural", label: "Natural", description: "Unhurried, unforced timing", promptHint: "with natural unhurried timing", term: "at a natural pace" },
330
+ { id: "dynamic", label: "Dynamic", description: "Assertive, energetic", promptHint: "with dynamic energy and assertive flourish", term: "energetically" },
331
+ { id: "crazy", label: "Crazy", description: "Extreme, wild, distorted", promptHint: "with extreme exaggerated energy, wild flourishes, and dramatic distortion", term: "wildly exaggerated" },
332
+ ] as const satisfies ReadonlyArray<CharacterFxTimingOption>
333
+
334
+ /**
335
+ * Index a timing catalog into the `Record<value, clause>` the composer reads.
336
+ *
337
+ * The key type is derived from the SAME array, so the record is total over the
338
+ * catalog by construction. That matters: the composer indexes these records
339
+ * without a fallback, and a missing key would push `undefined` into the parts
340
+ * list, which `join(", ")` renders as a dangling separator — a malformed prompt
341
+ * shipped to a provider with the user's chosen parameter silently dropped.
342
+ *
343
+ * Deliberately a private twin of the helper in `transitions.ts` rather than a
344
+ * shared import: the two nodes' timing catalogs must stay independent.
345
+ */
346
+ function clausesOf<T extends CharacterFxTimingOption>(
347
+ options: ReadonlyArray<T>,
348
+ ): Record<Exclude<T["id"], "auto">, string> {
349
+ return Object.fromEntries(
350
+ options.filter((o) => o.id !== "auto").map((o) => [o.id, o.promptHint]),
351
+ ) as Record<Exclude<T["id"], "auto">, string>
288
352
  }
289
353
 
354
+ const POSITION_CLAUSES = clausesOf(CHARACTER_FX_POSITIONS)
355
+ const DURATION_CLAUSES = clausesOf(CHARACTER_FX_DURATIONS)
356
+ const INTENSITY_CLAUSES = clausesOf(CHARACTER_FX_INTENSITIES)
357
+
358
+ /**
359
+ * Widening guard. `as const` on the three arrays above is load-bearing: drop
360
+ * it and every id widens to `string`, the derived unions stop constraining
361
+ * anything, and the clause records degrade to `Record<string, string>` — the
362
+ * totality guarantee is gone with no runtime symptom. This assignment stops
363
+ * compiling the moment that happens (tsup's DTS build runs the type checker).
364
+ */
365
+ type NarrowIds<T> = string extends T ? never : true
366
+ const _timingIdsStayNarrow: [
367
+ NarrowIds<CharacterFxPosition>,
368
+ NarrowIds<CharacterFxDuration>,
369
+ NarrowIds<CharacterFxIntensity>,
370
+ ] = [true, true, true]
371
+ void _timingIdsStayNarrow
372
+
290
373
  /**
291
374
  * Compose a character-fx prompt-hint sentence from an effect id (or array
292
375
  * of 1-2 ids for multi-pick) plus target-ref display names (from upstream
@@ -59,7 +59,14 @@ import {
59
59
  TRANSITION_DURATIONS,
60
60
  TRANSITION_INTENSITIES,
61
61
  } from "./transitions.js"
62
- import { CHARACTER_FX, CHARACTER_FX_CATEGORY_LABELS, CHARACTER_FX_CATEGORY_ORDER } from "./character-fx.js"
62
+ import {
63
+ CHARACTER_FX,
64
+ CHARACTER_FX_CATEGORY_LABELS,
65
+ CHARACTER_FX_CATEGORY_ORDER,
66
+ CHARACTER_FX_POSITIONS,
67
+ CHARACTER_FX_DURATIONS,
68
+ CHARACTER_FX_INTENSITIES,
69
+ } from "./character-fx.js"
63
70
  import { POSES, POSE_CATEGORY_LABELS, POSE_CATEGORY_ORDER } from "./pose.js"
64
71
  import { MATERIALS, MATERIAL_CATEGORY_LABELS, MATERIAL_CATEGORY_ORDER } from "./materials.js"
65
72
  import { ANIMALS, ANIMAL_SUBCATEGORY_LABELS, ANIMAL_SUBCATEGORY_ORDER } from "@nodaro/shared"
@@ -557,6 +564,16 @@ const SINGLE_CATALOGS: readonly PickerCatalog[] = [
557
564
  categoryOrder: CHARACTER_FX_CATEGORY_ORDER,
558
565
  categoryLabels: CHARACTER_FX_CATEGORY_LABELS,
559
566
  options: toOptions(CHARACTER_FX, "category"),
567
+ // The node's three timing parameters, alongside the effect itself — the
568
+ // same shape `transition` carries above, but the character-fx scales, not
569
+ // the transition ones: the wording is deliberately different (an effect
570
+ // manifests and persists; a transition occurs and spans), so an id-only
571
+ // consumer must read these rows, never reuse the transition rows.
572
+ dimensions: perFieldDims([
573
+ ["position", CHARACTER_FX_POSITIONS],
574
+ ["duration", CHARACTER_FX_DURATIONS],
575
+ ["intensity", CHARACTER_FX_INTENSITIES],
576
+ ]),
560
577
  },
561
578
 
562
579
  // -------- "Subject / Object" family --------
@@ -73,7 +73,7 @@ precise subject → action details → scene/environment → lighting & color to
73
73
  - Transitions and camera terms on 2.5: state a transition's trigger point AND method in one sentence — "At the 5-second mark, the camera quickly transitions leftward using a left wipe combined with a natural dissolve." Basic shot and camera terms are written directly (push in / pull out / pan / track / orbit / dolly zoom / whip pan / hard cut / dissolve / one-shot / speed ramp); only niche terms need [term + descriptive explanation] — which is exactly what the pickers' compact hint mode emits versus their long hints.
74
74
 
75
75
  **References (when reference media is attached)**
76
- - Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync.)
76
+ - Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync. An API caller that passes \`connectedReferences\` can instead name a reference by its own id — \`{ref:<id>}\` / \`{ref:<id>:label}\` — and the platform substitutes the \`@image_N\` seat after it has numbered the references, so the client never computes N; a token whose reference was not attached drops to its label or name.)
77
77
  - Define each subject once, then reuse the label consistently: 'Define the woman in the red dress in Image 1 as the courier' … 'the courier opens the door'. In multi-character scenes bind every character to its image ("the man from Image 1 hands the box to the woman from Image 2") and append: "do not generate duplicate copies of the same character".
78
78
  - Character identity: ONE close-up headshot + ONE full-body image is ideal. On the 2.0 SKUs do NOT attach multi-view/three-view character sheets — the model reads the views as separate people, causing identity drift and twin duplicates; 2.5 accepts multi-view images (see "Generation differences").
79
79
  - 4-5 assets total works best (1-2 character images + 1 scene image + 1 camera-movement video + 1 audio clip). Maxing out the 9-image/3-video/3-audio limits degrades feature priority and adherence.
@@ -177,7 +177,7 @@ precise subject → action details → scene/environment → lighting & color to
177
177
  - Nothing visual connected → text-to-video. A concrete aspect ratio is required (21:9 / 16:9 / 4:3 / 1:1 / 3:4 / 9:16 — no adaptive); Nodaro renders 16:9 unless one is picked.
178
178
 
179
179
  **References (when reference media is attached)**
180
- - Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync.)
180
+ - Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync. An API caller that passes \`connectedReferences\` can instead write \`{ref:<id>}\` / \`{ref:<id>:label}\` with the reference's own id — the platform substitutes the \`@image_N\` seat after numbering.)
181
181
  - Caps: 9 reference images; 3 reference videos, each 2-15s and ≤15s combined; 3 reference audio clips, ≤15s combined. Reference audio cannot be used alone — it must accompany an image or video reference.
182
182
  - Define each subject once, then reuse the label consistently ("the woman from @Image 1 … the woman opens the door"). A focused set of 4-5 assets beats maxing every cap.
183
183
  - Billing note: generated seconds AND reference-video input seconds bill at the same per-second rate; the first 5 input images are free and each extra image adds a small surcharge; audio input is free.
@@ -0,0 +1,45 @@
1
+ /**
2
+ * The reference-binding surface string — `@image_N` / `@video_N` / `@audio_N`
3
+ * — and the one identity sentence built on it. Split out of
4
+ * `video-reference-resolver.ts` so the id-addressed token resolver
5
+ * (`ref-id-tokens.ts`) can bind through the same arrows without a module
6
+ * cycle; the resolver re-exports both, so importers are unaffected.
7
+ */
8
+
9
+ /**
10
+ * The SINGLE swap-point for the reference-binding surface-string (design D1/D7).
11
+ *
12
+ * Every place that renders an `@image_N`-style binding into a video prompt — the
13
+ * per-image subject phrasing, the bare ordinal in a "Use these characters" /
14
+ * pair-back bullet, and the opening/closing frame directive — MUST go through
15
+ * these five arrows. The default form is `@image_N`; if the D7 probe shows a
16
+ * provider prefers the legacy `Image N` form, flipping is editing ONLY these five
17
+ * arrows (`@image_${n}` → `Image ${n}`), nothing downstream.
18
+ *
19
+ * This IS the live swap-point: `resolveVideoReferenceCore` routes the per-image
20
+ * subject phrasing, the "Use these characters" / pair-back bullet ordinals, and
21
+ * the frame directive through these arrows, and `resolveReferenceTokens` resolves
22
+ * the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
23
+ * are the ONLY emission sites for the binding surface string.
24
+ */
25
+ /**
26
+ * The identity-reference binding sentence shared by the flat-image-list
27
+ * resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
28
+ * says the two things a multimodal model needs to hear — match exactly, and
29
+ * these are not frames. One spelling; both resolvers ride it.
30
+ */
31
+ export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
32
+ return firstOrdinal === lastOrdinal
33
+ ? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
34
+ : `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
35
+ }
36
+
37
+ export const REF_BINDING = {
38
+ image: (label: string, n: number) => `the ${label} from @image_${n}`,
39
+ video: (label: string, n: number) => `the ${label} from @video_${n}`,
40
+ audio: (label: string, n: number) => `the ${label} from @audio_${n}`,
41
+ /** ordinal as it appears in a "Use these characters" bullet / pair-back */
42
+ ordinal: (n: number) => `@image_${n}`,
43
+ frame: (n: number, role: "opening" | "closing") =>
44
+ `Use @image_${n} as the ${role} (${role === "opening" ? "first" : "last"}) frame of the video.`,
45
+ } as const
@@ -0,0 +1,112 @@
1
+ /**
2
+ * `{ref:<id>}` / `{ref:<id>:<label>}` — id-addressed reference tokens.
3
+ *
4
+ * The API/Studio form of the positional `{image:N}` token: the client names a
5
+ * reference by its OWN `connectedReferences[].id` and the platform substitutes
6
+ * the `@image_N` seat after IT has done the numbering. Without it a client that
7
+ * wanted the binding inline had to mirror the numbering walk client-side — a
8
+ * duplicated rule that misbinds pictures the moment the walk changes.
9
+ *
10
+ * `resolveVideoReferenceCore` builds the `RefIdTokenContext` DURING its walk
11
+ * and calls `resolveRefIdTokens` before the `referenceOrder` reorder (so the
12
+ * binding follows the reference to its final seat); the video routes call it
13
+ * standalone on their no-image-reference early return (nothing seated, so
14
+ * every token degrades).
15
+ */
16
+
17
+ import { REF_BINDING } from "./ref-binding.js"
18
+
19
+ /** The label class of `REFERENCE_TOKEN_RE` (`{image:N:label}`), shared. */
20
+ const REF_TOKEN_LABEL_RE = /^[a-zA-Z0-9_ -]+$/
21
+ /** Cheap gate for the whole pass — a prompt without it is untouched. */
22
+ const HAS_REF_ID_TOKEN_RE = /\{ref:/i
23
+ /**
24
+ * One well-formed `{ref:…}` token: everything between `{ref:` and the next
25
+ * `}` that contains no brace. Greedy over a brace-free class, so the scan is
26
+ * LINEAR in the prompt length whatever the content — `prompt` is up to
27
+ * `PROMPT_HARD_CEILING` (30k) chars of caller-controlled text, so a lazy
28
+ * quantifier with a nested optional label group here would be a quadratic-time
29
+ * ReDoS surface. The id / label split happens in code (`splitLabel`), not in
30
+ * the regex. `ref` is case-insensitive; ids are not.
31
+ */
32
+ const REF_ID_TOKEN_RE = /\{[rR][eE][fF]:([^{}]*)\}/g
33
+ /**
34
+ * Last-resort net for a MALFORMED `{ref:` (a brace inside the id, or no
35
+ * closing `}`): drop the `{ref:` run up to the next whitespace or brace, so
36
+ * the prefix can never reach a model, without eating prose past the token.
37
+ */
38
+ const MALFORMED_REF_ID_TOKEN_RE = /\{[rR][eE][fF]:[^\s{}]*\}?/g
39
+
40
+ /** What `resolveRefIdTokens` resolves against — the numbering walk's output. */
41
+ export interface RefIdTokenContext {
42
+ /** Reference id → the 1-based `@image_N` seat the walk gave it. */
43
+ readonly slotById: ReadonlyMap<string, number>
44
+ /** Reference id → display name, the degrade target of a token that cannot bind. */
45
+ readonly nameById: ReadonlyMap<string, string>
46
+ /**
47
+ * How many image references actually ship — the same range gate `{image:N}`
48
+ * uses. A seat past it (a capped-out or duplicate-URL ref) must not bind.
49
+ */
50
+ readonly imageCount: number
51
+ }
52
+
53
+ /**
54
+ * Split a token's content into `<id>` and an optional `<label>` at the LAST
55
+ * colon — only when the tail is a well-formed label. Ids are opaque and may
56
+ * themselves contain `:` (`slug:variant`) or `/` (a URL), so nothing before
57
+ * the last colon is ever interpreted.
58
+ */
59
+ function splitLabel(content: string): { id: string; label?: string } {
60
+ const at = content.lastIndexOf(":")
61
+ if (at === -1) return { id: content }
62
+ const tail = content.slice(at + 1)
63
+ if (!REF_TOKEN_LABEL_RE.test(tail)) return { id: content }
64
+ return { id: content.slice(0, at), label: tail }
65
+ }
66
+
67
+ /**
68
+ * Rewrite id-addressed reference tokens into the `@image_N` binding of the
69
+ * reference the caller sent under that id.
70
+ *
71
+ * Ids are matched by IDENTITY against the known ids (seated or named), never
72
+ * parsed by character class: the whole content is tried as an id first (the
73
+ * longest reading — an id may itself end in something label-shaped), then
74
+ * `<id>:<label>` split at the last colon, then the token is unknown. The label
75
+ * class is the one `REFERENCE_TOKEN_RE` uses. An id containing `{`, `}`, an
76
+ * `@name:N` mention or a `{image:N}` token is unsupported (the mention pass
77
+ * runs first and would rewrite it; a brace ends the token).
78
+ *
79
+ * Per token:
80
+ * - id seated in range → `REF_BINDING.image(label, N)` when labeled, else the
81
+ * bare `REF_BINDING.ordinal(N)` — exactly what `{image:N[:label]}` emits.
82
+ * - otherwise (unknown id, ref skipped by the walk, capped out, or no image
83
+ * references at all) → the label if given, else the ref's display name if
84
+ * the id is known, else "". A token never ships raw — a malformed one is
85
+ * dropped by the last-resort net.
86
+ *
87
+ * No whitespace tidy here: every caller runs `resolveReferenceTokens` after
88
+ * this (the core does at every return), and that collapses the gap a dropped
89
+ * token leaves. Returns the input untouched when it carries no `{ref:` at all.
90
+ */
91
+ export function resolveRefIdTokens(
92
+ prompt: string | undefined,
93
+ ctx: RefIdTokenContext,
94
+ ): string | undefined {
95
+ if (!prompt || !HAS_REF_ID_TOKEN_RE.test(prompt)) return prompt
96
+ const known = (id: string): boolean => id.length > 0 && (ctx.slotById.has(id) || ctx.nameById.has(id))
97
+ const bind = (id: string, label: string | undefined): string => {
98
+ const slot = ctx.slotById.get(id)
99
+ if (slot !== undefined && slot >= 1 && slot <= ctx.imageCount) {
100
+ return label ? REF_BINDING.image(label, slot) : REF_BINDING.ordinal(slot)
101
+ }
102
+ return label ?? ctx.nameById.get(id) ?? ""
103
+ }
104
+ return prompt
105
+ .replace(REF_ID_TOKEN_RE, (_match, content: string) => {
106
+ if (known(content)) return bind(content, undefined)
107
+ const { id, label } = splitLabel(content)
108
+ if (known(id)) return bind(id, label)
109
+ return label ?? ""
110
+ })
111
+ .replace(MALFORMED_REF_ID_TOKEN_RE, "")
112
+ }
@@ -33,44 +33,15 @@ import { resolveCharacterMentions, applyReferenceOrderToVideo } from "./prompt-b
33
33
  import { roleToPhrase, REFERENCE_ROLE_PRESETS, resolveDefaultRole } from "@nodaro/shared"
34
34
  import { buildIdentityLockLine, withForcedIdentityLock } from "./identity-lock.js"
35
35
  import type { ConnectedReference } from "@nodaro/shared"
36
+ import { REF_BINDING } from "./ref-binding.js"
37
+ import { resolveRefIdTokens } from "./ref-id-tokens.js"
36
38
 
37
- /**
38
- * The SINGLE swap-point for the reference-binding surface-string (design D1/D7).
39
- *
40
- * Every place that renders an `@image_N`-style binding into a video prompt — the
41
- * per-image subject phrasing, the bare ordinal in a "Use these characters" /
42
- * pair-back bullet, and the opening/closing frame directive — MUST go through
43
- * these five arrows. The default form is `@image_N`; if the D7 probe shows a
44
- * provider prefers the legacy `Image N` form, flipping is editing ONLY these five
45
- * arrows (`@image_${n}` → `Image ${n}`), nothing downstream.
46
- *
47
- * This IS the live swap-point: `resolveVideoReferenceCore` routes the per-image
48
- * subject phrasing, the "Use these characters" / pair-back bullet ordinals, and
49
- * the frame directive through these arrows, and `resolveReferenceTokens` resolves
50
- * the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
51
- * are the ONLY emission sites for the binding surface string.
52
- */
53
- /**
54
- * The identity-reference binding sentence shared by the flat-image-list
55
- * resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
56
- * says the two things a multimodal model needs to hear — match exactly, and
57
- * these are not frames. One spelling; both resolvers ride it.
58
- */
59
- export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
60
- return firstOrdinal === lastOrdinal
61
- ? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
62
- : `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
63
- }
39
+ // The binding surface string and the id-addressed token resolver live in their
40
+ // own modules (see them for the contracts); re-exported here so every existing
41
+ // importer of this module — and the package index's `export *` — keeps working.
42
+ export { REF_BINDING, identityRefsSentence } from "./ref-binding.js"
43
+ export { resolveRefIdTokens, type RefIdTokenContext } from "./ref-id-tokens.js"
64
44
 
65
- export const REF_BINDING = {
66
- image: (label: string, n: number) => `the ${label} from @image_${n}`,
67
- video: (label: string, n: number) => `the ${label} from @video_${n}`,
68
- audio: (label: string, n: number) => `the ${label} from @audio_${n}`,
69
- /** ordinal as it appears in a "Use these characters" bullet / pair-back */
70
- ordinal: (n: number) => `@image_${n}`,
71
- frame: (n: number, role: "opening" | "closing") =>
72
- `Use @image_${n} as the ${role} (${role === "opening" ? "first" : "last"}) frame of the video.`,
73
- } as const
74
45
 
75
46
  /**
76
47
  * Positional reference counts the editor tokens are resolved against — how many
@@ -135,11 +106,19 @@ export function resolveReferenceTokens(
135
106
  )
136
107
  }
137
108
 
109
+
138
110
  /**
139
111
  * A user-attached "extra reference image" row. Layer-agnostic shape of the
140
112
  * frontend `ExtraRef` / backend extras: only the fields this core reads.
141
113
  */
142
114
  export interface VideoExtraRef {
115
+ /**
116
+ * The caller's own id for this reference (`connectedReferences[].id` on the
117
+ * route, `extraRefs[].id` on the canvas) — what a `{ref:<id>}` token in the
118
+ * prompt names. Slot-map only: the reorder's tile id stays `wired:<url>`.
119
+ * Absent → the extra cannot be addressed by id (it still numbers normally).
120
+ */
121
+ id?: string
143
122
  url: string
144
123
  description?: string
145
124
  characterSlug?: string
@@ -241,6 +220,15 @@ export interface ResolveVideoReferenceCoreArgs {
241
220
  * Wired in Phase B Tasks 2-3.
242
221
  */
243
222
  hybridRoles?: boolean
223
+ /**
224
+ * Display names for EVERY reference the caller knows by id — including the
225
+ * ones it did NOT hand to this walk (the route caps `connectedReferences` to
226
+ * the provider's image budget before calling in). A `{ref:<id>}` token that
227
+ * cannot bind degrades to `label ?? refNamesById[id] ?? ""`, so a capped-out
228
+ * or skipped reference keeps its name in the prose instead of vanishing.
229
+ * The wired character refs' `defaultName`s are known without this.
230
+ */
231
+ refNamesById?: ReadonlyMap<string, string>
244
232
  }
245
233
 
246
234
  /** Result of the HYBRID mention pass — inline role phrases + surfaced opt-in
@@ -455,16 +443,39 @@ export function resolveVideoReferenceCore(
455
443
  // early-return below is gated on (no chars AND no extras) so we don't skip
456
444
  // extras-only setups.
457
445
  const hasExtras = (args.extraRefs?.length ?? 0) > 0
446
+ // `{ref:<id>}` degrade names — every id this call knows: the wired refs'
447
+ // display names, the extras' ids (name-less, so they match by identity rather
448
+ // than falling to the catch-all), then the caller's own map on top (it knows
449
+ // the refs it capped out before calling in).
450
+ const nameByRefId = new Map<string, string>()
451
+ for (const r of args.wiredCharRefs) {
452
+ if (r.id && !nameByRefId.has(r.id)) nameByRefId.set(r.id, r.defaultName || r.characterSlug || "")
453
+ }
454
+ for (const ex of args.extraRefs ?? []) {
455
+ if (ex.id && !nameByRefId.has(ex.id)) nameByRefId.set(ex.id, "")
456
+ }
457
+ for (const [id, name] of args.refNamesById ?? []) {
458
+ if (id) nameByRefId.set(id, name)
459
+ }
460
+ // id → seat, recorded by the walk below as `position` advances — never
461
+ // recovered from URLs afterwards (the walk counts a duplicate-URL extra that
462
+ // `merged` dedups, so URL → index is ambiguous; id → position is not).
463
+ const slotByRefId = new Map<string, number>()
458
464
  if (wiredCharRefs.length === 0 && !hasExtras) {
459
465
  // No wired chars / extras, but the node can still carry plain base reference
460
466
  // images (leadingRefUrls), so `{image:N}` body tokens MUST still resolve. The
461
467
  // count is the leading-ref count (or the legacy `imageRefCount` when no
462
468
  // leading refs were passed); the leading URLs are returned for the payload.
469
+ // Nothing was seated, so every `{ref:}` degrades (label → name → "").
470
+ const counts = tokenCounts(leadingRefUrls.length)
463
471
  return {
464
472
  // tokenCounts(leadingRefUrls.length) → image count == offset (no assets here):
465
473
  // leadingRefUrls mode counts the leading refs; ordinalOffset mode counts the
466
474
  // caller-owned leading refs the offset stands in for.
467
- prompt: resolveReferenceTokens(args.prompt, tokenCounts(leadingRefUrls.length)),
475
+ prompt: resolveReferenceTokens(
476
+ resolveRefIdTokens(args.prompt, { slotById: slotByRefId, nameById: nameByRefId, imageCount: counts.image }),
477
+ counts,
478
+ ),
468
479
  additionalUrls: [...leadingRefUrls],
469
480
  }
470
481
  }
@@ -531,10 +542,17 @@ export function resolveVideoReferenceCore(
531
542
  let position = offset
532
543
  for (let i = 0; i < resolved.additionalUrls.length; i++) {
533
544
  position += 1
545
+ const url = resolved.additionalUrls[i]
534
546
  // Look up which ref this URL came from to learn its characterSlug.
535
- const ref = wiredCharRefs.find((r) => r.url === resolved.additionalUrls[i])
547
+ const ref = wiredCharRefs.find((r) => r.url === url)
536
548
  const slug = ref?.characterSlug
537
549
  if (slug && !positionsByChar.has(slug)) positionsByChar.set(slug, position)
550
+ // `{ref:<id>}`: every wired ref sharing this URL sits in this seat (the
551
+ // merge dedups by URL). First sight wins — the legacy mention pass does
552
+ // not dedup, so a re-mentioned URL advances `position` but keeps its seat.
553
+ for (const r of wiredCharRefs) {
554
+ if (r.url === url && r.id && !slotByRefId.has(r.id)) slotByRefId.set(r.id, position)
555
+ }
538
556
  }
539
557
  for (const r of wiredCharRefs) {
540
558
  if (r.source !== "wired-character") continue
@@ -547,6 +565,7 @@ export function resolveVideoReferenceCore(
547
565
  fallbackUrls.push(r.url)
548
566
  position += 1
549
567
  if (!positionsByChar.has(r.characterSlug)) positionsByChar.set(r.characterSlug, position)
568
+ if (r.id && !slotByRefId.has(r.id)) slotByRefId.set(r.id, position)
550
569
  // Hybrid: emit the inline role phrase (`the person from @image_N`) + opt-in
551
570
  // lock + wired element injection instead of a "Use these characters:"
552
571
  // bullet. The selection above (canonical entry only, deduped, skip mentioned)
@@ -616,6 +635,7 @@ export function resolveVideoReferenceCore(
616
635
  for (const ex of args.extraRefs!) {
617
636
  if (!ex.url) continue
618
637
  position += 1
638
+ if (ex.id && !slotByRefId.has(ex.id)) slotByRefId.set(ex.id, position)
619
639
  const desc = (ex.description ?? "").trim()
620
640
  if (ex.characterSlug) {
621
641
  // First sight of this character via an extra. Resolution chain
@@ -797,6 +817,23 @@ export function resolveVideoReferenceCore(
797
817
  if (u && !seen.has(u)) { seen.add(u); merged.push(u) }
798
818
  }
799
819
 
820
+ // `{ref:<id>}` tokens resolve HERE — after the walk has seated every reference
821
+ // (the slot map is complete) and BEFORE the user reorder below, so the
822
+ // reorder's `@image_N` renumber pass carries the freshly emitted binding to
823
+ // the ref's final seat. That is the point of an id token: "this reference,
824
+ // wherever it lands". `{image:N}` is deliberately the opposite — resolved
825
+ // LAST, after the reorder, so the author's literal N is kept (see the note at
826
+ // the reorder return). Range-gated by the same image count `{image:N}` uses,
827
+ // so a seat the payload never ships (capped out, duplicate URL) degrades to
828
+ // the name instead of binding a phantom `@image_N`. A prompt with no `{ref:`
829
+ // is untouched — byte-identical to before this token existed.
830
+ finalPrompt =
831
+ resolveRefIdTokens(finalPrompt, {
832
+ slotById: slotByRefId,
833
+ nameById: nameByRefId,
834
+ imageCount: tokenCounts(merged.length).image,
835
+ }) ?? finalPrompt
836
+
800
837
  // Apply user-defined reorder + renumber `Image N` tokens — parity with the
801
838
  // backend `resolveVideoPromptMentions` and the shared image builder.
802
839
  const referenceOrder = args.referenceOrder
@@ -825,7 +862,9 @@ export function resolveVideoReferenceCore(
825
862
  // Resolve body tokens LAST — AFTER the reorder's `@image_N` renumber pass, so
826
863
  // it can't miscorrect a freshly-resolved binding (the curly `{image:N}` tokens
827
864
  // are invisible to the reorder's `(@image_|Image )` regex, so they ride through
828
- // untouched and keep their author-typed N — documented v1 behavior).
865
+ // untouched and keep their author-typed N — documented v1 behavior). The
866
+ // id-addressed `{ref:<id>}` tokens are the deliberate opposite: resolved
867
+ // BEFORE the reorder (above), so their binding follows the reference.
829
868
  return {
830
869
  prompt: resolveReferenceTokens(reordered.prompt, tokenCounts(merged.length)),
831
870
  additionalUrls: [...leadingRefUrls, ...reordered.urls],