@nodaro/prompts 1.9.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +108 -23
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +241 -31
- package/dist/index.d.ts +241 -31
- package/dist/index.js +105 -24
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/__tests__/character-fx-timing-catalogs.test.ts +240 -0
- package/src/__tests__/transition-timing-catalogs.test.ts +3 -1
- package/src/__tests__/video-reference-ref-id-tokens.test.ts +301 -0
- package/src/character-fx.ts +102 -19
- package/src/picker-catalogs.ts +18 -1
- package/src/provider-prompt-doctrine.ts +2 -2
- package/src/ref-binding.ts +45 -0
- package/src/ref-id-tokens.ts +112 -0
- package/src/video-reference-resolver.ts +78 -39
package/src/character-fx.ts
CHANGED
|
@@ -37,9 +37,23 @@ export interface CharacterFx {
|
|
|
37
37
|
readonly term?: string
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
40
|
+
/**
|
|
41
|
+
* The three timing scales, each derived from the catalog that defines it (see
|
|
42
|
+
* `CHARACTER_FX_POSITIONS` and friends below).
|
|
43
|
+
*
|
|
44
|
+
* The direction matters. These used to be hand-written unions with the clause
|
|
45
|
+
* tables written out separately beside them, so the two could disagree: add a
|
|
46
|
+
* step to the union, forget the clause, and the composer indexed a missing key
|
|
47
|
+
* — pushing `undefined` into the parts list, which `join(", ")` renders as a
|
|
48
|
+
* dangling separator on a prompt that then ships to a provider with the user's
|
|
49
|
+
* chosen parameter silently dropped. Deriving the union FROM the catalog makes
|
|
50
|
+
* that unrepresentable: one array is the source of the type, the option list
|
|
51
|
+
* the API serves, and the clause table, so a new step reaches all three or
|
|
52
|
+
* none. The exact id sets are pinned by `character-fx-timing-catalogs.test.ts`.
|
|
53
|
+
*/
|
|
54
|
+
export type CharacterFxPosition = (typeof CHARACTER_FX_POSITIONS)[number]["id"]
|
|
55
|
+
export type CharacterFxDuration = (typeof CHARACTER_FX_DURATIONS)[number]["id"]
|
|
56
|
+
export type CharacterFxIntensity = (typeof CHARACTER_FX_INTENSITIES)[number]["id"]
|
|
43
57
|
|
|
44
58
|
export interface CharacterFxTiming {
|
|
45
59
|
position?: CharacterFxPosition
|
|
@@ -266,27 +280,96 @@ export const CHARACTER_FX_IDS: ReadonlyArray<string> = CHARACTER_FX.map((c) => c
|
|
|
266
280
|
// Graph-aware composer — target input handle + timing fields + multi-pick
|
|
267
281
|
// ---------------------------------------------------------------------------
|
|
268
282
|
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
283
|
+
/**
|
|
284
|
+
* The character-fx node's three timing parameters, as catalogs.
|
|
285
|
+
*
|
|
286
|
+
* Graded scales in the standard option shape, so a consumer that can only send
|
|
287
|
+
* ids (Studio, the SDK, MCP) can offer Position / Duration / Intensity without
|
|
288
|
+
* composing prompt text of its own. `auto` is the no-op head of each scale: an
|
|
289
|
+
* empty `promptHint`, so an unset parameter contributes nothing and the model
|
|
290
|
+
* is left to decide, exactly as before these were enumerable.
|
|
291
|
+
*
|
|
292
|
+
* These are NOT the transition node's scales, even though the ids match. The
|
|
293
|
+
* wording is deliberately different — a transition OCCURS and SPANS the clip,
|
|
294
|
+
* an effect MANIFESTS and PERSISTS — and the three intensity clauses coincide
|
|
295
|
+
* by accident, not by shared definition. Keep the two catalogs separate; do
|
|
296
|
+
* not fold one into the other.
|
|
297
|
+
*
|
|
298
|
+
* `POSITION_CLAUSES` / `DURATION_CLAUSES` / `INTENSITY_CLAUSES` below are
|
|
299
|
+
* DERIVED from these arrays, so the clause the composer injects and the hint
|
|
300
|
+
* the catalog advertises are the same string by construction and cannot drift.
|
|
301
|
+
*/
|
|
302
|
+
export interface CharacterFxTimingOption {
|
|
303
|
+
readonly id: string
|
|
304
|
+
readonly label: string
|
|
305
|
+
readonly description: string
|
|
306
|
+
readonly promptHint: string
|
|
307
|
+
readonly term?: string
|
|
274
308
|
}
|
|
275
309
|
|
|
276
|
-
const
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
}
|
|
310
|
+
export const CHARACTER_FX_POSITIONS = [
|
|
311
|
+
{ id: "auto", label: "Auto", description: "Let the model place the effect", promptHint: "", term: "" },
|
|
312
|
+
{ id: "start", label: "Start", description: "Occurs at the opening of the clip", promptHint: "the effect occurs at the opening of the clip", term: "at the opening of the clip" },
|
|
313
|
+
{ id: "middle", label: "Middle", description: "Occurs in the middle of the clip", promptHint: "the effect occurs in the middle of the clip", term: "mid-clip" },
|
|
314
|
+
{ id: "end", label: "End", description: "Occurs at the end of the clip", promptHint: "the effect occurs at the end of the clip", term: "at the end of the clip" },
|
|
315
|
+
{ id: "full", label: "Full", description: "Persists for the entire clip", promptHint: "the effect persists for the entire clip", term: "persisting for the whole clip" },
|
|
316
|
+
] as const satisfies ReadonlyArray<CharacterFxTimingOption>
|
|
317
|
+
|
|
318
|
+
export const CHARACTER_FX_DURATIONS = [
|
|
319
|
+
{ id: "auto", label: "Auto", description: "Let the model time the effect", promptHint: "", term: "" },
|
|
320
|
+
{ id: "instant", label: "Instant", description: "Manifests instantaneously", promptHint: "manifesting instantaneously", term: "manifesting instantly" },
|
|
321
|
+
{ id: "short", label: "Short (~1s)", description: "Manifests over approximately 1 second", promptHint: "manifesting over approximately 1 second", term: "manifesting over about 1 second" },
|
|
322
|
+
{ id: "medium", label: "Medium (~2s)", description: "Manifests over approximately 2 seconds", promptHint: "manifesting over approximately 2 seconds", term: "manifesting over about 2 seconds" },
|
|
323
|
+
{ id: "long", label: "Long (~3s)", description: "Manifests over approximately 3 seconds", promptHint: "manifesting over approximately 3 seconds", term: "manifesting over about 3 seconds" },
|
|
324
|
+
] as const satisfies ReadonlyArray<CharacterFxTimingOption>
|
|
282
325
|
|
|
283
|
-
const
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
326
|
+
export const CHARACTER_FX_INTENSITIES = [
|
|
327
|
+
{ id: "auto", label: "Auto", description: "Let the model judge the effect's energy", promptHint: "", term: "" },
|
|
328
|
+
{ id: "subtle", label: "Subtle", description: "Restrained, minimal flourish", promptHint: "with subtle restrained energy and minimal flourish", term: "subtly" },
|
|
329
|
+
{ id: "natural", label: "Natural", description: "Unhurried, unforced timing", promptHint: "with natural unhurried timing", term: "at a natural pace" },
|
|
330
|
+
{ id: "dynamic", label: "Dynamic", description: "Assertive, energetic", promptHint: "with dynamic energy and assertive flourish", term: "energetically" },
|
|
331
|
+
{ id: "crazy", label: "Crazy", description: "Extreme, wild, distorted", promptHint: "with extreme exaggerated energy, wild flourishes, and dramatic distortion", term: "wildly exaggerated" },
|
|
332
|
+
] as const satisfies ReadonlyArray<CharacterFxTimingOption>
|
|
333
|
+
|
|
334
|
+
/**
|
|
335
|
+
* Index a timing catalog into the `Record<value, clause>` the composer reads.
|
|
336
|
+
*
|
|
337
|
+
* The key type is derived from the SAME array, so the record is total over the
|
|
338
|
+
* catalog by construction. That matters: the composer indexes these records
|
|
339
|
+
* without a fallback, and a missing key would push `undefined` into the parts
|
|
340
|
+
* list, which `join(", ")` renders as a dangling separator — a malformed prompt
|
|
341
|
+
* shipped to a provider with the user's chosen parameter silently dropped.
|
|
342
|
+
*
|
|
343
|
+
* Deliberately a private twin of the helper in `transitions.ts` rather than a
|
|
344
|
+
* shared import: the two nodes' timing catalogs must stay independent.
|
|
345
|
+
*/
|
|
346
|
+
function clausesOf<T extends CharacterFxTimingOption>(
|
|
347
|
+
options: ReadonlyArray<T>,
|
|
348
|
+
): Record<Exclude<T["id"], "auto">, string> {
|
|
349
|
+
return Object.fromEntries(
|
|
350
|
+
options.filter((o) => o.id !== "auto").map((o) => [o.id, o.promptHint]),
|
|
351
|
+
) as Record<Exclude<T["id"], "auto">, string>
|
|
288
352
|
}
|
|
289
353
|
|
|
354
|
+
const POSITION_CLAUSES = clausesOf(CHARACTER_FX_POSITIONS)
|
|
355
|
+
const DURATION_CLAUSES = clausesOf(CHARACTER_FX_DURATIONS)
|
|
356
|
+
const INTENSITY_CLAUSES = clausesOf(CHARACTER_FX_INTENSITIES)
|
|
357
|
+
|
|
358
|
+
/**
|
|
359
|
+
* Widening guard. `as const` on the three arrays above is load-bearing: drop
|
|
360
|
+
* it and every id widens to `string`, the derived unions stop constraining
|
|
361
|
+
* anything, and the clause records degrade to `Record<string, string>` — the
|
|
362
|
+
* totality guarantee is gone with no runtime symptom. This assignment stops
|
|
363
|
+
* compiling the moment that happens (tsup's DTS build runs the type checker).
|
|
364
|
+
*/
|
|
365
|
+
type NarrowIds<T> = string extends T ? never : true
|
|
366
|
+
const _timingIdsStayNarrow: [
|
|
367
|
+
NarrowIds<CharacterFxPosition>,
|
|
368
|
+
NarrowIds<CharacterFxDuration>,
|
|
369
|
+
NarrowIds<CharacterFxIntensity>,
|
|
370
|
+
] = [true, true, true]
|
|
371
|
+
void _timingIdsStayNarrow
|
|
372
|
+
|
|
290
373
|
/**
|
|
291
374
|
* Compose a character-fx prompt-hint sentence from an effect id (or array
|
|
292
375
|
* of 1-2 ids for multi-pick) plus target-ref display names (from upstream
|
package/src/picker-catalogs.ts
CHANGED
|
@@ -59,7 +59,14 @@ import {
|
|
|
59
59
|
TRANSITION_DURATIONS,
|
|
60
60
|
TRANSITION_INTENSITIES,
|
|
61
61
|
} from "./transitions.js"
|
|
62
|
-
import {
|
|
62
|
+
import {
|
|
63
|
+
CHARACTER_FX,
|
|
64
|
+
CHARACTER_FX_CATEGORY_LABELS,
|
|
65
|
+
CHARACTER_FX_CATEGORY_ORDER,
|
|
66
|
+
CHARACTER_FX_POSITIONS,
|
|
67
|
+
CHARACTER_FX_DURATIONS,
|
|
68
|
+
CHARACTER_FX_INTENSITIES,
|
|
69
|
+
} from "./character-fx.js"
|
|
63
70
|
import { POSES, POSE_CATEGORY_LABELS, POSE_CATEGORY_ORDER } from "./pose.js"
|
|
64
71
|
import { MATERIALS, MATERIAL_CATEGORY_LABELS, MATERIAL_CATEGORY_ORDER } from "./materials.js"
|
|
65
72
|
import { ANIMALS, ANIMAL_SUBCATEGORY_LABELS, ANIMAL_SUBCATEGORY_ORDER } from "@nodaro/shared"
|
|
@@ -557,6 +564,16 @@ const SINGLE_CATALOGS: readonly PickerCatalog[] = [
|
|
|
557
564
|
categoryOrder: CHARACTER_FX_CATEGORY_ORDER,
|
|
558
565
|
categoryLabels: CHARACTER_FX_CATEGORY_LABELS,
|
|
559
566
|
options: toOptions(CHARACTER_FX, "category"),
|
|
567
|
+
// The node's three timing parameters, alongside the effect itself — the
|
|
568
|
+
// same shape `transition` carries above, but the character-fx scales, not
|
|
569
|
+
// the transition ones: the wording is deliberately different (an effect
|
|
570
|
+
// manifests and persists; a transition occurs and spans), so an id-only
|
|
571
|
+
// consumer must read these rows, never reuse the transition rows.
|
|
572
|
+
dimensions: perFieldDims([
|
|
573
|
+
["position", CHARACTER_FX_POSITIONS],
|
|
574
|
+
["duration", CHARACTER_FX_DURATIONS],
|
|
575
|
+
["intensity", CHARACTER_FX_INTENSITIES],
|
|
576
|
+
]),
|
|
560
577
|
},
|
|
561
578
|
|
|
562
579
|
// -------- "Subject / Object" family --------
|
|
@@ -73,7 +73,7 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
73
73
|
- Transitions and camera terms on 2.5: state a transition's trigger point AND method in one sentence — "At the 5-second mark, the camera quickly transitions leftward using a left wipe combined with a natural dissolve." Basic shot and camera terms are written directly (push in / pull out / pan / track / orbit / dolly zoom / whip pan / hard cut / dissolve / one-shot / speed ramp); only niche terms need [term + descriptive explanation] — which is exactly what the pickers' compact hint mode emits versus their long hints.
|
|
74
74
|
|
|
75
75
|
**References (when reference media is attached)**
|
|
76
|
-
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync.)
|
|
76
|
+
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync. An API caller that passes \`connectedReferences\` can instead name a reference by its own id — \`{ref:<id>}\` / \`{ref:<id>:label}\` — and the platform substitutes the \`@image_N\` seat after it has numbered the references, so the client never computes N; a token whose reference was not attached drops to its label or name.)
|
|
77
77
|
- Define each subject once, then reuse the label consistently: 'Define the woman in the red dress in Image 1 as the courier' … 'the courier opens the door'. In multi-character scenes bind every character to its image ("the man from Image 1 hands the box to the woman from Image 2") and append: "do not generate duplicate copies of the same character".
|
|
78
78
|
- Character identity: ONE close-up headshot + ONE full-body image is ideal. On the 2.0 SKUs do NOT attach multi-view/three-view character sheets — the model reads the views as separate people, causing identity drift and twin duplicates; 2.5 accepts multi-view images (see "Generation differences").
|
|
79
79
|
- 4-5 assets total works best (1-2 character images + 1 scene image + 1 camera-movement video + 1 audio clip). Maxing out the 9-image/3-video/3-audio limits degrades feature priority and adherence.
|
|
@@ -177,7 +177,7 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
177
177
|
- Nothing visual connected → text-to-video. A concrete aspect ratio is required (21:9 / 16:9 / 4:3 / 1:1 / 3:4 / 9:16 — no adaptive); Nodaro renders 16:9 unless one is picked.
|
|
178
178
|
|
|
179
179
|
**References (when reference media is attached)**
|
|
180
|
-
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync.)
|
|
180
|
+
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync. An API caller that passes \`connectedReferences\` can instead write \`{ref:<id>}\` / \`{ref:<id>:label}\` with the reference's own id — the platform substitutes the \`@image_N\` seat after numbering.)
|
|
181
181
|
- Caps: 9 reference images; 3 reference videos, each 2-15s and ≤15s combined; 3 reference audio clips, ≤15s combined. Reference audio cannot be used alone — it must accompany an image or video reference.
|
|
182
182
|
- Define each subject once, then reuse the label consistently ("the woman from @Image 1 … the woman opens the door"). A focused set of 4-5 assets beats maxing every cap.
|
|
183
183
|
- Billing note: generated seconds AND reference-video input seconds bill at the same per-second rate; the first 5 input images are free and each extra image adds a small surcharge; audio input is free.
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The reference-binding surface string — `@image_N` / `@video_N` / `@audio_N`
|
|
3
|
+
* — and the one identity sentence built on it. Split out of
|
|
4
|
+
* `video-reference-resolver.ts` so the id-addressed token resolver
|
|
5
|
+
* (`ref-id-tokens.ts`) can bind through the same arrows without a module
|
|
6
|
+
* cycle; the resolver re-exports both, so importers are unaffected.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* The SINGLE swap-point for the reference-binding surface-string (design D1/D7).
|
|
11
|
+
*
|
|
12
|
+
* Every place that renders an `@image_N`-style binding into a video prompt — the
|
|
13
|
+
* per-image subject phrasing, the bare ordinal in a "Use these characters" /
|
|
14
|
+
* pair-back bullet, and the opening/closing frame directive — MUST go through
|
|
15
|
+
* these five arrows. The default form is `@image_N`; if the D7 probe shows a
|
|
16
|
+
* provider prefers the legacy `Image N` form, flipping is editing ONLY these five
|
|
17
|
+
* arrows (`@image_${n}` → `Image ${n}`), nothing downstream.
|
|
18
|
+
*
|
|
19
|
+
* This IS the live swap-point: `resolveVideoReferenceCore` routes the per-image
|
|
20
|
+
* subject phrasing, the "Use these characters" / pair-back bullet ordinals, and
|
|
21
|
+
* the frame directive through these arrows, and `resolveReferenceTokens` resolves
|
|
22
|
+
* the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
|
|
23
|
+
* are the ONLY emission sites for the binding surface string.
|
|
24
|
+
*/
|
|
25
|
+
/**
|
|
26
|
+
* The identity-reference binding sentence shared by the flat-image-list
|
|
27
|
+
* resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
|
|
28
|
+
* says the two things a multimodal model needs to hear — match exactly, and
|
|
29
|
+
* these are not frames. One spelling; both resolvers ride it.
|
|
30
|
+
*/
|
|
31
|
+
export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
|
|
32
|
+
return firstOrdinal === lastOrdinal
|
|
33
|
+
? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
|
|
34
|
+
: `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export const REF_BINDING = {
|
|
38
|
+
image: (label: string, n: number) => `the ${label} from @image_${n}`,
|
|
39
|
+
video: (label: string, n: number) => `the ${label} from @video_${n}`,
|
|
40
|
+
audio: (label: string, n: number) => `the ${label} from @audio_${n}`,
|
|
41
|
+
/** ordinal as it appears in a "Use these characters" bullet / pair-back */
|
|
42
|
+
ordinal: (n: number) => `@image_${n}`,
|
|
43
|
+
frame: (n: number, role: "opening" | "closing") =>
|
|
44
|
+
`Use @image_${n} as the ${role} (${role === "opening" ? "first" : "last"}) frame of the video.`,
|
|
45
|
+
} as const
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `{ref:<id>}` / `{ref:<id>:<label>}` — id-addressed reference tokens.
|
|
3
|
+
*
|
|
4
|
+
* The API/Studio form of the positional `{image:N}` token: the client names a
|
|
5
|
+
* reference by its OWN `connectedReferences[].id` and the platform substitutes
|
|
6
|
+
* the `@image_N` seat after IT has done the numbering. Without it a client that
|
|
7
|
+
* wanted the binding inline had to mirror the numbering walk client-side — a
|
|
8
|
+
* duplicated rule that misbinds pictures the moment the walk changes.
|
|
9
|
+
*
|
|
10
|
+
* `resolveVideoReferenceCore` builds the `RefIdTokenContext` DURING its walk
|
|
11
|
+
* and calls `resolveRefIdTokens` before the `referenceOrder` reorder (so the
|
|
12
|
+
* binding follows the reference to its final seat); the video routes call it
|
|
13
|
+
* standalone on their no-image-reference early return (nothing seated, so
|
|
14
|
+
* every token degrades).
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { REF_BINDING } from "./ref-binding.js"
|
|
18
|
+
|
|
19
|
+
/** The label class of `REFERENCE_TOKEN_RE` (`{image:N:label}`), shared. */
|
|
20
|
+
const REF_TOKEN_LABEL_RE = /^[a-zA-Z0-9_ -]+$/
|
|
21
|
+
/** Cheap gate for the whole pass — a prompt without it is untouched. */
|
|
22
|
+
const HAS_REF_ID_TOKEN_RE = /\{ref:/i
|
|
23
|
+
/**
|
|
24
|
+
* One well-formed `{ref:…}` token: everything between `{ref:` and the next
|
|
25
|
+
* `}` that contains no brace. Greedy over a brace-free class, so the scan is
|
|
26
|
+
* LINEAR in the prompt length whatever the content — `prompt` is up to
|
|
27
|
+
* `PROMPT_HARD_CEILING` (30k) chars of caller-controlled text, so a lazy
|
|
28
|
+
* quantifier with a nested optional label group here would be a quadratic-time
|
|
29
|
+
* ReDoS surface. The id / label split happens in code (`splitLabel`), not in
|
|
30
|
+
* the regex. `ref` is case-insensitive; ids are not.
|
|
31
|
+
*/
|
|
32
|
+
const REF_ID_TOKEN_RE = /\{[rR][eE][fF]:([^{}]*)\}/g
|
|
33
|
+
/**
|
|
34
|
+
* Last-resort net for a MALFORMED `{ref:` (a brace inside the id, or no
|
|
35
|
+
* closing `}`): drop the `{ref:` run up to the next whitespace or brace, so
|
|
36
|
+
* the prefix can never reach a model, without eating prose past the token.
|
|
37
|
+
*/
|
|
38
|
+
const MALFORMED_REF_ID_TOKEN_RE = /\{[rR][eE][fF]:[^\s{}]*\}?/g
|
|
39
|
+
|
|
40
|
+
/** What `resolveRefIdTokens` resolves against — the numbering walk's output. */
|
|
41
|
+
export interface RefIdTokenContext {
|
|
42
|
+
/** Reference id → the 1-based `@image_N` seat the walk gave it. */
|
|
43
|
+
readonly slotById: ReadonlyMap<string, number>
|
|
44
|
+
/** Reference id → display name, the degrade target of a token that cannot bind. */
|
|
45
|
+
readonly nameById: ReadonlyMap<string, string>
|
|
46
|
+
/**
|
|
47
|
+
* How many image references actually ship — the same range gate `{image:N}`
|
|
48
|
+
* uses. A seat past it (a capped-out or duplicate-URL ref) must not bind.
|
|
49
|
+
*/
|
|
50
|
+
readonly imageCount: number
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Split a token's content into `<id>` and an optional `<label>` at the LAST
|
|
55
|
+
* colon — only when the tail is a well-formed label. Ids are opaque and may
|
|
56
|
+
* themselves contain `:` (`slug:variant`) or `/` (a URL), so nothing before
|
|
57
|
+
* the last colon is ever interpreted.
|
|
58
|
+
*/
|
|
59
|
+
function splitLabel(content: string): { id: string; label?: string } {
|
|
60
|
+
const at = content.lastIndexOf(":")
|
|
61
|
+
if (at === -1) return { id: content }
|
|
62
|
+
const tail = content.slice(at + 1)
|
|
63
|
+
if (!REF_TOKEN_LABEL_RE.test(tail)) return { id: content }
|
|
64
|
+
return { id: content.slice(0, at), label: tail }
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Rewrite id-addressed reference tokens into the `@image_N` binding of the
|
|
69
|
+
* reference the caller sent under that id.
|
|
70
|
+
*
|
|
71
|
+
* Ids are matched by IDENTITY against the known ids (seated or named), never
|
|
72
|
+
* parsed by character class: the whole content is tried as an id first (the
|
|
73
|
+
* longest reading — an id may itself end in something label-shaped), then
|
|
74
|
+
* `<id>:<label>` split at the last colon, then the token is unknown. The label
|
|
75
|
+
* class is the one `REFERENCE_TOKEN_RE` uses. An id containing `{`, `}`, an
|
|
76
|
+
* `@name:N` mention or a `{image:N}` token is unsupported (the mention pass
|
|
77
|
+
* runs first and would rewrite it; a brace ends the token).
|
|
78
|
+
*
|
|
79
|
+
* Per token:
|
|
80
|
+
* - id seated in range → `REF_BINDING.image(label, N)` when labeled, else the
|
|
81
|
+
* bare `REF_BINDING.ordinal(N)` — exactly what `{image:N[:label]}` emits.
|
|
82
|
+
* - otherwise (unknown id, ref skipped by the walk, capped out, or no image
|
|
83
|
+
* references at all) → the label if given, else the ref's display name if
|
|
84
|
+
* the id is known, else "". A token never ships raw — a malformed one is
|
|
85
|
+
* dropped by the last-resort net.
|
|
86
|
+
*
|
|
87
|
+
* No whitespace tidy here: every caller runs `resolveReferenceTokens` after
|
|
88
|
+
* this (the core does at every return), and that collapses the gap a dropped
|
|
89
|
+
* token leaves. Returns the input untouched when it carries no `{ref:` at all.
|
|
90
|
+
*/
|
|
91
|
+
export function resolveRefIdTokens(
|
|
92
|
+
prompt: string | undefined,
|
|
93
|
+
ctx: RefIdTokenContext,
|
|
94
|
+
): string | undefined {
|
|
95
|
+
if (!prompt || !HAS_REF_ID_TOKEN_RE.test(prompt)) return prompt
|
|
96
|
+
const known = (id: string): boolean => id.length > 0 && (ctx.slotById.has(id) || ctx.nameById.has(id))
|
|
97
|
+
const bind = (id: string, label: string | undefined): string => {
|
|
98
|
+
const slot = ctx.slotById.get(id)
|
|
99
|
+
if (slot !== undefined && slot >= 1 && slot <= ctx.imageCount) {
|
|
100
|
+
return label ? REF_BINDING.image(label, slot) : REF_BINDING.ordinal(slot)
|
|
101
|
+
}
|
|
102
|
+
return label ?? ctx.nameById.get(id) ?? ""
|
|
103
|
+
}
|
|
104
|
+
return prompt
|
|
105
|
+
.replace(REF_ID_TOKEN_RE, (_match, content: string) => {
|
|
106
|
+
if (known(content)) return bind(content, undefined)
|
|
107
|
+
const { id, label } = splitLabel(content)
|
|
108
|
+
if (known(id)) return bind(id, label)
|
|
109
|
+
return label ?? ""
|
|
110
|
+
})
|
|
111
|
+
.replace(MALFORMED_REF_ID_TOKEN_RE, "")
|
|
112
|
+
}
|
|
@@ -33,44 +33,15 @@ import { resolveCharacterMentions, applyReferenceOrderToVideo } from "./prompt-b
|
|
|
33
33
|
import { roleToPhrase, REFERENCE_ROLE_PRESETS, resolveDefaultRole } from "@nodaro/shared"
|
|
34
34
|
import { buildIdentityLockLine, withForcedIdentityLock } from "./identity-lock.js"
|
|
35
35
|
import type { ConnectedReference } from "@nodaro/shared"
|
|
36
|
+
import { REF_BINDING } from "./ref-binding.js"
|
|
37
|
+
import { resolveRefIdTokens } from "./ref-id-tokens.js"
|
|
36
38
|
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
* pair-back bullet, and the opening/closing frame directive — MUST go through
|
|
43
|
-
* these five arrows. The default form is `@image_N`; if the D7 probe shows a
|
|
44
|
-
* provider prefers the legacy `Image N` form, flipping is editing ONLY these five
|
|
45
|
-
* arrows (`@image_${n}` → `Image ${n}`), nothing downstream.
|
|
46
|
-
*
|
|
47
|
-
* This IS the live swap-point: `resolveVideoReferenceCore` routes the per-image
|
|
48
|
-
* subject phrasing, the "Use these characters" / pair-back bullet ordinals, and
|
|
49
|
-
* the frame directive through these arrows, and `resolveReferenceTokens` resolves
|
|
50
|
-
* the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
|
|
51
|
-
* are the ONLY emission sites for the binding surface string.
|
|
52
|
-
*/
|
|
53
|
-
/**
|
|
54
|
-
* The identity-reference binding sentence shared by the flat-image-list
|
|
55
|
-
* resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
|
|
56
|
-
* says the two things a multimodal model needs to hear — match exactly, and
|
|
57
|
-
* these are not frames. One spelling; both resolvers ride it.
|
|
58
|
-
*/
|
|
59
|
-
export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
|
|
60
|
-
return firstOrdinal === lastOrdinal
|
|
61
|
-
? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
|
|
62
|
-
: `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
|
|
63
|
-
}
|
|
39
|
+
// The binding surface string and the id-addressed token resolver live in their
|
|
40
|
+
// own modules (see them for the contracts); re-exported here so every existing
|
|
41
|
+
// importer of this module — and the package index's `export *` — keeps working.
|
|
42
|
+
export { REF_BINDING, identityRefsSentence } from "./ref-binding.js"
|
|
43
|
+
export { resolveRefIdTokens, type RefIdTokenContext } from "./ref-id-tokens.js"
|
|
64
44
|
|
|
65
|
-
export const REF_BINDING = {
|
|
66
|
-
image: (label: string, n: number) => `the ${label} from @image_${n}`,
|
|
67
|
-
video: (label: string, n: number) => `the ${label} from @video_${n}`,
|
|
68
|
-
audio: (label: string, n: number) => `the ${label} from @audio_${n}`,
|
|
69
|
-
/** ordinal as it appears in a "Use these characters" bullet / pair-back */
|
|
70
|
-
ordinal: (n: number) => `@image_${n}`,
|
|
71
|
-
frame: (n: number, role: "opening" | "closing") =>
|
|
72
|
-
`Use @image_${n} as the ${role} (${role === "opening" ? "first" : "last"}) frame of the video.`,
|
|
73
|
-
} as const
|
|
74
45
|
|
|
75
46
|
/**
|
|
76
47
|
* Positional reference counts the editor tokens are resolved against — how many
|
|
@@ -135,11 +106,19 @@ export function resolveReferenceTokens(
|
|
|
135
106
|
)
|
|
136
107
|
}
|
|
137
108
|
|
|
109
|
+
|
|
138
110
|
/**
|
|
139
111
|
* A user-attached "extra reference image" row. Layer-agnostic shape of the
|
|
140
112
|
* frontend `ExtraRef` / backend extras: only the fields this core reads.
|
|
141
113
|
*/
|
|
142
114
|
export interface VideoExtraRef {
|
|
115
|
+
/**
|
|
116
|
+
* The caller's own id for this reference (`connectedReferences[].id` on the
|
|
117
|
+
* route, `extraRefs[].id` on the canvas) — what a `{ref:<id>}` token in the
|
|
118
|
+
* prompt names. Slot-map only: the reorder's tile id stays `wired:<url>`.
|
|
119
|
+
* Absent → the extra cannot be addressed by id (it still numbers normally).
|
|
120
|
+
*/
|
|
121
|
+
id?: string
|
|
143
122
|
url: string
|
|
144
123
|
description?: string
|
|
145
124
|
characterSlug?: string
|
|
@@ -241,6 +220,15 @@ export interface ResolveVideoReferenceCoreArgs {
|
|
|
241
220
|
* Wired in Phase B Tasks 2-3.
|
|
242
221
|
*/
|
|
243
222
|
hybridRoles?: boolean
|
|
223
|
+
/**
|
|
224
|
+
* Display names for EVERY reference the caller knows by id — including the
|
|
225
|
+
* ones it did NOT hand to this walk (the route caps `connectedReferences` to
|
|
226
|
+
* the provider's image budget before calling in). A `{ref:<id>}` token that
|
|
227
|
+
* cannot bind degrades to `label ?? refNamesById[id] ?? ""`, so a capped-out
|
|
228
|
+
* or skipped reference keeps its name in the prose instead of vanishing.
|
|
229
|
+
* The wired character refs' `defaultName`s are known without this.
|
|
230
|
+
*/
|
|
231
|
+
refNamesById?: ReadonlyMap<string, string>
|
|
244
232
|
}
|
|
245
233
|
|
|
246
234
|
/** Result of the HYBRID mention pass — inline role phrases + surfaced opt-in
|
|
@@ -455,16 +443,39 @@ export function resolveVideoReferenceCore(
|
|
|
455
443
|
// early-return below is gated on (no chars AND no extras) so we don't skip
|
|
456
444
|
// extras-only setups.
|
|
457
445
|
const hasExtras = (args.extraRefs?.length ?? 0) > 0
|
|
446
|
+
// `{ref:<id>}` degrade names — every id this call knows: the wired refs'
|
|
447
|
+
// display names, the extras' ids (name-less, so they match by identity rather
|
|
448
|
+
// than falling to the catch-all), then the caller's own map on top (it knows
|
|
449
|
+
// the refs it capped out before calling in).
|
|
450
|
+
const nameByRefId = new Map<string, string>()
|
|
451
|
+
for (const r of args.wiredCharRefs) {
|
|
452
|
+
if (r.id && !nameByRefId.has(r.id)) nameByRefId.set(r.id, r.defaultName || r.characterSlug || "")
|
|
453
|
+
}
|
|
454
|
+
for (const ex of args.extraRefs ?? []) {
|
|
455
|
+
if (ex.id && !nameByRefId.has(ex.id)) nameByRefId.set(ex.id, "")
|
|
456
|
+
}
|
|
457
|
+
for (const [id, name] of args.refNamesById ?? []) {
|
|
458
|
+
if (id) nameByRefId.set(id, name)
|
|
459
|
+
}
|
|
460
|
+
// id → seat, recorded by the walk below as `position` advances — never
|
|
461
|
+
// recovered from URLs afterwards (the walk counts a duplicate-URL extra that
|
|
462
|
+
// `merged` dedups, so URL → index is ambiguous; id → position is not).
|
|
463
|
+
const slotByRefId = new Map<string, number>()
|
|
458
464
|
if (wiredCharRefs.length === 0 && !hasExtras) {
|
|
459
465
|
// No wired chars / extras, but the node can still carry plain base reference
|
|
460
466
|
// images (leadingRefUrls), so `{image:N}` body tokens MUST still resolve. The
|
|
461
467
|
// count is the leading-ref count (or the legacy `imageRefCount` when no
|
|
462
468
|
// leading refs were passed); the leading URLs are returned for the payload.
|
|
469
|
+
// Nothing was seated, so every `{ref:}` degrades (label → name → "").
|
|
470
|
+
const counts = tokenCounts(leadingRefUrls.length)
|
|
463
471
|
return {
|
|
464
472
|
// tokenCounts(leadingRefUrls.length) → image count == offset (no assets here):
|
|
465
473
|
// leadingRefUrls mode counts the leading refs; ordinalOffset mode counts the
|
|
466
474
|
// caller-owned leading refs the offset stands in for.
|
|
467
|
-
prompt: resolveReferenceTokens(
|
|
475
|
+
prompt: resolveReferenceTokens(
|
|
476
|
+
resolveRefIdTokens(args.prompt, { slotById: slotByRefId, nameById: nameByRefId, imageCount: counts.image }),
|
|
477
|
+
counts,
|
|
478
|
+
),
|
|
468
479
|
additionalUrls: [...leadingRefUrls],
|
|
469
480
|
}
|
|
470
481
|
}
|
|
@@ -531,10 +542,17 @@ export function resolveVideoReferenceCore(
|
|
|
531
542
|
let position = offset
|
|
532
543
|
for (let i = 0; i < resolved.additionalUrls.length; i++) {
|
|
533
544
|
position += 1
|
|
545
|
+
const url = resolved.additionalUrls[i]
|
|
534
546
|
// Look up which ref this URL came from to learn its characterSlug.
|
|
535
|
-
const ref = wiredCharRefs.find((r) => r.url ===
|
|
547
|
+
const ref = wiredCharRefs.find((r) => r.url === url)
|
|
536
548
|
const slug = ref?.characterSlug
|
|
537
549
|
if (slug && !positionsByChar.has(slug)) positionsByChar.set(slug, position)
|
|
550
|
+
// `{ref:<id>}`: every wired ref sharing this URL sits in this seat (the
|
|
551
|
+
// merge dedups by URL). First sight wins — the legacy mention pass does
|
|
552
|
+
// not dedup, so a re-mentioned URL advances `position` but keeps its seat.
|
|
553
|
+
for (const r of wiredCharRefs) {
|
|
554
|
+
if (r.url === url && r.id && !slotByRefId.has(r.id)) slotByRefId.set(r.id, position)
|
|
555
|
+
}
|
|
538
556
|
}
|
|
539
557
|
for (const r of wiredCharRefs) {
|
|
540
558
|
if (r.source !== "wired-character") continue
|
|
@@ -547,6 +565,7 @@ export function resolveVideoReferenceCore(
|
|
|
547
565
|
fallbackUrls.push(r.url)
|
|
548
566
|
position += 1
|
|
549
567
|
if (!positionsByChar.has(r.characterSlug)) positionsByChar.set(r.characterSlug, position)
|
|
568
|
+
if (r.id && !slotByRefId.has(r.id)) slotByRefId.set(r.id, position)
|
|
550
569
|
// Hybrid: emit the inline role phrase (`the person from @image_N`) + opt-in
|
|
551
570
|
// lock + wired element injection instead of a "Use these characters:"
|
|
552
571
|
// bullet. The selection above (canonical entry only, deduped, skip mentioned)
|
|
@@ -616,6 +635,7 @@ export function resolveVideoReferenceCore(
|
|
|
616
635
|
for (const ex of args.extraRefs!) {
|
|
617
636
|
if (!ex.url) continue
|
|
618
637
|
position += 1
|
|
638
|
+
if (ex.id && !slotByRefId.has(ex.id)) slotByRefId.set(ex.id, position)
|
|
619
639
|
const desc = (ex.description ?? "").trim()
|
|
620
640
|
if (ex.characterSlug) {
|
|
621
641
|
// First sight of this character via an extra. Resolution chain
|
|
@@ -797,6 +817,23 @@ export function resolveVideoReferenceCore(
|
|
|
797
817
|
if (u && !seen.has(u)) { seen.add(u); merged.push(u) }
|
|
798
818
|
}
|
|
799
819
|
|
|
820
|
+
// `{ref:<id>}` tokens resolve HERE — after the walk has seated every reference
|
|
821
|
+
// (the slot map is complete) and BEFORE the user reorder below, so the
|
|
822
|
+
// reorder's `@image_N` renumber pass carries the freshly emitted binding to
|
|
823
|
+
// the ref's final seat. That is the point of an id token: "this reference,
|
|
824
|
+
// wherever it lands". `{image:N}` is deliberately the opposite — resolved
|
|
825
|
+
// LAST, after the reorder, so the author's literal N is kept (see the note at
|
|
826
|
+
// the reorder return). Range-gated by the same image count `{image:N}` uses,
|
|
827
|
+
// so a seat the payload never ships (capped out, duplicate URL) degrades to
|
|
828
|
+
// the name instead of binding a phantom `@image_N`. A prompt with no `{ref:`
|
|
829
|
+
// is untouched — byte-identical to before this token existed.
|
|
830
|
+
finalPrompt =
|
|
831
|
+
resolveRefIdTokens(finalPrompt, {
|
|
832
|
+
slotById: slotByRefId,
|
|
833
|
+
nameById: nameByRefId,
|
|
834
|
+
imageCount: tokenCounts(merged.length).image,
|
|
835
|
+
}) ?? finalPrompt
|
|
836
|
+
|
|
800
837
|
// Apply user-defined reorder + renumber `Image N` tokens — parity with the
|
|
801
838
|
// backend `resolveVideoPromptMentions` and the shared image builder.
|
|
802
839
|
const referenceOrder = args.referenceOrder
|
|
@@ -825,7 +862,9 @@ export function resolveVideoReferenceCore(
|
|
|
825
862
|
// Resolve body tokens LAST — AFTER the reorder's `@image_N` renumber pass, so
|
|
826
863
|
// it can't miscorrect a freshly-resolved binding (the curly `{image:N}` tokens
|
|
827
864
|
// are invisible to the reorder's `(@image_|Image )` regex, so they ride through
|
|
828
|
-
// untouched and keep their author-typed N — documented v1 behavior).
|
|
865
|
+
// untouched and keep their author-typed N — documented v1 behavior). The
|
|
866
|
+
// id-addressed `{ref:<id>}` tokens are the deliberate opposite: resolved
|
|
867
|
+
// BEFORE the reorder (above), so their binding follows the reference.
|
|
829
868
|
return {
|
|
830
869
|
prompt: resolveReferenceTokens(reordered.prompt, tokenCounts(merged.length)),
|
|
831
870
|
additionalUrls: [...leadingRefUrls, ...reordered.urls],
|