@nodaro/prompts 1.8.1 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -42,9 +42,23 @@ export interface Transition {
42
42
  readonly term?: string
43
43
  }
44
44
 
45
- export type TransitionPosition = "auto" | "start" | "middle" | "end" | "full"
46
- export type TransitionDuration = "auto" | "instant" | "short" | "medium" | "long"
47
- export type TransitionIntensity = "auto" | "subtle" | "natural" | "dynamic" | "crazy"
45
+ /**
46
+ * The three timing scales, each derived from the catalog that defines it (see
47
+ * `TRANSITION_POSITIONS` and friends below).
48
+ *
49
+ * The direction matters. These used to be hand-written unions with the clause
50
+ * tables written out separately beside them, so the two could disagree: add a
51
+ * step to the union, forget the clause, and the composer indexed a missing key
52
+ * — pushing `undefined` into the parts list, which `join(", ")` renders as a
53
+ * dangling separator on a prompt that then ships to a provider with the user's
54
+ * chosen parameter silently dropped. Deriving the union FROM the catalog makes
55
+ * that unrepresentable: one array is the source of the type, the option list
56
+ * the API serves, and the clause table, so a new step reaches all three or
57
+ * none. The exact id sets are pinned by `transition-timing-catalogs.test.ts`.
58
+ */
59
+ export type TransitionPosition = (typeof TRANSITION_POSITIONS)[number]["id"]
60
+ export type TransitionDuration = (typeof TRANSITION_DURATIONS)[number]["id"]
61
+ export type TransitionIntensity = (typeof TRANSITION_INTENSITIES)[number]["id"]
48
62
 
49
63
  export interface TransitionTiming {
50
64
  position?: TransitionPosition
@@ -303,27 +317,73 @@ export const TRANSITION_IDS: ReadonlyArray<string> = TRANSITIONS.map((t) => t.id
303
317
  // Graph-aware composer — start/end input handles + timing fields + multi-pick
304
318
  // ---------------------------------------------------------------------------
305
319
 
306
- const POSITION_CLAUSES: Record<Exclude<TransitionPosition, "auto">, string> = {
307
- start: "the transition occurs at the opening of the clip",
308
- middle: "the transition occurs in the middle of the clip",
309
- end: "the transition occurs at the end of the clip",
310
- full: "the transition spans the entire clip",
320
+ /**
321
+ * The transition node's three timing parameters, as catalogs.
322
+ *
323
+ * These are graded scales, not free numbers — the same shape as
324
+ * `exposure-settings`' aperture or `temporal`'s speed — so a consumer that can
325
+ * only send ids (Studio, the SDK, MCP) can offer them without composing prompt
326
+ * text of its own. `auto` is the no-op head of each scale: an empty
327
+ * `promptHint`, so an unset parameter contributes nothing and the model is left
328
+ * to decide, exactly as before these were enumerable.
329
+ *
330
+ * `POSITION_CLAUSES` / `DURATION_CLAUSES` / `INTENSITY_CLAUSES` below are
331
+ * DERIVED from these arrays, so the clause the composer injects and the hint
332
+ * the catalog advertises are the same string by construction and cannot drift.
333
+ */
334
+ export interface TransitionTimingOption {
335
+ readonly id: string
336
+ readonly label: string
337
+ readonly description: string
338
+ readonly promptHint: string
339
+ readonly term?: string
311
340
  }
312
341
 
313
- const DURATION_CLAUSES: Record<Exclude<TransitionDuration, "auto">, string> = {
314
- instant: "occurring instantaneously",
315
- short: "lasting approximately 1 second",
316
- medium: "lasting approximately 2 seconds",
317
- long: "lasting approximately 3 seconds",
318
- }
342
+ export const TRANSITION_POSITIONS = [
343
+ { id: "auto", label: "Auto", description: "Let the model place it", promptHint: "", term: "" },
344
+ { id: "start", label: "Start", description: "At the opening of the clip", promptHint: "the transition occurs at the opening of the clip", term: "at the opening of the clip" },
345
+ { id: "middle", label: "Middle", description: "In the middle of the clip", promptHint: "the transition occurs in the middle of the clip", term: "mid-clip" },
346
+ { id: "end", label: "End", description: "At the end of the clip", promptHint: "the transition occurs at the end of the clip", term: "at the end of the clip" },
347
+ { id: "full", label: "Full", description: "Spans the entire clip", promptHint: "the transition spans the entire clip", term: "across the whole clip" },
348
+ ] as const satisfies ReadonlyArray<TransitionTimingOption>
349
+
350
+ export const TRANSITION_DURATIONS = [
351
+ { id: "auto", label: "Auto", description: "Let the model time it", promptHint: "", term: "" },
352
+ { id: "instant", label: "Instant", description: "No perceptible duration", promptHint: "occurring instantaneously", term: "instantaneous" },
353
+ { id: "short", label: "Short (~1s)", description: "Approximately 1 second", promptHint: "lasting approximately 1 second", term: "about 1 second" },
354
+ { id: "medium", label: "Medium (~2s)", description: "Approximately 2 seconds", promptHint: "lasting approximately 2 seconds", term: "about 2 seconds" },
355
+ { id: "long", label: "Long (~3s)", description: "Approximately 3 seconds", promptHint: "lasting approximately 3 seconds", term: "about 3 seconds" },
356
+ ] as const satisfies ReadonlyArray<TransitionTimingOption>
357
+
358
+ export const TRANSITION_INTENSITIES = [
359
+ { id: "auto", label: "Auto", description: "Let the model judge it", promptHint: "", term: "" },
360
+ { id: "subtle", label: "Subtle", description: "Restrained, minimal flourish", promptHint: "with subtle restrained energy and minimal flourish", term: "subtly" },
361
+ { id: "natural", label: "Natural", description: "Unhurried, unforced timing", promptHint: "with natural unhurried timing", term: "at a natural pace" },
362
+ { id: "dynamic", label: "Dynamic", description: "Assertive, energetic", promptHint: "with dynamic energy and assertive flourish", term: "energetically" },
363
+ { id: "crazy", label: "Crazy", description: "Extreme, wild, distorted", promptHint: "with extreme exaggerated energy, wild flourishes, and dramatic distortion", term: "wildly exaggerated" },
364
+ ] as const satisfies ReadonlyArray<TransitionTimingOption>
319
365
 
320
- const INTENSITY_CLAUSES: Record<Exclude<TransitionIntensity, "auto">, string> = {
321
- subtle: "with subtle restrained energy and minimal flourish",
322
- natural: "with natural unhurried timing",
323
- dynamic: "with dynamic energy and assertive flourish",
324
- crazy: "with extreme exaggerated energy, wild flourishes, and dramatic distortion",
366
+ /**
367
+ * Index a timing catalog into the `Record<value, clause>` the composer reads.
368
+ *
369
+ * The key type is derived from the SAME array, so the record is total over the
370
+ * catalog by construction. That matters: the composer indexes these records
371
+ * without a fallback, and a missing key would push `undefined` into the parts
372
+ * list, which `join(", ")` renders as a dangling separator — a malformed prompt
373
+ * shipped to a provider with the user's chosen parameter silently dropped.
374
+ */
375
+ function clausesOf<T extends TransitionTimingOption>(
376
+ options: ReadonlyArray<T>,
377
+ ): Record<Exclude<T["id"], "auto">, string> {
378
+ return Object.fromEntries(
379
+ options.filter((o) => o.id !== "auto").map((o) => [o.id, o.promptHint]),
380
+ ) as Record<Exclude<T["id"], "auto">, string>
325
381
  }
326
382
 
383
+ const POSITION_CLAUSES = clausesOf(TRANSITION_POSITIONS)
384
+ const DURATION_CLAUSES = clausesOf(TRANSITION_DURATIONS)
385
+ const INTENSITY_CLAUSES = clausesOf(TRANSITION_INTENSITIES)
386
+
327
387
  /**
328
388
  * Compose a structural prompt-hint sentence from a transition id (or array
329
389
  * of 1-2 ids for multi-pick) plus optional start-state/end-state hints
@@ -33,44 +33,15 @@ import { resolveCharacterMentions, applyReferenceOrderToVideo } from "./prompt-b
33
33
  import { roleToPhrase, REFERENCE_ROLE_PRESETS, resolveDefaultRole } from "@nodaro/shared"
34
34
  import { buildIdentityLockLine, withForcedIdentityLock } from "./identity-lock.js"
35
35
  import type { ConnectedReference } from "@nodaro/shared"
36
+ import { REF_BINDING } from "./ref-binding.js"
37
+ import { resolveRefIdTokens } from "./ref-id-tokens.js"
36
38
 
37
- /**
38
- * The SINGLE swap-point for the reference-binding surface-string (design D1/D7).
39
- *
40
- * Every place that renders an `@image_N`-style binding into a video prompt — the
41
- * per-image subject phrasing, the bare ordinal in a "Use these characters" /
42
- * pair-back bullet, and the opening/closing frame directive — MUST go through
43
- * these five arrows. The default form is `@image_N`; if the D7 probe shows a
44
- * provider prefers the legacy `Image N` form, flipping is editing ONLY these five
45
- * arrows (`@image_${n}` → `Image ${n}`), nothing downstream.
46
- *
47
- * This IS the live swap-point: `resolveVideoReferenceCore` routes the per-image
48
- * subject phrasing, the "Use these characters" / pair-back bullet ordinals, and
49
- * the frame directive through these arrows, and `resolveReferenceTokens` resolves
50
- * the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
51
- * are the ONLY emission sites for the binding surface string.
52
- */
53
- /**
54
- * The identity-reference binding sentence shared by the flat-image-list
55
- * resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
56
- * says the two things a multimodal model needs to hear — match exactly, and
57
- * these are not frames. One spelling; both resolvers ride it.
58
- */
59
- export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
60
- return firstOrdinal === lastOrdinal
61
- ? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
62
- : `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
63
- }
39
+ // The binding surface string and the id-addressed token resolver live in their
40
+ // own modules (see them for the contracts); re-exported here so every existing
41
+ // importer of this module — and the package index's `export *` — keeps working.
42
+ export { REF_BINDING, identityRefsSentence } from "./ref-binding.js"
43
+ export { resolveRefIdTokens, type RefIdTokenContext } from "./ref-id-tokens.js"
64
44
 
65
- export const REF_BINDING = {
66
- image: (label: string, n: number) => `the ${label} from @image_${n}`,
67
- video: (label: string, n: number) => `the ${label} from @video_${n}`,
68
- audio: (label: string, n: number) => `the ${label} from @audio_${n}`,
69
- /** ordinal as it appears in a "Use these characters" bullet / pair-back */
70
- ordinal: (n: number) => `@image_${n}`,
71
- frame: (n: number, role: "opening" | "closing") =>
72
- `Use @image_${n} as the ${role} (${role === "opening" ? "first" : "last"}) frame of the video.`,
73
- } as const
74
45
 
75
46
  /**
76
47
  * Positional reference counts the editor tokens are resolved against — how many
@@ -135,11 +106,19 @@ export function resolveReferenceTokens(
135
106
  )
136
107
  }
137
108
 
109
+
138
110
  /**
139
111
  * A user-attached "extra reference image" row. Layer-agnostic shape of the
140
112
  * frontend `ExtraRef` / backend extras: only the fields this core reads.
141
113
  */
142
114
  export interface VideoExtraRef {
115
+ /**
116
+ * The caller's own id for this reference (`connectedReferences[].id` on the
117
+ * route, `extraRefs[].id` on the canvas) — what a `{ref:<id>}` token in the
118
+ * prompt names. Slot-map only: the reorder's tile id stays `wired:<url>`.
119
+ * Absent → the extra cannot be addressed by id (it still numbers normally).
120
+ */
121
+ id?: string
143
122
  url: string
144
123
  description?: string
145
124
  characterSlug?: string
@@ -241,6 +220,15 @@ export interface ResolveVideoReferenceCoreArgs {
241
220
  * Wired in Phase B Tasks 2-3.
242
221
  */
243
222
  hybridRoles?: boolean
223
+ /**
224
+ * Display names for EVERY reference the caller knows by id — including the
225
+ * ones it did NOT hand to this walk (the route caps `connectedReferences` to
226
+ * the provider's image budget before calling in). A `{ref:<id>}` token that
227
+ * cannot bind degrades to `label ?? refNamesById[id] ?? ""`, so a capped-out
228
+ * or skipped reference keeps its name in the prose instead of vanishing.
229
+ * The wired character refs' `defaultName`s are known without this.
230
+ */
231
+ refNamesById?: ReadonlyMap<string, string>
244
232
  }
245
233
 
246
234
  /** Result of the HYBRID mention pass — inline role phrases + surfaced opt-in
@@ -455,16 +443,39 @@ export function resolveVideoReferenceCore(
455
443
  // early-return below is gated on (no chars AND no extras) so we don't skip
456
444
  // extras-only setups.
457
445
  const hasExtras = (args.extraRefs?.length ?? 0) > 0
446
+ // `{ref:<id>}` degrade names — every id this call knows: the wired refs'
447
+ // display names, the extras' ids (name-less, so they match by identity rather
448
+ // than falling to the catch-all), then the caller's own map on top (it knows
449
+ // the refs it capped out before calling in).
450
+ const nameByRefId = new Map<string, string>()
451
+ for (const r of args.wiredCharRefs) {
452
+ if (r.id && !nameByRefId.has(r.id)) nameByRefId.set(r.id, r.defaultName || r.characterSlug || "")
453
+ }
454
+ for (const ex of args.extraRefs ?? []) {
455
+ if (ex.id && !nameByRefId.has(ex.id)) nameByRefId.set(ex.id, "")
456
+ }
457
+ for (const [id, name] of args.refNamesById ?? []) {
458
+ if (id) nameByRefId.set(id, name)
459
+ }
460
+ // id → seat, recorded by the walk below as `position` advances — never
461
+ // recovered from URLs afterwards (the walk counts a duplicate-URL extra that
462
+ // `merged` dedups, so URL → index is ambiguous; id → position is not).
463
+ const slotByRefId = new Map<string, number>()
458
464
  if (wiredCharRefs.length === 0 && !hasExtras) {
459
465
  // No wired chars / extras, but the node can still carry plain base reference
460
466
  // images (leadingRefUrls), so `{image:N}` body tokens MUST still resolve. The
461
467
  // count is the leading-ref count (or the legacy `imageRefCount` when no
462
468
  // leading refs were passed); the leading URLs are returned for the payload.
469
+ // Nothing was seated, so every `{ref:}` degrades (label → name → "").
470
+ const counts = tokenCounts(leadingRefUrls.length)
463
471
  return {
464
472
  // tokenCounts(leadingRefUrls.length) → image count == offset (no assets here):
465
473
  // leadingRefUrls mode counts the leading refs; ordinalOffset mode counts the
466
474
  // caller-owned leading refs the offset stands in for.
467
- prompt: resolveReferenceTokens(args.prompt, tokenCounts(leadingRefUrls.length)),
475
+ prompt: resolveReferenceTokens(
476
+ resolveRefIdTokens(args.prompt, { slotById: slotByRefId, nameById: nameByRefId, imageCount: counts.image }),
477
+ counts,
478
+ ),
468
479
  additionalUrls: [...leadingRefUrls],
469
480
  }
470
481
  }
@@ -531,10 +542,17 @@ export function resolveVideoReferenceCore(
531
542
  let position = offset
532
543
  for (let i = 0; i < resolved.additionalUrls.length; i++) {
533
544
  position += 1
545
+ const url = resolved.additionalUrls[i]
534
546
  // Look up which ref this URL came from to learn its characterSlug.
535
- const ref = wiredCharRefs.find((r) => r.url === resolved.additionalUrls[i])
547
+ const ref = wiredCharRefs.find((r) => r.url === url)
536
548
  const slug = ref?.characterSlug
537
549
  if (slug && !positionsByChar.has(slug)) positionsByChar.set(slug, position)
550
+ // `{ref:<id>}`: every wired ref sharing this URL sits in this seat (the
551
+ // merge dedups by URL). First sight wins — the legacy mention pass does
552
+ // not dedup, so a re-mentioned URL advances `position` but keeps its seat.
553
+ for (const r of wiredCharRefs) {
554
+ if (r.url === url && r.id && !slotByRefId.has(r.id)) slotByRefId.set(r.id, position)
555
+ }
538
556
  }
539
557
  for (const r of wiredCharRefs) {
540
558
  if (r.source !== "wired-character") continue
@@ -547,6 +565,7 @@ export function resolveVideoReferenceCore(
547
565
  fallbackUrls.push(r.url)
548
566
  position += 1
549
567
  if (!positionsByChar.has(r.characterSlug)) positionsByChar.set(r.characterSlug, position)
568
+ if (r.id && !slotByRefId.has(r.id)) slotByRefId.set(r.id, position)
550
569
  // Hybrid: emit the inline role phrase (`the person from @image_N`) + opt-in
551
570
  // lock + wired element injection instead of a "Use these characters:"
552
571
  // bullet. The selection above (canonical entry only, deduped, skip mentioned)
@@ -616,6 +635,7 @@ export function resolveVideoReferenceCore(
616
635
  for (const ex of args.extraRefs!) {
617
636
  if (!ex.url) continue
618
637
  position += 1
638
+ if (ex.id && !slotByRefId.has(ex.id)) slotByRefId.set(ex.id, position)
619
639
  const desc = (ex.description ?? "").trim()
620
640
  if (ex.characterSlug) {
621
641
  // First sight of this character via an extra. Resolution chain
@@ -797,6 +817,23 @@ export function resolveVideoReferenceCore(
797
817
  if (u && !seen.has(u)) { seen.add(u); merged.push(u) }
798
818
  }
799
819
 
820
+ // `{ref:<id>}` tokens resolve HERE — after the walk has seated every reference
821
+ // (the slot map is complete) and BEFORE the user reorder below, so the
822
+ // reorder's `@image_N` renumber pass carries the freshly emitted binding to
823
+ // the ref's final seat. That is the point of an id token: "this reference,
824
+ // wherever it lands". `{image:N}` is deliberately the opposite — resolved
825
+ // LAST, after the reorder, so the author's literal N is kept (see the note at
826
+ // the reorder return). Range-gated by the same image count `{image:N}` uses,
827
+ // so a seat the payload never ships (capped out, duplicate URL) degrades to
828
+ // the name instead of binding a phantom `@image_N`. A prompt with no `{ref:`
829
+ // is untouched — byte-identical to before this token existed.
830
+ finalPrompt =
831
+ resolveRefIdTokens(finalPrompt, {
832
+ slotById: slotByRefId,
833
+ nameById: nameByRefId,
834
+ imageCount: tokenCounts(merged.length).image,
835
+ }) ?? finalPrompt
836
+
800
837
  // Apply user-defined reorder + renumber `Image N` tokens — parity with the
801
838
  // backend `resolveVideoPromptMentions` and the shared image builder.
802
839
  const referenceOrder = args.referenceOrder
@@ -825,7 +862,9 @@ export function resolveVideoReferenceCore(
825
862
  // Resolve body tokens LAST — AFTER the reorder's `@image_N` renumber pass, so
826
863
  // it can't miscorrect a freshly-resolved binding (the curly `{image:N}` tokens
827
864
  // are invisible to the reorder's `(@image_|Image )` regex, so they ride through
828
- // untouched and keep their author-typed N — documented v1 behavior).
865
+ // untouched and keep their author-typed N — documented v1 behavior). The
866
+ // id-addressed `{ref:<id>}` tokens are the deliberate opposite: resolved
867
+ // BEFORE the reorder (above), so their binding follows the reference.
829
868
  return {
830
869
  prompt: resolveReferenceTokens(reordered.prompt, tokenCounts(merged.length)),
831
870
  additionalUrls: [...leadingRefUrls, ...reordered.urls],