@nodaro/prompts 1.9.0 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,6 +12,7 @@ import { usageModeDirective, DEFAULT_USAGE_MODE, type UsageMode } from "@nodaro/
12
12
  import { roleToPhrase, defaultRoleForSource, REFERENCE_ROLE_PRESETS, normalizeRoleSlug, resolveDefaultRole } from "@nodaro/shared"
13
13
  import { buildIdentityLockLine, withForcedIdentityLock } from "./identity-lock.js"
14
14
  import { findLocationMentionTokens, DEFAULT_LOCATION_USAGE_MODE, type LocationMentionTokenInfo, type LocationUsageMode } from "@nodaro/shared"
15
+ import { findImageMentionTokens, imageMentionSlugForRef, knownImageSlugsFromRefs, type ImageMentionTokenInfo } from "@nodaro/shared"
15
16
  import type { CharacterDef, ConnectedReference, IdentityFidelity, IdentityMeta, ReferenceSource, SceneData } from "@nodaro/shared"
16
17
  import { locationReferencePhotoKindLabel, type LocationReferencePhotoKind } from "@nodaro/shared"
17
18
 
@@ -694,6 +695,131 @@ function resolveLocationMentionsHybrid(
694
695
  return { prompt: resolvedPrompt, additionalUrls, mentionedLocationSlugs, lockLines, elementDirectives }
695
696
  }
696
697
 
698
+ interface ResolveImageMentionsHybridResult {
699
+ /** Body with each `@<image-name>` mention replaced INLINE by its role phrase
700
+ * ("the {role} from reference image {LETTER}", or the bare binding when the
701
+ * role resolves to the media default `""`). No directive block. */
702
+ prompt: string
703
+ /** Matched image URLs in mention order (deduped by the caller). */
704
+ additionalUrls: string[]
705
+ /** NO `mentionedImageSlugs` — the location result's analog exists to FILTER
706
+ * mentioned refs out of `connectedReferences`, and this pass deliberately
707
+ * does no such filtering (see the caller's NOTE). Carrying the set anyway
708
+ * would advertise a filter that does not exist. */
709
+ /** Per-reference identity-lock lines (deduped per URL). Caller prepends them
710
+ * as ONE block, merged with the character/location lock lines. */
711
+ lockLines: string[]
712
+ /** Non-empty `elementInjection` fragments (deduped per URL). Caller appends
713
+ * them as trailing scene directives. */
714
+ elementDirectives: string[]
715
+ }
716
+
717
+ /**
718
+ * HYBRID named-image mention convergence (P3). The media analog of
719
+ * `resolveLocationMentionsHybrid`, for `wired-image` / `manual` references
720
+ * addressed by the slug of their `defaultName` (an upload node's label on the
721
+ * canvas, or the name a thin client puts on the reference).
722
+ *
723
+ * ASYMMETRY vs. characters and locations: a media ref AUTO-ATTACHES its URL
724
+ * through the New path whether or not it is mentioned, and carries NO canonical
725
+ * prose. So a mention here is BINDING + RE-SEATING, never attach-gating — which
726
+ * is why the caller does NOT filter mentioned refs out of `connectedReferences`
727
+ * the way the location pass does.
728
+ *
729
+ * NO LEGACY COUNTERPART. The Phase-0 arm that reaches this is hybrid-gated, so
730
+ * an `@name:N` token under the legacy reference format stays literal text and
731
+ * the ref attaches exactly as it does today (the `resolveLocationMentions`
732
+ * role-token precedent, where `if (t.role) continue` leaves the token alone).
733
+ *
734
+ * DUPLICATE SLUGS: FIRST WINS (`if (!bySlug.has(...))`), matching
735
+ * `buildTileIdForUrl`. Every unrenamed upload node shares its default label, so
736
+ * ties are the common case, not an edge.
737
+ *
738
+ * ROLE precedence: the per-mention 3rd segment (VERBATIM — media role presets
739
+ * are all single-word, so `normalizeRoleSlug` is location-only and must NOT be
740
+ * called here) → the node default via `resolveDefaultRole` → `""`, which
741
+ * `roleToPhrase` renders as the BARE binding ("reference image C"), i.e. today's
742
+ * ref-only default with a name attached to it.
743
+ *
744
+ * CAPPED REFS: `imageReferenceLimit(provider)` truncates `connectedReferences`
745
+ * BEFORE Phase 0, so a mention whose ref was capped out silently falls through
746
+ * as literal text — matching how a capped character mention behaves today.
747
+ */
748
+ function resolveImageMentionsHybrid(
749
+ prompt: string,
750
+ tokens: readonly ImageMentionTokenInfo[],
751
+ refs: readonly ConnectedReference[],
752
+ existingUrls: readonly string[],
753
+ ): ResolveImageMentionsHybridResult {
754
+ const bySlug = new Map<string, ConnectedReference>()
755
+ for (const r of refs) {
756
+ // `imageMentionSlugForRef` is the SAME predicate `knownImageSlugsFromRefs`
757
+ // applies to build the finder's known-slug set — one gate, so this map and
758
+ // that set can never admit different refs. (It also drops extras, which
759
+ // render through `renderExtraRefsHybrid` with their own body lines, and
760
+ // grammar-invalid slugs, where emptiness is NOT the gate.)
761
+ const slug = imageMentionSlugForRef(r)
762
+ if (!slug) continue
763
+ if (!bySlug.has(slug)) bySlug.set(slug, r)
764
+ }
765
+
766
+ const additionalUrls: string[] = []
767
+ const refByUrl = new Map<string, ConnectedReference>()
768
+ // Per-mention `~lock` / `~nolock`: the tri-state lock OVERRIDE per attached
769
+ // URL, fed to `withForcedIdentityLock` below. Only sentinel-bearing mentions
770
+ // write here (last sentinel wins), mirroring the location resolver.
771
+ const lockOverrideByUrl = new Map<string, boolean>()
772
+ const matched: Array<{ token: string; offset: number; url: string; role: string }> = []
773
+
774
+ for (const t of tokens) {
775
+ const match = bySlug.get(t.imageSlug)
776
+ if (!match || !match.url) continue
777
+ additionalUrls.push(match.url)
778
+ refByUrl.set(match.url, match)
779
+ if (t.lock !== undefined) lockOverrideByUrl.set(match.url, t.lock)
780
+ const role = (t.role ?? "").trim()
781
+ || resolveDefaultRole(match.defaultRole, match.defaultUsageMode, match.source)
782
+ matched.push({ token: t.token, offset: t.offset, url: match.url, role })
783
+ }
784
+
785
+ // Slot letters from the deduped [existing, mention] URL list — the prefix of
786
+ // the caller's `finalIndexByUrl`, so the letters agree.
787
+ const slotByUrl = new Map<string, number>()
788
+ for (const u of [...existingUrls, ...additionalUrls]) {
789
+ if (!slotByUrl.has(u)) slotByUrl.set(u, slotByUrl.size + 1)
790
+ }
791
+ const bindingFor = (url: string): string => {
792
+ const slot = slotByUrl.get(url)
793
+ return slot ? `reference image ${slotToLetter(slot)}` : "the reference image"
794
+ }
795
+
796
+ // Replace mention tokens right-to-left so earlier offsets stay valid.
797
+ let resolvedPrompt = prompt
798
+ for (const m of [...matched].sort((a, b) => b.offset - a.offset)) {
799
+ const phrase = roleToPhrase(m.role, bindingFor(m.url))
800
+ resolvedPrompt =
801
+ resolvedPrompt.slice(0, m.offset) + phrase + resolvedPrompt.slice(m.offset + m.token.length)
802
+ }
803
+
804
+ // One lock + one element directive per UNIQUE attached URL.
805
+ const lockLines: string[] = []
806
+ const elementDirectives: string[] = []
807
+ const seenUrls = new Set<string>()
808
+ for (const m of matched) {
809
+ if (seenUrls.has(m.url)) continue
810
+ seenUrls.add(m.url)
811
+ const ref = refByUrl.get(m.url)
812
+ if (!ref) continue
813
+ const binding = bindingFor(m.url)
814
+ const lock = buildIdentityLockLine(withForcedIdentityLock(ref, lockOverrideByUrl.get(m.url)), binding)
815
+ if (lock) lockLines.push(lock)
816
+ const inject = ref.elementInjection?.trim()
817
+ if (inject) elementDirectives.push(inject)
818
+ }
819
+
820
+ return { prompt: resolvedPrompt, additionalUrls, lockLines, elementDirectives }
821
+ }
822
+
697
823
  /**
698
824
  * Build the canonical-fallback directive lines + URLs for wired characters
699
825
  * that were NOT @-mentioned in the prompt. Matches the pre-mention behavior:
@@ -1495,6 +1621,12 @@ export interface BuildImagePromptConfig {
1495
1621
  * (`flux-lora-character`) — the trigger word + LoRA carries identity, so
1496
1622
  * the directive bullets are redundant and the wired-character refs
1497
1623
  * shouldn't be injected as `Image N`.
1624
+ *
1625
+ * The strip regex is grammar-agnostic — it eats LOCATION and named-IMAGE
1626
+ * mentions (`@town:3:background`) as well as character ones. That is
1627
+ * INTENDED, not an oversight: this path drops `connectedReferences` outright,
1628
+ * so there is no reference left for any mention to bind to and a surviving
1629
+ * token would reach the model as literal noise. Do not narrow it.
1498
1630
  */
1499
1631
  skipCharacterMentions?: boolean
1500
1632
  }
@@ -1714,7 +1846,22 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
1714
1846
  .filter((s): s is string => typeof s === "string" && s.length > 0)
1715
1847
  )
1716
1848
  )
1717
- if (knownCharacterSlugs.length > 0 || hasExtraRefs || knownLocationSlugs.length > 0) {
1849
+ // Named-image mentions (P3). Slugs derive from each media ref's
1850
+ // `defaultName` — there is no `imageSlug` wire field; the derivation is
1851
+ // shared with the backend orchestrator's structured-branch gate via
1852
+ // `knownImageSlugsFromRefs`, so the two can never disagree about whether a
1853
+ // prompt carries a resolvable image mention.
1854
+ const knownImageSlugs = knownImageSlugsFromRefs(connectedReferences)
1855
+ // EXISTENCE-ONLY precheck, gating the Phase-0 arm below. Unlike characters
1856
+ // (whose canonical fallback must run even with ZERO mentions), images have
1857
+ // NO canonical fallback, so an image-only graph with no mention has no
1858
+ // reason to enter Phase 0 — gating on TOKEN presence keeps today's control
1859
+ // flow and byte output for every unmentioned graph BY CONSTRUCTION.
1860
+ // HYBRID-only: under the legacy format an `@name:N` token stays literal text.
1861
+ const hasImageMentionTokens = isHybrid
1862
+ && knownImageSlugs.length > 0
1863
+ && findImageMentionTokens(config.prompt, knownImageSlugs).length > 0
1864
+ if (knownCharacterSlugs.length > 0 || hasExtraRefs || knownLocationSlugs.length > 0 || hasImageMentionTokens) {
1718
1865
  const mentionTokens = knownCharacterSlugs.length > 0
1719
1866
  ? findCharacterMentionTokens(config.prompt, knownCharacterSlugs)
1720
1867
  : []
@@ -1806,6 +1953,44 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
1806
1953
  if (r.locationVariantBucket) return false
1807
1954
  return !mentionedLocationSlugs.has(r.locationSlug)
1808
1955
  })
1956
+ // Pass 3 (P3): named-image mentions, resolved on the POST-character,
1957
+ // POST-location prompt. The finder MUST re-run here — the earlier passes
1958
+ // spliced their own tokens out, so any offset computed before them is
1959
+ // stale. PRECEDENCE character → location → image falls out of this pass
1960
+ // order: a name shared by a character and an image resolves as the
1961
+ // character, and the image token never fires.
1962
+ //
1963
+ // `existingUrls` is already location-inclusive because `additionalUrls`
1964
+ // absorbed the location URLs just above, so the mention slot letters stay
1965
+ // a prefix of the final `finalIndexByUrl`.
1966
+ let hybridImageLockLines: string[] = []
1967
+ let hybridImageElementDirectives: string[] = []
1968
+ if (hasImageMentionTokens) {
1969
+ const imageTokens = findImageMentionTokens(resolved.prompt, knownImageSlugs)
1970
+ if (imageTokens.length > 0) {
1971
+ const hi = resolveImageMentionsHybrid(
1972
+ resolved.prompt,
1973
+ imageTokens,
1974
+ connectedReferences,
1975
+ [...(referenceImageUrls || []), ...resolved.additionalUrls],
1976
+ )
1977
+ resolved.prompt = hi.prompt
1978
+ resolved.additionalUrls = [...resolved.additionalUrls, ...hi.additionalUrls]
1979
+ hybridImageLockLines = hi.lockLines
1980
+ hybridImageElementDirectives = hi.elementDirectives
1981
+ // Inline role phrases now live in the body → skip line-initial
1982
+ // capitalization, which would otherwise corrupt a mid-sentence
1983
+ // "the background from reference image C".
1984
+ if (hi.additionalUrls.length > 0) hybridBodyConverged = true
1985
+ }
1986
+ }
1987
+ // NOTE — deliberately NO `connectedReferences` filter for mentioned image
1988
+ // refs (the location pass filters, just above). The New-path URL merge
1989
+ // dedups by URL so the ref keeps its earlier Phase-0 slot; wired-image /
1990
+ // manual refs have no canonical-fallback prose to double-emit (that loop
1991
+ // is gated to wired-location/object/creature); and leaving the ref in
1992
+ // place keeps `nonCharacterRefs[N-1]` positional indexing stable for
1993
+ // `{image:N}` tokens.
1809
1994
  // Default-fallback canonical URLs + directives for any wired character
1810
1995
  // that has zero mentions in the prompt. Mirrors the legacy behavior the
1811
1996
  // mention feature replaced — wire a character with no typing required.
@@ -1944,12 +2129,14 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
1944
2129
  const allLockLines = [...new Set([
1945
2130
  ...hybridLockLines,
1946
2131
  ...hybridLocationLockLines,
2132
+ ...hybridImageLockLines,
1947
2133
  ...canonical.lockLines,
1948
2134
  ...extrasRendered.lockLines,
1949
2135
  ])]
1950
2136
  const trailingLines = [
1951
2137
  ...hybridElementDirectives,
1952
2138
  ...hybridLocationElementDirectives,
2139
+ ...hybridImageElementDirectives,
1953
2140
  ...canonical.phrases,
1954
2141
  ...canonical.elementDirectives,
1955
2142
  ...extrasRendered.bodyLines,
@@ -0,0 +1,30 @@
1
+ /**
2
+ * The measured hint join, shared by the image (`composePromptText`) and video
3
+ * (`composeVideoPromptText`) composers so the two can never drift.
4
+ *
5
+ * EXACT NO-OP: zero hints → the user's prompt back VERBATIM AND UNTRIMMED (the
6
+ * platform-caller byte-parity contract — the old platform path passed the
7
+ * prompt straight to `buildImagePrompt`, which never trims, so trimming here
8
+ * would change the assembled prompt and the recorded `jobs.input_data`
9
+ * byte-for-byte). With hints → trim the body so the join reads cleanly
10
+ * ("prompt. hint", not "prompt . hint"), and drop a blank body so the result
11
+ * never starts with ". ".
12
+ *
13
+ * Never mutates its inputs.
14
+ */
15
+
16
+ /** The measured separator between the user's prompt and each folded hint. */
17
+ export const PROMPT_HINT_SEPARATOR = ". "
18
+
19
+ /**
20
+ * Join a user prompt with its folded hint clauses.
21
+ *
22
+ * The trailing `.filter((p) => p.length > 0)` on the join array is
23
+ * parity-critical — do NOT remove it as "redundant": `hints` arrives
24
+ * pre-filtered but `userPrompt` does not, so a blank user prompt would
25
+ * otherwise make the result start with ". ".
26
+ */
27
+ export function joinPromptHints(userPrompt: string, hints: readonly string[]): string {
28
+ if (hints.length === 0) return userPrompt
29
+ return [userPrompt.trim(), ...hints].filter((p) => p.length > 0).join(PROMPT_HINT_SEPARATOR)
30
+ }
@@ -73,7 +73,7 @@ precise subject → action details → scene/environment → lighting & color to
73
73
  - Transitions and camera terms on 2.5: state a transition's trigger point AND method in one sentence — "At the 5-second mark, the camera quickly transitions leftward using a left wipe combined with a natural dissolve." Basic shot and camera terms are written directly (push in / pull out / pan / track / orbit / dolly zoom / whip pan / hard cut / dissolve / one-shot / speed ramp); only niche terms need [term + descriptive explanation] — which is exactly what the pickers' compact hint mode emits versus their long hints.
74
74
 
75
75
  **References (when reference media is attached)**
76
- - Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync.)
76
+ - Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync. An API caller that passes \`connectedReferences\` can instead name a reference by its own id — \`{ref:<id>}\` / \`{ref:<id>:label}\` — and the platform substitutes the \`@image_N\` seat after it has numbered the references, so the client never computes N; a token whose reference was not attached drops to its label or name.)
77
77
  - Define each subject once, then reuse the label consistently: 'Define the woman in the red dress in Image 1 as the courier' … 'the courier opens the door'. In multi-character scenes bind every character to its image ("the man from Image 1 hands the box to the woman from Image 2") and append: "do not generate duplicate copies of the same character".
78
78
  - Character identity: ONE close-up headshot + ONE full-body image is ideal. On the 2.0 SKUs do NOT attach multi-view/three-view character sheets — the model reads the views as separate people, causing identity drift and twin duplicates; 2.5 accepts multi-view images (see "Generation differences").
79
79
  - 4-5 assets total works best (1-2 character images + 1 scene image + 1 camera-movement video + 1 audio clip). Maxing out the 9-image/3-video/3-audio limits degrades feature priority and adherence.
@@ -177,7 +177,7 @@ precise subject → action details → scene/environment → lighting & color to
177
177
  - Nothing visual connected → text-to-video. A concrete aspect ratio is required (21:9 / 16:9 / 4:3 / 1:1 / 3:4 / 9:16 — no adaptive); Nodaro renders 16:9 unless one is picked.
178
178
 
179
179
  **References (when reference media is attached)**
180
- - Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync.)
180
+ - Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync. An API caller that passes \`connectedReferences\` can instead write \`{ref:<id>}\` / \`{ref:<id>:label}\` with the reference's own id — the platform substitutes the \`@image_N\` seat after numbering.)
181
181
  - Caps: 9 reference images; 3 reference videos, each 2-15s and ≤15s combined; 3 reference audio clips, ≤15s combined. Reference audio cannot be used alone — it must accompany an image or video reference.
182
182
  - Define each subject once, then reuse the label consistently ("the woman from @Image 1 … the woman opens the door"). A focused set of 4-5 assets beats maxing every cap.
183
183
  - Billing note: generated seconds AND reference-video input seconds bill at the same per-second rate; the first 5 input images are free and each extra image adds a small surcharge; audio input is free.
@@ -0,0 +1,174 @@
1
+ /**
2
+ * Narrow readers turning UNTRUSTED persisted node data (`workflows.nodes`
3
+ * JSONB — import, MCP write, node preset, a Studio-emitted graph) into the
4
+ * typed `direction` / `structured` levers `assembleImageInput` accepts.
5
+ * `buildPayload` has no zod and workflow writes are
6
+ * `z.record(z.string(), z.unknown())`, so this blob may have been written years
7
+ * ago by any client.
8
+ *
9
+ * Used by ALL THREE image-assembly sites so what the canvas accepts cannot
10
+ * drift between them: the frontend single-node executor (`execute-node.ts`),
11
+ * the orchestrator (`payload-builder.ts`), and the config-panel final-prompt
12
+ * preview (`build-image-assemble-input.ts`).
13
+ *
14
+ * `readDirectionFields` is DERIVED FROM `DIRECTION_FIELDS`, never a hand list:
15
+ * a dimension added to the registry is honored here by construction, so the
16
+ * reader can never silently drop a key the wire schema and the renderer both
17
+ * know about. Anything unrecognised is DROPPED, never thrown on — a malformed
18
+ * blob must not fail a canvas run, and an unknown catalog ID already degrades
19
+ * to `""` inside each `get*PromptHint`, so this validates SHAPE only.
20
+ *
21
+ * BOUNDS MATCH THE WIRE SCHEMA where both exist, by SHARED CONSTANT:
22
+ * `DIRECTION_ID_MAX_CHARS` and `DIRECTION_ARRAY_CEILING` are the same two
23
+ * literals the `generate-image` route's `directionSchema` enforces, imported
24
+ * from the registry rather than re-typed here — a body the route accepts and a
25
+ * node the canvas re-runs must not disagree about which strings are ids.
26
+ *
27
+ * DELIBERATELY STRICTER THAN THE WIRE SCHEMA on `structured` values: the
28
+ * route's `structuredPromptFieldsSchema` declares every field as a bare
29
+ * `z.string().optional()` (unbounded), while `MAX_STRUCTURED_VALUE_CHARS` below
30
+ * bounds them at 200. Do not "fix" that asymmetry by loosening HERE — this
31
+ * reads a blob any client may have written years ago, and the values land
32
+ * verbatim in `jobs.input_data.prompt`. Tightening the ROUTE instead would be a
33
+ * new 400 on currently-accepted input, so a >200-char structured value stored
34
+ * on a node is dropped on a canvas run while `POST /v1/generate-image` still
35
+ * renders it. That is the one known divergence; it is bounded to a field the
36
+ * canvas UI never writes that long.
37
+ *
38
+ * RETURNS `undefined`, NEVER `{}`: an empty object would still be a *defined*
39
+ * `direction`, and the call sites' `...(x !== undefined ? { x } : {})` spread
40
+ * would then hand `assembleImageInput` a defined-but-empty lever. Returning
41
+ * `undefined` keeps that spread honest and the exact no-op branch taken.
42
+ */
43
+ import {
44
+ DIRECTION_ARRAY_CEILING,
45
+ DIRECTION_FIELDS,
46
+ DIRECTION_ID_MAX_CHARS,
47
+ type DirectionFields,
48
+ } from "./direction-registry.js"
49
+ import type { StructuredPromptFields } from "./prompt-builder-structured-fields.js"
50
+
51
+ /**
52
+ * Read a node's stored cinematic-direction ids. Accepts a single id OR an array
53
+ * on every key (multi-pick dimensions carry arrays; a single-pick key may
54
+ * legitimately carry one).
55
+ *
56
+ * The SEMANTIC per-dimension cap stays the renderer's slice (`maxPicks`) — this
57
+ * reader does not know a row's pick budget and must not guess it. But it DOES
58
+ * bound cardinality at `DIRECTION_ARRAY_CEILING`, the same ceiling the wire
59
+ * schema enforces: the value is untrusted persisted JSONB (a node blob is
60
+ * validated only as `z.record(z.string(), z.unknown())` on write), and handing
61
+ * an unbounded array to `renderDirectionHints` would put an `includes`-dedupe
62
+ * scan in front of that slice. Keeping the FIRST `DIRECTION_ARRAY_CEILING`
63
+ * survivors is what the wire door already does to the same input.
64
+ *
65
+ * Junk is filtered BEFORE the cap, so valid ids sitting behind malformed
66
+ * entries survive rather than being crowded out by them.
67
+ */
68
+ export function readDirectionFields(value: unknown): DirectionFields | undefined {
69
+ if (typeof value !== "object" || value === null || Array.isArray(value)) return undefined
70
+ const src = value as Record<string, unknown>
71
+ const out: Record<string, string | string[]> = {}
72
+ for (const spec of DIRECTION_FIELDS) {
73
+ const v = src[spec.key]
74
+ if (typeof v === "string") {
75
+ if (v.length > 0 && v.length <= DIRECTION_ID_MAX_CHARS) out[spec.key] = v
76
+ } else if (Array.isArray(v)) {
77
+ const kept: string[] = []
78
+ for (const x of v) {
79
+ if (kept.length >= DIRECTION_ARRAY_CEILING) break
80
+ if (typeof x === "string" && x.length > 0 && x.length <= DIRECTION_ID_MAX_CHARS) {
81
+ kept.push(x)
82
+ }
83
+ }
84
+ if (kept.length > 0) out[spec.key] = kept
85
+ }
86
+ }
87
+ return Object.keys(out).length > 0 ? (out as DirectionFields) : undefined
88
+ }
89
+
90
+ // ── structured ───────────────────────────────────────────────────────────────
91
+
92
+ type FieldKind = "string" | "number" | "gender"
93
+ const GENDERS = ["man", "woman", "child", "non-binary"] as const
94
+ const MAX_STRUCTURED_VALUE_CHARS = 200
95
+
96
+ type Group = Exclude<keyof StructuredPromptFields, "mood">
97
+ type GroupFields<K extends Group> = Record<keyof NonNullable<StructuredPromptFields[K]>, FieldKind>
98
+
99
+ /**
100
+ * A FIELD-BY-FIELD table (not a registry walk) on purpose:
101
+ * `StructuredPromptFields` is a small HAND-AUTHORED type, not catalog-derived,
102
+ * and the table is what blocks `person: { age: "drop table" }` from rendering
103
+ * verbatim into `jobs.input_data.prompt` — `renderStructuredFields` never
104
+ * throws on junk, but it does render it. Totality is enforced by the
105
+ * `GroupFields<K>` / `{ [K in Group]: … }` mapped types: a new field or group
106
+ * on the published type fails to typecheck here.
107
+ */
108
+ const PERSON_FIELDS: GroupFields<"person"> = {
109
+ age: "number",
110
+ gender: "gender",
111
+ hair: "string",
112
+ eyes: "string",
113
+ expression: "string",
114
+ profession: "string",
115
+ warriorType: "string",
116
+ }
117
+ const STYLING_FIELDS: GroupFields<"styling"> = {
118
+ mood: "string",
119
+ lighting: "string",
120
+ aesthetic: "string",
121
+ colorLook: "string",
122
+ }
123
+ const SETTING_FIELDS: GroupFields<"setting"> = {
124
+ era: "string",
125
+ atmosphere: "string",
126
+ backdrop: "string",
127
+ }
128
+ const CAMERA_FIELDS: GroupFields<"camera"> = {
129
+ framing: "string",
130
+ motion: "string",
131
+ format: "string",
132
+ }
133
+ const LENS_FIELDS: GroupFields<"lens"> = { focalLength: "string", aperture: "string" }
134
+
135
+ const STRUCTURED_GROUPS: { [K in Group]: GroupFields<K> } = {
136
+ person: PERSON_FIELDS,
137
+ styling: STYLING_FIELDS,
138
+ setting: SETTING_FIELDS,
139
+ camera: CAMERA_FIELDS,
140
+ lens: LENS_FIELDS,
141
+ }
142
+
143
+ function readField(v: unknown, kind: FieldKind): string | number | undefined {
144
+ if (kind === "number") return typeof v === "number" && Number.isFinite(v) ? v : undefined
145
+ if (typeof v !== "string" || v.length === 0 || v.length > MAX_STRUCTURED_VALUE_CHARS) {
146
+ return undefined
147
+ }
148
+ if (kind === "gender") return (GENDERS as readonly string[]).includes(v) ? v : undefined
149
+ return v
150
+ }
151
+
152
+ /** Read a node's stored Path-1 structured prompt fields. Same drop-never-throw
153
+ * and `undefined`-never-`{}` contract as `readDirectionFields`. */
154
+ export function readStructuredFields(value: unknown): StructuredPromptFields | undefined {
155
+ if (typeof value !== "object" || value === null || Array.isArray(value)) return undefined
156
+ const src = value as Record<string, unknown>
157
+ const out: Record<string, unknown> = {}
158
+ for (const group of Object.keys(STRUCTURED_GROUPS) as Group[]) {
159
+ const raw = src[group]
160
+ if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue
161
+ const rawRec = raw as Record<string, unknown>
162
+ const kept: Record<string, unknown> = {}
163
+ for (const [field, kind] of Object.entries(
164
+ STRUCTURED_GROUPS[group] as Record<string, FieldKind>,
165
+ )) {
166
+ const v = readField(rawRec[field], kind)
167
+ if (v !== undefined) kept[field] = v
168
+ }
169
+ if (Object.keys(kept).length > 0) out[group] = kept
170
+ }
171
+ const mood = readField(src.mood, "string")
172
+ if (mood !== undefined) out.mood = mood
173
+ return Object.keys(out).length > 0 ? (out as StructuredPromptFields) : undefined
174
+ }
@@ -0,0 +1,45 @@
1
+ /**
2
+ * The reference-binding surface string — `@image_N` / `@video_N` / `@audio_N`
3
+ * — and the one identity sentence built on it. Split out of
4
+ * `video-reference-resolver.ts` so the id-addressed token resolver
5
+ * (`ref-id-tokens.ts`) can bind through the same arrows without a module
6
+ * cycle; the resolver re-exports both, so importers are unaffected.
7
+ */
8
+
9
+ /**
10
+ * The SINGLE swap-point for the reference-binding surface-string (design D1/D7).
11
+ *
12
+ * Every place that renders an `@image_N`-style binding into a video prompt — the
13
+ * per-image subject phrasing, the bare ordinal in a "Use these characters" /
14
+ * pair-back bullet, and the opening/closing frame directive — MUST go through
15
+ * these five arrows. The default form is `@image_N`; if the D7 probe shows a
16
+ * provider prefers the legacy `Image N` form, flipping is editing ONLY these five
17
+ * arrows (`@image_${n}` → `Image ${n}`), nothing downstream.
18
+ *
19
+ * This IS the live swap-point: `resolveVideoReferenceCore` routes the per-image
20
+ * subject phrasing, the "Use these characters" / pair-back bullet ordinals, and
21
+ * the frame directive through these arrows, and `resolveReferenceTokens` resolves
22
+ * the body `{image:N}` tokens through `REF_BINDING[kind]` — so the five arrows
23
+ * are the ONLY emission sites for the binding surface string.
24
+ */
25
+ /**
26
+ * The identity-reference binding sentence shared by the flat-image-list
27
+ * resolvers (gemini-omni, veo i2v): names the ordinal span as identities and
28
+ * says the two things a multimodal model needs to hear — match exactly, and
29
+ * these are not frames. One spelling; both resolvers ride it.
30
+ */
31
+ export function identityRefsSentence(firstOrdinal: number, lastOrdinal: number): string {
32
+ return firstOrdinal === lastOrdinal
33
+ ? `${REF_BINDING.ordinal(firstOrdinal)} is an identity reference for this shot's subjects — match its subject's exact appearance; it is not a frame.`
34
+ : `${REF_BINDING.ordinal(firstOrdinal)} through ${REF_BINDING.ordinal(lastOrdinal)} are identity references for this shot's subjects — match each subject's exact appearance; they are not frames.`
35
+ }
36
+
37
+ export const REF_BINDING = {
38
+ image: (label: string, n: number) => `the ${label} from @image_${n}`,
39
+ video: (label: string, n: number) => `the ${label} from @video_${n}`,
40
+ audio: (label: string, n: number) => `the ${label} from @audio_${n}`,
41
+ /** ordinal as it appears in a "Use these characters" bullet / pair-back */
42
+ ordinal: (n: number) => `@image_${n}`,
43
+ frame: (n: number, role: "opening" | "closing") =>
44
+ `Use @image_${n} as the ${role} (${role === "opening" ? "first" : "last"}) frame of the video.`,
45
+ } as const
@@ -0,0 +1,112 @@
1
+ /**
2
+ * `{ref:<id>}` / `{ref:<id>:<label>}` — id-addressed reference tokens.
3
+ *
4
+ * The API/Studio form of the positional `{image:N}` token: the client names a
5
+ * reference by its OWN `connectedReferences[].id` and the platform substitutes
6
+ * the `@image_N` seat after IT has done the numbering. Without it a client that
7
+ * wanted the binding inline had to mirror the numbering walk client-side — a
8
+ * duplicated rule that misbinds pictures the moment the walk changes.
9
+ *
10
+ * `resolveVideoReferenceCore` builds the `RefIdTokenContext` DURING its walk
11
+ * and calls `resolveRefIdTokens` before the `referenceOrder` reorder (so the
12
+ * binding follows the reference to its final seat); the video routes call it
13
+ * standalone on their no-image-reference early return (nothing seated, so
14
+ * every token degrades).
15
+ */
16
+
17
+ import { REF_BINDING } from "./ref-binding.js"
18
+
19
+ /** The label class of `REFERENCE_TOKEN_RE` (`{image:N:label}`), shared. */
20
+ const REF_TOKEN_LABEL_RE = /^[a-zA-Z0-9_ -]+$/
21
+ /** Cheap gate for the whole pass — a prompt without it is untouched. */
22
+ const HAS_REF_ID_TOKEN_RE = /\{ref:/i
23
+ /**
24
+ * One well-formed `{ref:…}` token: everything between `{ref:` and the next
25
+ * `}` that contains no brace. Greedy over a brace-free class, so the scan is
26
+ * LINEAR in the prompt length whatever the content — `prompt` is up to
27
+ * `PROMPT_HARD_CEILING` (30k) chars of caller-controlled text, so a lazy
28
+ * quantifier with a nested optional label group here would be a quadratic-time
29
+ * ReDoS surface. The id / label split happens in code (`splitLabel`), not in
30
+ * the regex. `ref` is case-insensitive; ids are not.
31
+ */
32
+ const REF_ID_TOKEN_RE = /\{[rR][eE][fF]:([^{}]*)\}/g
33
+ /**
34
+ * Last-resort net for a MALFORMED `{ref:` (a brace inside the id, or no
35
+ * closing `}`): drop the `{ref:` run up to the next whitespace or brace, so
36
+ * the prefix can never reach a model, without eating prose past the token.
37
+ */
38
+ const MALFORMED_REF_ID_TOKEN_RE = /\{[rR][eE][fF]:[^\s{}]*\}?/g
39
+
40
+ /** What `resolveRefIdTokens` resolves against — the numbering walk's output. */
41
+ export interface RefIdTokenContext {
42
+ /** Reference id → the 1-based `@image_N` seat the walk gave it. */
43
+ readonly slotById: ReadonlyMap<string, number>
44
+ /** Reference id → display name, the degrade target of a token that cannot bind. */
45
+ readonly nameById: ReadonlyMap<string, string>
46
+ /**
47
+ * How many image references actually ship — the same range gate `{image:N}`
48
+ * uses. A seat past it (a capped-out or duplicate-URL ref) must not bind.
49
+ */
50
+ readonly imageCount: number
51
+ }
52
+
53
+ /**
54
+ * Split a token's content into `<id>` and an optional `<label>` at the LAST
55
+ * colon — only when the tail is a well-formed label. Ids are opaque and may
56
+ * themselves contain `:` (`slug:variant`) or `/` (a URL), so nothing before
57
+ * the last colon is ever interpreted.
58
+ */
59
+ function splitLabel(content: string): { id: string; label?: string } {
60
+ const at = content.lastIndexOf(":")
61
+ if (at === -1) return { id: content }
62
+ const tail = content.slice(at + 1)
63
+ if (!REF_TOKEN_LABEL_RE.test(tail)) return { id: content }
64
+ return { id: content.slice(0, at), label: tail }
65
+ }
66
+
67
+ /**
68
+ * Rewrite id-addressed reference tokens into the `@image_N` binding of the
69
+ * reference the caller sent under that id.
70
+ *
71
+ * Ids are matched by IDENTITY against the known ids (seated or named), never
72
+ * parsed by character class: the whole content is tried as an id first (the
73
+ * longest reading — an id may itself end in something label-shaped), then
74
+ * `<id>:<label>` split at the last colon, then the token is unknown. The label
75
+ * class is the one `REFERENCE_TOKEN_RE` uses. An id containing `{`, `}`, an
76
+ * `@name:N` mention or a `{image:N}` token is unsupported (the mention pass
77
+ * runs first and would rewrite it; a brace ends the token).
78
+ *
79
+ * Per token:
80
+ * - id seated in range → `REF_BINDING.image(label, N)` when labeled, else the
81
+ * bare `REF_BINDING.ordinal(N)` — exactly what `{image:N[:label]}` emits.
82
+ * - otherwise (unknown id, ref skipped by the walk, capped out, or no image
83
+ * references at all) → the label if given, else the ref's display name if
84
+ * the id is known, else "". A token never ships raw — a malformed one is
85
+ * dropped by the last-resort net.
86
+ *
87
+ * No whitespace tidy here: every caller runs `resolveReferenceTokens` after
88
+ * this (the core does at every return), and that collapses the gap a dropped
89
+ * token leaves. Returns the input untouched when it carries no `{ref:` at all.
90
+ */
91
+ export function resolveRefIdTokens(
92
+ prompt: string | undefined,
93
+ ctx: RefIdTokenContext,
94
+ ): string | undefined {
95
+ if (!prompt || !HAS_REF_ID_TOKEN_RE.test(prompt)) return prompt
96
+ const known = (id: string): boolean => id.length > 0 && (ctx.slotById.has(id) || ctx.nameById.has(id))
97
+ const bind = (id: string, label: string | undefined): string => {
98
+ const slot = ctx.slotById.get(id)
99
+ if (slot !== undefined && slot >= 1 && slot <= ctx.imageCount) {
100
+ return label ? REF_BINDING.image(label, slot) : REF_BINDING.ordinal(slot)
101
+ }
102
+ return label ?? ctx.nameById.get(id) ?? ""
103
+ }
104
+ return prompt
105
+ .replace(REF_ID_TOKEN_RE, (_match, content: string) => {
106
+ if (known(content)) return bind(content, undefined)
107
+ const { id, label } = splitLabel(content)
108
+ if (known(id)) return bind(id, label)
109
+ return label ?? ""
110
+ })
111
+ .replace(MALFORMED_REF_ID_TOKEN_RE, "")
112
+ }