@slatesvideo/shared 0.5.4 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +13 -0
  2. package/dist/index.d.ts +1 -1
  3. package/dist/index.js +2 -2
  4. package/dist/operations/index.d.ts +84 -9
  5. package/dist/operations/index.js +455 -47
  6. package/dist/prompts/character-sheet.d.ts +11 -21
  7. package/dist/prompts/character-sheet.js +124 -59
  8. package/dist/prompts/model-facts.d.ts +1 -1
  9. package/dist/prompts/model-facts.js +32 -0
  10. package/dist/prompts/partials.generated.js +2 -2
  11. package/dist/prompts/prompting-tips.d.ts +1 -1
  12. package/dist/prompts/prompting-tips.js +196 -2
  13. package/dist/prompts/reference-composer.d.ts +1 -1
  14. package/dist/prompts/reference-composer.js +3 -4
  15. package/dist/prompts/reference-rules.d.ts +19 -2
  16. package/dist/prompts/reference-rules.js +21 -4
  17. package/dist/skills/content.js +14 -11
  18. package/exports/slates-prompt-builder/generated/SKILL.md +59 -0
  19. package/{skills/slates-character-turnaround.md → exports/slates-prompt-builder/generated/reference-character.md} +26 -33
  20. package/exports/slates-prompt-builder/generated/reference-content-policy.md +75 -0
  21. package/exports/slates-prompt-builder/generated/reference-kling.md +212 -0
  22. package/exports/slates-prompt-builder/generated/reference-nano-banana.md +182 -0
  23. package/exports/slates-prompt-builder/generated/reference-seedance.md +353 -0
  24. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +79 -0
  25. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  26. package/package.json +7 -3
  27. package/skills/_partials/reference-rules-core.md +1 -1
  28. package/skills/_partials/reference-tips-short.md +1 -1
  29. package/skills/slates-character-identity.md +105 -0
  30. package/skills/slates-edit-and-iterate.md +1 -1
  31. package/skills/slates-model-selection.md +26 -0
  32. package/skills/slates-one-prompt-film.md +3 -3
  33. package/skills/slates-prompting-elevenlabs.md +131 -0
  34. package/skills/slates-prompting-flux-2-max.md +1 -1
  35. package/skills/slates-prompting-gpt-image-2.md +1 -1
  36. package/skills/slates-prompting-kling-v3.md +8 -6
  37. package/skills/slates-prompting-nano-banana-2.md +7 -5
  38. package/skills/slates-prompting-omni-flash.md +1 -1
  39. package/skills/slates-prompting-seed-audio.md +110 -0
  40. package/skills/slates-prompting-seedance.md +15 -9
  41. package/skills/slates-prompting-suno.md +110 -0
  42. package/skills/slates-prompting-veo-3.md +1 -1
@@ -4,33 +4,23 @@
4
4
  * pixels, so it gets the resolution. The body panels exist for build,
5
5
  * proportion, wardrobe and hair, not for the face.
6
6
  *
7
- * The back panel KEEPS its head on purpose: it has no face to compete with the
8
- * portrait, and it is the only panel where hair fall reads.
7
+ * The front panel is headless and the back panel KEEPS its head — that
8
+ * asymmetry is the whole rule. A front-facing body panel renders a ~40px face
9
+ * that cannot match the portrait's, so the sheet would carry two competing
10
+ * identities and the model averages them. A back view has no face to compete
11
+ * with, and it is the only panel where hair fall reads. See the header comment
12
+ * for the receipt, and for why the phrasing must stay FRAMING rather than
13
+ * removal and must be SCOPED TO THE FACE rather than to the whole body.
9
14
  */
10
15
  export declare const CHARACTER_SHEET_PANELS_DESC: string;
11
16
  /** Panel identifiers, in sheet order. */
12
17
  export declare const BODY_POSE_LABELS: readonly ["portrait", "front", "back"];
13
18
  /**
14
- * Expression-sheet close-ups — LEGACY. Kept so characters built before the
15
- * 2026-07-21 single-sheet architecture keep regenerating correctly, and so the
16
- * expression slot remains usable for a character that genuinely needs a
17
- * dedicated expression range.
18
- */
19
- export declare const CHARACTER_EXPRESSIONS_DESC = "neutral expression on left, genuine smile showing teeth in center, serious frown on right";
20
- export declare const EXPRESSION_LABELS: readonly ["neutral", "smile", "serious"];
21
- /**
22
- * The character identity sheet — one asset, three panels, bound to the
23
- * turnaround slot. Named `buildCharacterTurnaroundPrompt` for continuity with
24
- * every existing caller and with the slot it binds to.
19
+ * The character identity sheet — one asset, three panels.
25
20
  *
26
21
  * @param userStyle optional natural-language style transform (e.g. "make her a real person")
27
22
  */
28
- export declare function buildCharacterTurnaroundPrompt(userStyle?: string | null): string;
29
- /**
30
- * Expression sheet — LEGACY close-up face reference. Since 2026-07-21 the
31
- * identity sheet above is the default and this slot is normally left null.
32
- * Generate one only when a character needs an explicit expression range;
33
- * attaching it costs a second reference slot on every generation.
34
- */
35
- export declare function buildExpressionSheetPrompt(userStyle?: string | null): string;
23
+ export declare function buildCharacterIdentityPrompt(userStyle?: string | null): string;
24
+ /** @deprecated Use buildCharacterIdentityPrompt. */
25
+ export declare const buildCharacterTurnaroundPrompt: typeof buildCharacterIdentityPrompt;
36
26
  //# sourceMappingURL=character-sheet.d.ts.map
@@ -2,41 +2,118 @@
2
2
  //
3
3
  // ARCHITECTURE (locked by Eric 2026-07-21): ONE identity sheet per character.
4
4
  // A dominant off-frontal chest-up portrait carries the face; two full-body
5
- // panels (front + back) carry build, wardrobe and hair. The sheet binds to the
6
- // character's TURNAROUND slot and the expression slot is left null.
5
+ // panels (front + back) carry build, wardrobe and hair. The result is the
6
+ // character's one canonical identity reference.
7
7
  //
8
8
  // Why one sheet and not two — three arguments, none of which depend on a
9
9
  // comparison generation:
10
- // 1. Reference-cap economics. Every `@character` mention pushes BOTH bound
11
- // sheets into one reference group, so a two-sheet character costs TWO
12
- // reference slots on every generation. Against real caps that is brutal:
13
- // Kling 3.0 takes 4 ingredients (2 characters, zero room for an
14
- // environment), NB2 has 4 character slots, Seedance 9. One sheet each
15
- // DOUBLES the cast you can stage on every model we route to.
16
- // 2. Competing face renderings drop 6 → 2. The old pair sent three large
17
- // portraits plus three postage-stamp faces (turnaround front + both
18
- // profiles). The model cannot tell which rendering is authoritative and
19
- // averages them; ByteDance documents the same root cause for its
20
- // duplicate-character failure (ModelArk :1959) and prescribes fewer
21
- // competing views. Both profile panels disappear with the shape change.
22
- // 3. One generation instead of two per character — half the sheet spend, one
23
- // asset to inspect and bind.
10
+ // 1. Reference-cap economics. Every character uses one reference slot.
11
+ // 2. One authoritative face avoids averaging competing renderings.
12
+ // 3. One generation means one asset to inspect and bind.
24
13
  //
25
- // KNOWN COST, accepted for v1: a neutral chest-up portrait carries no dental
26
- // information, so a character who smiles in a shot gets invented teeth. The
27
- // 2026-06-26 doctrine already holds that the user's prompt owns expression.
28
- // Revisit only if a receipt shows invented teeth.
14
+ // EXPRESSION IS A SLIGHT SMILE WITH THE TEETH JUST VISIBLE (Eric, 2026-07-30,
15
+ // replacing the v1 neutral default). The v1 note called neutral's missing
16
+ // dental information a "known cost, revisit only if a receipt shows invented
17
+ // teeth" — this is that revisit, and it arrived from the other direction:
18
+ // Eric hand-edited a sheet to smiling and the result "worked really well".
29
19
  //
30
- // DEFERRED, not rejected: cropping the face off the front body panel (the
31
- // "ghost mannequin" treatment). Adopting this layout takes competing faces
32
- // 6 → 2 for free; the crop buys only 2 → 1, and it is contradicted by the only
33
- // visible output in the source corpus. Flipping it later is a one-line change
34
- // here — no migration, no data touched.
20
+ // WHY: a closed-mouth portrait carries ZERO dental information, so every
21
+ // downstream shot where the character smiles has to invent teeth, and teeth are
22
+ // person-specific and stable — inventing them is a highly visible identity
23
+ // break. The smile also records the nasolabial fold, the eye crinkle and where
24
+ // the cheeks sit raised, none of which a neutral mouth shows.
35
25
  //
36
- // BACK-COMPAT IS MANDATORY: ~20 live users have characters bound to BOTH
37
- // slots. `buildExpressionSheetPrompt` stays exported and the expression slot
38
- // stays readable; `mentions.ts` only pushes non-null paths, so old two-sheet
39
- // characters and new one-sheet characters both work unchanged.
26
+ // THE COST, NAMED: `references-read-literally.md` says a baked-in property is
27
+ // read as a property of the SUBJECT, so a smile risks a character who smiles
28
+ // through a beat that asked for grief. Survivable because the 2026-06-26
29
+ // doctrine still holds — the user's prompt owns expression, and downstream
30
+ // prompts name an emotional register on every delivered line.
31
+ //
32
+ // SLIGHT, never a grin: a broad smile deforms eyes, cheeks and mouth enough
33
+ // that the model has to un-deform it for any neutral shot.
34
+ //
35
+ // NON-HUMANS ARE CARVED OUT (Eric, 2026-07-30). A smile clause on a horse, a
36
+ // dragon or a robot produces bared teeth. Non-human characters get a natural
37
+ // neutral expression instead. NOTE THE SCOPE DIFFERS from the A-pose carve-out
38
+ // beside it: that one is anatomical (quadruped / non-bipedal), this one is
39
+ // about having a human mouth — so a BIPEDAL robot or humanoid alien is covered
40
+ // by this carve-out and not by that one. Two carve-outs, deliberately, because
41
+ // one predicate does not fit both.
42
+ //
43
+ // Receipt strength: N=1, and it is a "looked right" judgement, not a scored
44
+ // identity-hold comparison against the neutral sheets. HOW YOU'D KNOW THIS IS
45
+ // BEATEN: a character smiling through a beat prompted angry or grieving, or
46
+ // identity holding measurably worse than neutral did. Reverting is one line.
47
+ //
48
+ // THE FRONT PANEL IS HEADLESS (shipped 2026-07-22, receipt-gated). The face is
49
+ // cropped off the front body panel, taking competing face renderings 2 → 1.
50
+ // ONLY the face — the body, neck, arms and hands render normally; see the
51
+ // 2026-07-30 (b) note below for what happens when that isn't said. The BACK
52
+ // panel keeps its head: it has no
53
+ // face to compete with the portrait, and it is the only panel where hair fall
54
+ // reads. The rule is "kill every competing rendering of the FACE", not "kill
55
+ // every head".
56
+ //
57
+ // Two things had to be true before this shipped, and both were verified on
58
+ // real generations (research/model-prompting-research.md, "Head-crop receipt"):
59
+ // 1. It is prompt-reachable. NB2 renders a clean invisible-mannequin panel
60
+ // with no refusal. THIS DEPENDS ON THE PHRASING: it is framed as FRAMING
61
+ // ("cropped at the collarbone, invisible-mannequin presentation"), a
62
+ // standard e-commerce genre with deep training data. Never phrase it as
63
+ // removal or decapitation.
64
+ //
65
+ // 2026-07-30 (a) — THE PHRASING NARROWED, and the receipt above was
66
+ // MODEL-SCOPED. "framed from the collarbone down with the head not shown"
67
+ // passed NB2 and is a HARD 422 on gpt-image-2: fal returns
68
+ // content_policy_violation with loc ["body","prompt"], so the text is
69
+ // rejected before any image is read. Diagnosis: "the head not shown"
70
+ // states an anatomical ABSENCE — a headless human body — which reads as
71
+ // gore to OpenAI's classifier. "cropped at the collarbone" states a
72
+ // CAMERA fact and carries the same instruction.
73
+ // RULE: describe an exclusion as a framing choice, never as a missing
74
+ // body part. The "invisible-mannequin" genre anchor is KEPT because the
75
+ // NB2 receipt says it is what makes the panel reachable there.
76
+ // HOW YOU'D KNOW THIS IS BEATEN: a front panel that comes back with a
77
+ // head on it, meaning "cropped at the collarbone" alone is too weak
78
+ // without the absence clause. If that happens, the fix is a
79
+ // model-conditional phrasing, not restoring the 422.
80
+ //
81
+ // 2026-07-30 (b) — THE GENRE ANCHOR HAS TO BE SCOPED TO THE FACE, and
82
+ // this one cost real generations. "an invisible-mannequin presentation
83
+ // WHERE THE CLOTHING HOLDS ITS OWN SHAPE" is the e-commerce genre stated
84
+ // in full — and the full genre means NO BODY AT ALL, an empty outfit
85
+ // photographed on nothing. Eric's sheets came back with the skin removed:
86
+ // no neck, no hands, no forearms, a floating garment. The genre anchor
87
+ // was doing exactly what it says.
88
+ // Eric's replacement, verbatim, and the default since: "an invisible-
89
+ // mannequin presentation with just the face cropped out." Same anchor
90
+ // (still what makes the panel reachable on NB2), bounded so the only
91
+ // thing missing is the face.
92
+ // RULE: a genre anchor imports the WHOLE genre unless you bound it —
93
+ // name what STAYS, not just what goes. This is the same failure shape as
94
+ // (a) from the opposite side: (a) was an exclusion phrased too
95
+ // anatomically, (b) was an exclusion scoped too widely.
96
+ // HOW YOU'D KNOW THIS IS BEATEN: front panels returning with a head on
97
+ // them — "just the face cropped out" would then be reading as a face
98
+ // edit rather than a crop, leaving the collarbone clause to carry it
99
+ // alone.
100
+ // 2. The literal-reading law does NOT fire on it. This was the real risk and
101
+ // it is ours, not the source corpus's: `references-read-literally.md` says
102
+ // a baked-in property is read as a property of the SUBJECT, and this panel
103
+ // renders headlessness as CONTENT (empty plate above the collar), not as
104
+ // photographic framing. The predicted failure was a headless or
105
+ // neck-glitched downstream character. A Kling shot from a bound sheet came
106
+ // back head intact, identity holding, wardrobe held — in a MULTI-CHARACTER
107
+ // frame, which is the scope ByteDance's averaging failure lives in
108
+ // (ModelArk :1948-1994), so it is the hardest form of the test.
109
+ //
110
+ // Receipt strength: N=1 character, decisive for filterability, strong for the
111
+ // literal-reading risk, NOT a scored V2-vs-V3 comparison — nobody measured
112
+ // whether 2 → 1 improves identity hold, only that it doesn't break. HOW YOU'D
113
+ // KNOW THIS IS BEATEN: a downstream character generating headless,
114
+ // neck-glitched, or with a floating collar; or a scored run where heads-kept
115
+ // holds identity better. Reverting is a one-line change here — no migration,
116
+ // no data touched. Quadrupeds are carved out below (a horse has no collarbone).
40
117
  //
41
118
  // SOURCE OF TRUTH. The desktop imports these builders through
42
119
  // `slate/src/shared/prompts/character-sheet.ts` (a thin re-export since 1.2.1)
@@ -50,22 +127,20 @@ import { renderStyleInstruction } from './style-library.js';
50
127
  * pixels, so it gets the resolution. The body panels exist for build,
51
128
  * proportion, wardrobe and hair, not for the face.
52
129
  *
53
- * The back panel KEEPS its head on purpose: it has no face to compete with the
54
- * portrait, and it is the only panel where hair fall reads.
130
+ * The front panel is headless and the back panel KEEPS its head — that
131
+ * asymmetry is the whole rule. A front-facing body panel renders a ~40px face
132
+ * that cannot match the portrait's, so the sheet would carry two competing
133
+ * identities and the model averages them. A back view has no face to compete
134
+ * with, and it is the only panel where hair fall reads. See the header comment
135
+ * for the receipt, and for why the phrasing must stay FRAMING rather than
136
+ * removal and must be SCOPED TO THE FACE rather than to the whole body.
55
137
  */
56
138
  export const CHARACTER_SHEET_PANELS_DESC = 'a large chest-up portrait on the left at a three-quarter angle (never dead-on), ' +
57
- 'a full-body front view in a relaxed A-pose in the centre, ' +
58
- 'and a full-body back view on the right';
139
+ 'a full-body front view in a relaxed A-pose in the centre, cropped at the collarbone — ' +
140
+ 'an invisible-mannequin presentation with just the face cropped out, ' +
141
+ 'and a full-body back view on the right with the head and hair fully visible';
59
142
  /** Panel identifiers, in sheet order. */
60
143
  export const BODY_POSE_LABELS = ['portrait', 'front', 'back'];
61
- /**
62
- * Expression-sheet close-ups — LEGACY. Kept so characters built before the
63
- * 2026-07-21 single-sheet architecture keep regenerating correctly, and so the
64
- * expression slot remains usable for a character that genuinely needs a
65
- * dedicated expression range.
66
- */
67
- export const CHARACTER_EXPRESSIONS_DESC = 'neutral expression on left, genuine smile showing teeth in center, serious frown on right';
68
- export const EXPRESSION_LABELS = ['neutral', 'smile', 'serious'];
69
144
  // The sheet's style directive: a user transform REPLACES the inherit-source
70
145
  // instruction (so the model isn't told to both preserve the medium AND change
71
146
  // it); otherwise inherit the source medium.
@@ -73,31 +148,21 @@ function styleDirective(userStyle) {
73
148
  return renderStyleInstruction(userStyle).trim() || INHERIT_SOURCE_STYLE;
74
149
  }
75
150
  /**
76
- * The character identity sheet — one asset, three panels, bound to the
77
- * turnaround slot. Named `buildCharacterTurnaroundPrompt` for continuity with
78
- * every existing caller and with the slot it binds to.
151
+ * The character identity sheet — one asset, three panels.
79
152
  *
80
153
  * @param userStyle optional natural-language style transform (e.g. "make her a real person")
81
154
  */
82
- export function buildCharacterTurnaroundPrompt(userStyle) {
155
+ export function buildCharacterIdentityPrompt(userStyle) {
83
156
  return (`A single character identity reference sheet of one character, three panels side by side on one plate: ` +
84
157
  `${CHARACTER_SHEET_PANELS_DESC}. ` +
85
158
  `The portrait is the largest panel and occupies roughly a quarter to a third of the sheet — it is the sole authority for the face, so render it at maximum facial detail. ` +
86
- `Neutral expression and identical appearance, wardrobe and hair across all three panels. ` +
159
+ `No second rendering of the face anywhere on the sheet. ` +
160
+ `A slight natural smile with the teeth just visible, and identical appearance, wardrobe and hair across all three panels. ` +
87
161
  `${styleDirective(userStyle)} ${IDENTITY_LIGHTING_CLAUSE} ${IDENTITY_CRAFT_CLAUSE} ` +
88
- `For quadruped or non-bipedal characters, replace the A-pose with a natural standing stance and keep the same three-panel layout. ` +
162
+ `For non-human characters, use a natural neutral expression instead of a smile. ` +
163
+ `For quadruped or non-bipedal characters, replace the A-pose with a natural standing stance, show the whole animal including the head on both body panels, and keep the same three-panel layout. ` +
89
164
  `No text, no labels, no captions, no panel borders.`);
90
165
  }
91
- /**
92
- * Expression sheet — LEGACY close-up face reference. Since 2026-07-21 the
93
- * identity sheet above is the default and this slot is normally left null.
94
- * Generate one only when a character needs an explicit expression range;
95
- * attaching it costs a second reference slot on every generation.
96
- */
97
- export function buildExpressionSheetPrompt(userStyle) {
98
- return (`Character expression reference sheet with 3 head and shoulder portraits arranged side by side horizontally: ${CHARACTER_EXPRESSIONS_DESC}. ` +
99
- `Consistent character appearance across all three. ` +
100
- `${styleDirective(userStyle)} ${IDENTITY_LIGHTING_CLAUSE} ${IDENTITY_CRAFT_CLAUSE} Same framing for each. ` +
101
- `No text, no labels, no captions.`);
102
- }
166
+ /** @deprecated Use buildCharacterIdentityPrompt. */
167
+ export const buildCharacterTurnaroundPrompt = buildCharacterIdentityPrompt;
103
168
  //# sourceMappingURL=character-sheet.js.map
@@ -1,7 +1,7 @@
1
1
  export interface ModelFact {
2
2
  id: string;
3
3
  label: string;
4
- kind: 'image' | 'video';
4
+ kind: 'image' | 'video' | 'audio';
5
5
  /** Max reference images (image models) — null if not applicable. */
6
6
  maxRefImages: number | null;
7
7
  /** Max ingredient images (video models) — null if not applicable. */
@@ -103,6 +103,38 @@ export const MODEL_FACTS = [
103
103
  maxIngredients: 0, // prompt + source clip ONLY — no element/style refs on this endpoint
104
104
  notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (7/09 head-to-head vs Kling edit on real talking footage: lips held perfectly, audio near-identical, both action beats landed). Takes an EXISTING 3-10s clip and changes what the prompt names, footage-synced (prop/effect/environment/lighting swaps). Fidelity is EARNED by prompt discipline: ONE short instruction + "Keep everything else the same." — long descriptive prompts DESTROY it (Google-documented + 7/09 receipt). Never name objects as metaphors ("candle-like" → literal candle). Quirk: occasional tail jitter/doubled last speech beat — trim the tail. NO reference images (identity swaps needing refs → Kling edit); bit-exact audio needs → Kling keep_audio or segment-splice. 720p output, cheapest edit seat (~2/3 of Kling edit Std).',
105
105
  },
106
+ {
107
+ id: 'seed-audio',
108
+ label: 'Seed Audio 1.0',
109
+ kind: 'audio',
110
+ maxRefImages: 1, // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
111
+ maxIngredients: null,
112
+ notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
113
+ },
114
+ {
115
+ id: 'eleven-v3',
116
+ label: 'ElevenLabs Eleven v3 (TTS)',
117
+ kind: 'audio',
118
+ maxRefImages: null,
119
+ maxIngredients: null,
120
+ notes: 'CONTROLLED, REPEATABLE named-voice VOICEOVER — route here whenever the exact words matter and must be re-renderable in the same voice (ad reads, narration, character lines to lip-sync against). AUDIO-ONLY. The text field IS the script: it is spoken verbatim, so never put stage directions in it. 1-5000 characters, billed per 100-character bucket, so trimming a sentence genuinely saves credits. 20 preset voices (Rachel default) — pick one and keep it for the whole piece. stability 0-1 trades consistency against expressiveness (low = more emotive and more variable). No voice cloning on this route. Word-level timestamps come back free and are what a future caption pass consumes. For scene ambience or SFX rather than speech, use seed-audio / eleven-sfx.',
121
+ },
122
+ {
123
+ id: 'eleven-sfx',
124
+ label: 'ElevenLabs Sound Effects v2',
125
+ kind: 'audio',
126
+ maxRefImages: null,
127
+ maxIngredients: null,
128
+ notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
129
+ },
130
+ {
131
+ id: 'suno',
132
+ label: 'Suno',
133
+ kind: 'audio',
134
+ maxRefImages: null,
135
+ maxIngredients: null,
136
+ notes: 'FULL MUSIC TRACKS with structure — route here for anything a listener would call a song or a score. AUDIO-ONLY. EVERY call returns TWO variations for one flat price, and duration is FREE up to 360s (measured 2026-07-31: a 240s track costs the same as a default one), which makes it the cheapest way to get a bed that outlasts a cut. Two modes: DESCRIPTION mode (customMode=false, prompt is a <=500-char description and the lyrics get written for you) and CUSTOM mode (customMode=true, needs style + title; prompt then holds the EXACT LYRICS, sung as written — never put a description there). instrumental=true scores a scene with no vocals. Steer with style/negativeTags rather than piling adjectives into the prompt. duration 10-360s is V5_5 + custom mode only. Unofficial API wrapper, so treat availability as best-effort.',
137
+ },
106
138
  ];
107
139
  const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
108
140
  export function getModelFact(id) {
@@ -6,8 +6,8 @@
6
6
  // no longer disagree. Edit the partial, not this file, not the skills.
7
7
  export const PARTIALS = {
8
8
  "decision-log": "When you surface the plan, include a short **decision log** — one line per decision *you* made that the user did not specify:\n\n```\nsource phrase or declared default → what you wrote → what it resolves\n\"in a diner\" → chrome-and-vinyl booth, 3/4 on the counter → fixes the anchor so blocking is repeatable\n(no time of day) → late afternoon, low warm key → default; say the word and it changes\n(no camera) → slow push-in, single move → one move per shot; stacking increases instability\n```\n\n**Hard rule: never silently add weather, props, style, or camera movement.** If it wasn't in the brief and you added it, it goes in the log. This is the \"why did you add that?\" affordance — for an agent that writes prompts on the user's behalf and spends their credits, it is what keeps the model in assembly and the user in the director's chair.\n\n> ❌ **Do NOT turn this into a question gate.** Clarifying questions before optimizing directly fight the locked fast-path rule: *if intent is clear, generate immediately with sane defaults, don't ask questions; only ask for production intent, and batch every question into one message.* Log the decisions, then go. The log is an **output**, not an interrogation — surfaced alongside the plan, never as a separate ceremony, and never as a reason to wait.",
9
- "reference-rules-core": "Identity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates composes the naming for you from your `@mentions` / `#tags` — you never hand-write role labels.\n3. **One identity sheet per character — and whatever you do attach for a subject, NAME it as one entity.** A character's identity sheet is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is always better, because the model cannot tell which one is authoritative and averages them.** Where a character carries a second bound sheet — an explicit expression range, or a legacy turnaround+expression pair — cite BOTH under the SAME name, `Marcus (images 1 and 2)`. That shared name, not a role essay, is what tells the model they are ONE person and stops the varied expressions from averaging the face. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.",
10
- "reference-tips-short": "Name each reference inline; never write role essays. Slates does this for you: `@mention` a subject or environment and it composes `Marcus (images 1 and 2) in the cafe (image 3)`, citing them in the exact order it sends them. Citing both of a character's sheets under the SAME name is what tells the model they are one person — a \"Reference Image Instructions\" block does the opposite and drags the sheet's studio lighting into your scene. Start with 2-3 focused refs.",
9
+ "reference-rules-core": "Identity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates composes the naming for you from your `@mentions` / `#tags` — you never hand-write role labels.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.",
10
+ "reference-tips-short": "Name each reference inline; never write role essays. Slates does this for you: `@mention` a subject or environment and it composes `Marcus (image 1) in the cafe (image 2)`, citing them in the exact order it sends them. One canonical identity image avoids competing facial renderings; a \"Reference Image Instructions\" block drags reference lighting into your scene. Start with 2-3 focused refs.",
11
11
  "references-read-literally": "> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.",
12
12
  "still-gate": "**A visible defect in the still is already a STOP.** Do not animate it. Fix the frame first, then move to motion — and go to motion only when the crop passes the still scan and you genuinely need movement to confirm an uncertain edge, reflection, or object.\n\nThis is a **cost** rule as much as a craft rule: a 1080p/10s premium video generation costs many multiples of an image re-roll, and video is where a defect stops being fixable. Anything wrong in the still gets worse in motion — soft geometry mushes, broken-but-plausible objects fall apart, oily textures start crawling. **Animating a known-bad frame is the single most expensive mistake in the pipeline.** Re-rolling the image is the cheap move; re-rolling the video is not.",
13
13
  };
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
16
16
  /** Footer callout paragraphs. */
17
17
  footer?: string[];
18
18
  }
19
- export type PromptingTipsKey = 'seedance' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite';
19
+ export type PromptingTipsKey = 'seedance' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-v3' | 'eleven-sfx' | 'suno';
20
20
  export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
21
21
  /** Null when no tips exist for the key — callers render an honest fallback. */
22
22
  export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
@@ -212,7 +212,7 @@ const OMNI_FLASH = {
212
212
  [
213
213
  {
214
214
  heading: 'Name references inline',
215
- example: 'Marcus (images 1 and 2) walks into the cafe...',
215
+ example: 'Marcus (image 1) walks into the cafe...',
216
216
  note: 'Up to 7 reference images merge into one list — refer to them by number in the prompt.',
217
217
  },
218
218
  {
@@ -325,7 +325,7 @@ const NANO_BANANA = {
325
325
  },
326
326
  {
327
327
  heading: 'Reference images — name them, never label roles',
328
- example: 'Marcus (images 1 and 2) sits across from the woman (image 3) in the cafe (image 4).',
328
+ example: 'Marcus (image 1) sits across from the woman (image 2) in the cafe (image 3).',
329
329
  note: `Up to 14 refs (10 object + 4 character — caps don't trade). ${PARTIALS['reference-tips-short']}`,
330
330
  },
331
331
  {
@@ -359,6 +359,196 @@ const NANO_BANANA_LITE = {
359
359
  : card),
360
360
  ],
361
361
  };
362
+ // ── Audio lane ──────────────────────────────────────────────────
363
+ // The four audio surfaces prompt NOTHING like the video models. The single
364
+ // most expensive mistake is bringing Kling's "SFX:" / "Ambient noise:" syntax
365
+ // to Seed Audio, which reads it as literal text. Every entry below leads with
366
+ // what the surface actually wants.
367
+ const SEED_AUDIO = {
368
+ label: 'Seed Audio 1.0',
369
+ intro: [
370
+ 'Seed Audio builds a whole audio scene — dialogue, effects and ambience together — from one plain sentence. Write it the way you would describe the moment to a person standing next to you, not the way you would write a video prompt.',
371
+ 'It has no duration setting. Length comes from the words, so Slates appends your chosen duration to the prompt ("… 15 seconds") and bills exactly that. Set the duration control to what you actually want and let the sentence stay clean.',
372
+ ],
373
+ columns: [
374
+ [
375
+ {
376
+ heading: 'One plain sentence',
377
+ example: 'nature soundscape, wide open field cicadas and birds and a loon.',
378
+ note: 'No shot language, no production jargon, no formatting. Plain description outperforms anything that reads like a spec sheet.',
379
+ },
380
+ {
381
+ heading: 'Duration lives in the prompt',
382
+ example: 'tiny applause of 2 or 3 people at an open mic. 15 seconds',
383
+ note: 'The duration control writes this for you. Do not also type a different length into your sentence — the two will fight and you pay for the one you selected.',
384
+ critical: true,
385
+ },
386
+ {
387
+ heading: 'Say the crowd size',
388
+ example: 'tiny applause of 2 or 3 people · a packed arena roaring',
389
+ note: '"Applause" alone returns a full room. Scale words are the single highest-leverage edit on any crowd, traffic or nature bed.',
390
+ },
391
+ {
392
+ heading: 'Cut beds longer than the shot',
393
+ note: 'Ask for a few seconds more than the clip needs so the edit has handles to fade in and out of. Beds that end exactly on the cut always sound clipped.',
394
+ },
395
+ ],
396
+ [
397
+ {
398
+ heading: 'No Kling syntax here',
399
+ example: '✗ SFX: heavy boots\n✓ heavy boots on wet pavement, a siren far off',
400
+ note: 'The "SFX:" and "Ambient noise:" prefixes belong to Kling video prompts. Seed Audio treats them as words in the scene and the result gets worse.',
401
+ critical: true,
402
+ },
403
+ {
404
+ heading: 'Dialogue in quotes',
405
+ example: 'a tired bartender says, "we closed twenty minutes ago", glasses clinking behind him',
406
+ note: 'Speech goes in quotes inside the same sentence as the room. Pick a preset voice for a specific speaker, or leave it unset and let the scene cast itself.',
407
+ },
408
+ {
409
+ heading: 'References',
410
+ example: 'match the room tone of @Audio1',
411
+ note: 'Up to 3 audio clips (max 30s each), referenced as @Audio1–@Audio3 — OR one image to score what is in frame. Never both in the same generation.',
412
+ },
413
+ {
414
+ heading: 'Know its seat',
415
+ note: 'Scenes, beds and room tone in one pass. For an exact script in a repeatable voice use Eleven v3; for a single effect on a specific frame use Sound Effects; for a song use Suno.',
416
+ },
417
+ ],
418
+ ],
419
+ };
420
+ const ELEVEN_V3 = {
421
+ label: 'ElevenLabs Eleven v3',
422
+ intro: [
423
+ 'Eleven v3 speaks your text verbatim in a named voice. It is the controlled, repeatable lane: the same text and the same voice give you a read you can regenerate after a script tweak without the performance drifting.',
424
+ 'Billing is per 100 characters of text, rounded up — so tightening a sentence genuinely costs less, and a stray pasted paragraph genuinely costs more.',
425
+ ],
426
+ columns: [
427
+ [
428
+ {
429
+ heading: 'The text field is the script',
430
+ example: '✗ (excited) Say this fast: Grab yours today!\n✓ Grab yours today!',
431
+ note: 'Everything you type gets spoken. Stage directions, character names and bracketed notes will be read out loud.',
432
+ critical: true,
433
+ },
434
+ {
435
+ heading: 'Punctuate for pace',
436
+ example: 'It works. Every time. · It works — every time…',
437
+ note: 'Full stops, commas, dashes and ellipses are your only timing controls. Rewrite the punctuation before you touch the settings.',
438
+ },
439
+ {
440
+ heading: 'Pick a voice and stay',
441
+ note: '20 preset voices. Choosing one per character or per piece is what makes a series sound deliberate; swapping voices mid-piece reads as a mistake.',
442
+ },
443
+ ],
444
+ [
445
+ {
446
+ heading: 'Stability',
447
+ example: '0.3 emotive · 0.5 default · 0.8 steady',
448
+ note: 'Lower is more expressive and more variable take-to-take. Higher is flatter and more repeatable. Raise it for long narration, lower it for a single dramatic line.',
449
+ },
450
+ {
451
+ heading: 'Spell out the tricky bits',
452
+ example: 'SKU → "ess kay you" · 2026 → "twenty twenty six"',
453
+ note: 'Acronyms, product names, prices and years are where TTS embarrasses itself. Write the pronunciation you want.',
454
+ },
455
+ {
456
+ heading: 'Know its seat',
457
+ note: 'Exact words, repeatable voice, lines you will lip-sync against. Ambience, crowds and rooms belong to Seed Audio; one-shot effects to Sound Effects.',
458
+ },
459
+ ],
460
+ ],
461
+ footer: [
462
+ 'Word-level timestamps come back with every generation at no extra cost — that is what a future caption pass will read, so there is no reason to turn them off.',
463
+ ],
464
+ };
465
+ const ELEVEN_SFX = {
466
+ label: 'ElevenLabs Sound Effects',
467
+ intro: [
468
+ 'Sound Effects makes one short sound with an exact length — the lane for a hit that has to land on a specific frame, or a seamless loop you can lay under a whole scene.',
469
+ 'Duration is always sent explicitly (0.5–22s). Billing is per second, so the length you pick is the price you pay.',
470
+ ],
471
+ columns: [
472
+ [
473
+ {
474
+ heading: 'Describe the cause, not the label',
475
+ example: '✗ door sound\n✓ heavy oak door slams shut in a stone hallway',
476
+ note: 'Material, weight and room are what separate a usable effect from a stock-library shrug. Name all three.',
477
+ critical: true,
478
+ },
479
+ {
480
+ heading: 'One sound per generation',
481
+ note: 'This surface makes a single event. A door, then footsteps, then a siren is three generations layered on the timeline — or one Seed Audio scene.',
482
+ },
483
+ {
484
+ heading: 'Duration is the edit',
485
+ example: '0.8s for an impact · 4s for a whoosh · 22s for a bed',
486
+ note: 'Ask for roughly the length you need. A 4-second request for a door slam pads the tail with room tone you then have to trim.',
487
+ },
488
+ ],
489
+ [
490
+ {
491
+ heading: 'Loops',
492
+ example: 'steady rain on a canvas tent (loop on, 12s)',
493
+ note: 'Turn loop on for anything continuous — rain, engine hum, crowd murmur — and it will tile without a seam.',
494
+ },
495
+ {
496
+ heading: 'Prompt influence',
497
+ example: '0.3 default · 0.7 literal',
498
+ note: 'Higher hugs your wording with less variation between takes; lower explores. Raise it when a re-roll keeps wandering off the brief.',
499
+ },
500
+ {
501
+ heading: 'Know its seat',
502
+ note: 'One precise effect on a known frame. Full rooms and layered scenes are cheaper and better in one Seed Audio pass.',
503
+ },
504
+ ],
505
+ ],
506
+ };
507
+ const SUNO = {
508
+ label: 'Suno',
509
+ intro: [
510
+ 'Suno writes full music. Every generation returns TWO variations for one flat price, and length is free up to six minutes — so there is never a reason to generate a bed that is shorter than your edit.',
511
+ 'The one thing to get right is which mode you are in, because it changes what the prompt field means.',
512
+ ],
513
+ columns: [
514
+ [
515
+ {
516
+ heading: 'Description mode',
517
+ example: 'brooding synthwave for a night drive, analog bass, no vocals',
518
+ note: 'The prompt is a description (max 500 characters) and the lyrics get written for you. This is the fast path when you just need a mood.',
519
+ },
520
+ {
521
+ heading: 'Custom mode — the prompt IS the lyrics',
522
+ example: 'Style: dream pop, hazy\nTitle: Blue Hour\nPrompt: [Verse 1] The lights come on…',
523
+ note: 'In custom mode the prompt is sung exactly as written. Putting a description there gets your description sung back at you.',
524
+ critical: true,
525
+ },
526
+ {
527
+ heading: 'Instrumental',
528
+ note: 'Turn instrumental on to score a scene with no vocals — style and title still steer it, and the prompt field is ignored.',
529
+ },
530
+ ],
531
+ [
532
+ {
533
+ heading: 'Steer with style, not adjectives',
534
+ example: 'Style: 90s trip-hop, dusty breakbeat, Rhodes\nAvoid: brass, EDM drops',
535
+ note: 'Genre, era, instrumentation and tempo belong in the style field. Negative tags remove what keeps creeping in.',
536
+ },
537
+ {
538
+ heading: 'Length is free',
539
+ example: '10–360 seconds',
540
+ note: 'A six-minute track costs exactly what a default one does. Ask for longer than the cut needs and trim on the timeline.',
541
+ },
542
+ {
543
+ heading: 'Two songs, both yours',
544
+ note: 'Both variations land in the gallery. They are genuinely different takes on the same brief — audition both before re-rolling.',
545
+ },
546
+ ],
547
+ ],
548
+ footer: [
549
+ 'Files are hosted by the provider for a limited window — Slates downloads and stores them on your machine as soon as the track finishes, so nothing expires out from under a project.',
550
+ ],
551
+ };
362
552
  export const PROMPTING_TIPS = {
363
553
  seedance: SEEDANCE,
364
554
  kling: KLING,
@@ -368,6 +558,10 @@ export const PROMPTING_TIPS = {
368
558
  'omni-flash-edit': OMNI_FLASH_EDIT,
369
559
  'nano-banana': NANO_BANANA,
370
560
  'nano-banana-lite': NANO_BANANA_LITE,
561
+ 'seed-audio': SEED_AUDIO,
562
+ 'eleven-v3': ELEVEN_V3,
563
+ 'eleven-sfx': ELEVEN_SFX,
564
+ suno: SUNO,
371
565
  };
372
566
  /** Null when no tips exist for the key — callers render an honest fallback. */
373
567
  export function getPromptingTips(key) {
@@ -14,7 +14,7 @@ export interface ReferenceGroup {
14
14
  /** Display + citation name: 'Marcus' | 'the cafe' | 'noir'. Used verbatim. */
15
15
  name: string;
16
16
  kind: ReferenceKind;
17
- /** A group can carry several images (e.g. turnaround + expression). */
17
+ /** A group can carry several images for workflows that genuinely need them. */
18
18
  media: ReferenceMedia[];
19
19
  }
20
20
  export interface ComposedReferences {
@@ -14,10 +14,9 @@
14
14
  // the desktop generation + rail read from this one function, so the rail's badge
15
15
  // numbers and the prompt's "image N" citations can never desync.
16
16
  //
17
- // Naming is the ONLY identity signal. Citing both of a subject's images as the
18
- // SAME name ("Marcus (images 1 and 2)") tells the model they are ONE entity —
19
- // which is what prevents a multi-image bucket from averaging into a blended
20
- // face. This IS each model's own official consistency lever (NB2 "assign a
17
+ // Naming is the identity signal. Citing the canonical subject image inline
18
+ // ("Marcus (image 1)") tells the model which reference owns that entity. This
19
+ // is each model's own official consistency lever (NB2 "assign a
21
20
  // distinct name", Seedance "Reference Subject_N in Image_N", Kling "reuse a fixed
22
21
  // label verbatim"); the heavy role-essay block was the off-doctrine part.
23
22
  // Normalize a name/token for matching: drop the sigil, lowercase, strip