@slatesvideo/shared 0.6.11 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/dist/auth.js +2 -2
  2. package/dist/clients/cloud.js +1 -1
  3. package/dist/index.d.ts +1 -1
  4. package/dist/index.js +1 -1
  5. package/dist/manual/content.d.ts +1 -1
  6. package/dist/manual/content.js +1 -1
  7. package/dist/operations/index.d.ts +817 -16
  8. package/dist/operations/index.js +1410 -360
  9. package/dist/operations/surface.d.ts +3 -1
  10. package/dist/operations/surface.js +37 -10
  11. package/dist/prompts/ad-presets.d.ts +77 -0
  12. package/dist/prompts/ad-presets.js +43 -0
  13. package/dist/prompts/agent-doctrine.js +5 -4
  14. package/dist/prompts/banned-tokens.d.ts +4 -29
  15. package/dist/prompts/banned-tokens.js +29 -204
  16. package/dist/prompts/craft-cards.js +2 -2
  17. package/dist/prompts/generation-policy.d.ts +41 -0
  18. package/dist/prompts/generation-policy.js +53 -0
  19. package/dist/prompts/guide-retrieval.d.ts +9 -0
  20. package/dist/prompts/guide-retrieval.js +53 -0
  21. package/dist/prompts/index.d.ts +1 -0
  22. package/dist/prompts/index.js +1 -0
  23. package/dist/prompts/model-capabilities.d.ts +18 -1
  24. package/dist/prompts/model-capabilities.js +72 -19
  25. package/dist/prompts/model-facts.d.ts +34 -2
  26. package/dist/prompts/model-facts.js +66 -5
  27. package/dist/prompts/partials.generated.js +8 -2
  28. package/dist/prompts/prompting-tips.d.ts +1 -1
  29. package/dist/prompts/prompting-tips.js +61 -16
  30. package/dist/prompts/reference-composer.d.ts +2 -0
  31. package/dist/prompts/reference-composer.js +51 -50
  32. package/dist/prompts/script-document.d.ts +165 -0
  33. package/dist/prompts/script-document.js +11 -0
  34. package/dist/prompts/shot-grammar.d.ts +4 -4
  35. package/dist/prompts/shot-grammar.js +3 -3
  36. package/dist/prompts/shot-spec.d.ts +13 -0
  37. package/dist/prompts/shot-spec.js +23 -5
  38. package/dist/skills/content.js +26 -23
  39. package/exports/slates-chatgpt-images/generated/SKILL.md +107 -0
  40. package/exports/slates-chatgpt-images/generated/slates-chatgpt-images.skill +0 -0
  41. package/exports/slates-prompt-builder/generated/SKILL.md +1 -1
  42. package/exports/slates-prompt-builder/generated/reference-character.md +9 -1
  43. package/exports/slates-prompt-builder/generated/reference-kling.md +3 -3
  44. package/exports/slates-prompt-builder/generated/reference-nano-banana.md +22 -10
  45. package/exports/slates-prompt-builder/generated/reference-seedance.md +4 -4
  46. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +17 -17
  47. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  48. package/package.json +10 -4
  49. package/skills/_partials/cinematic-card.md +8 -0
  50. package/skills/_partials/cinematic-routes-short.md +2 -0
  51. package/skills/_partials/cinematic-tips-short.md +2 -0
  52. package/skills/_partials/decision-log.md +1 -13
  53. package/skills/_partials/image-defaults.md +11 -0
  54. package/skills/_partials/lens-video-split.md +1 -0
  55. package/skills/_partials/reference-rules-core.md +1 -1
  56. package/skills/_partials/sheet-tool-defaults.md +6 -0
  57. package/skills/slates-character-identity.md +9 -1
  58. package/skills/slates-chatgpt-images.md +107 -0
  59. package/skills/slates-cinematic-look.md +237 -0
  60. package/skills/slates-cost-discipline.md +18 -12
  61. package/skills/slates-direct-response-ad.md +13 -53
  62. package/skills/slates-edit-and-iterate.md +1 -1
  63. package/skills/slates-model-selection.md +20 -14
  64. package/skills/slates-one-prompt-film.md +19 -77
  65. package/skills/slates-project-organization.md +7 -3
  66. package/skills/slates-prompting-flux-2-max.md +15 -4
  67. package/skills/slates-prompting-gpt-image-2-5.md +41 -28
  68. package/skills/slates-prompting-kling-v3.md +3 -3
  69. package/skills/slates-prompting-lip-sync.md +1 -1
  70. package/skills/slates-prompting-minimax-h3.md +30 -17
  71. package/skills/slates-prompting-motion-transfer.md +1 -1
  72. package/skills/slates-prompting-nano-banana-2.md +24 -11
  73. package/skills/slates-prompting-seedance-2-5.md +7 -6
  74. package/skills/slates-prompting-seedance.md +5 -5
  75. package/skills/slates-prompting-seedream-5-lite.md +14 -3
  76. package/skills/slates-prompting-veo-3.md +1 -1
  77. package/skills/slates-script-craft.md +45 -0
  78. package/skills/slates-shot-variety.md +11 -40
  79. package/skills/slates-storyboard-from-script.md +14 -66
  80. package/skills/slates-style-prompting.md +54 -54
  81. package/skills/slates-ugc-influencer-ad.md +32 -309
  82. package/skills/slates-vision-feedback-loop.md +2 -1
@@ -5,12 +5,18 @@
5
5
  // per-model skills, so the TS consumers and the markdown consumers can
6
6
  // no longer disagree. Edit the partial, not this file, not the skills.
7
7
  export const PARTIALS = {
8
- "decision-log": "When you surface the plan, include a short **decision log** — one line per decision *you* made that the user did not specify **and that no row already records**:\n\n```\nsource phrase or declared default → what you wrote → what it resolves\n\"in a diner\" → warm, and the light is the reason → why the anchor was chosen, not what it is\n(no time of day) → late afternoon, low warm key → default; say the word and it changes\n```\n\n🚨 **Keep it to what is NOT already data — and almost everything now IS.** A Shot holds the references and their roles, the model, every param, the shot size, the camera, the prop, the action and the spoken line, and `slates_list_shots` reads the whole board back in order with its variety counts. Narrating any of those is retelling a row the user can open. **Write the Shot, and let the log carry only the judgement no field holds** — why this world, why this light, why this register.\n\n**Hard rule: never silently add weather, props, style, or camera movement.** Four of those are now FIELDS: put the value on the Shot (`prop`, `camera`, `shotSize`, `action`) so the user can read and change it, and put the *reason* in the log only when you invented it rather than being told it. The rule has not softened — it moved from narration into data, which is stronger, because a field can be corrected and a sentence in chat cannot.\n\n> ❌ **Do NOT turn this into a question gate.** Clarifying questions before optimizing directly fight the locked fast-path rule: *if intent is clear, generate immediately with sane defaults, don't ask questions; only ask for production intent, and batch every question into one message.* Log the decisions, then go. The log is an **output**, not an interrogation — surfaced alongside the plan, never as a separate ceremony, and never as a reason to wait.",
9
- "reference-rules-core": "Identity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates composes the naming for you from your `@mentions` / `#tags` — you never hand-write role labels.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.",
8
+ "cinematic-card": "**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->",
9
+ "cinematic-routes-short": "Two routes: describe a new frame, or change a frame you own. For a new scene, name references where you use them, write the look reference's grade and imperfections in plain words, then describe one light system, visible exposure, lens plus effect, every garment and a closed foreground. Use only what the shot needs. For your own plate, sheet, photo, footage or Blender render, say only what changes and what stays. Never use a released film frame as the edit base; use it as an art-direction brief for a new scene.",
10
+ "cinematic-tips-short": "Image models tend toward clean, evenly lit, fully exposed pictures. For a filmed look, describe what you see: one light source and its effect, a face almost in silhouette, a sky burned white. Name the lens and its visible effect together. Look at every reference first and describe its own darkness, contrast, colour, softness, noise and subject separation. Keep muddy blacks muddy; never clean up or brighten the reference's grade unless that is the change you want. Name every garment and exactly what is in the foreground. Use only what the shot needs; references are optional.",
11
+ "decision-log": "Record production choices in the editable shot fields. Explain only consequential judgments the user did not specify and no field already records: for example, why a particular light or performance register supports the brief. Do not repeat the shot list in prose or turn this explanation into an approval gate. Follow the separate generation authorization policy before spending.",
12
+ "image-defaults": "**Image default:** gpt-image-2-5-sunburst, quality `high`, 3k. User overrides take priority. Without a project, generation uses the headless Nano Banana 2 seat.\n\n| Model | Default resolution |\n|---|---|\n| nano-banana-2 | 2k |\n| nano-banana-2-lite | 1k |\n| nano-banana-pro | 2k |\n| gpt-image-2-5-flare | 2k |\n| gpt-image-2-5-sunburst | 3k |\n| flux-2-max | 1k |\n| seedream-5-lite | 2k |",
13
+ "lens-video-split": "Named lenses, apertures, film stocks and camera bodies (`85mm f/1.4`, `Kodak Portra 400`, `ARRI Alexa 65`) are an image-model lever. On a video model, translate the look instead of pasting the gear list: `85mm f/1.4, Portra 400` becomes `close-up, shallow depth of field, warm natural colors, cinematic texture, film-grain texture`. ByteDance's Seedance 2.0 guide never mentions fps, shutter angle, f-stop or lens millimetres. Its Seedance 2.5 guide does, once: the visual-style line of its own storyboard example names one camera body and one 35 mm cinema lens. On 2.5 a single line like that is vendor-sanctioned; a stacked gear list still is not.",
14
+ "reference-rules-core": "Identity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.",
10
15
  "reference-tips-short": "Name each reference inline; never write role essays. Slates does this for you: `@mention` a subject or environment and it composes `Marcus (image 1) in the cafe (image 2)`, citing them in the exact order it sends them. One canonical identity image avoids competing facial renderings; a \"Reference Image Instructions\" block drags reference lighting into your scene. Start with 2-3 focused refs.",
11
16
  "references-read-literally": "> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.",
12
17
  "seedance-25-timestamps": "**2.0 does not respond to timestamps and answers only to shot numbers. 2.5 responds to\ninteger-second timestamps.** That is ByteDance's own first line under \"Differences from Seedance\n2.0\", and it is why a 30-second take is usable at all: the length is only worth buying if you can\nsay *when* things happen inside it.\n\nBoth formats are valid on 2.5, and you can mix them — `Shot N` blocks for a storyboard whose\npacing you are happy to leave to the model, timestamps when a beat has to land at a moment.\n\n**Three ways to control time, all first-party:**\n\n| Form | Write it like |\n|---|---|\n| **Interval** | `0-3 seconds… 3-7 seconds… 7-15 seconds` or `[1s-4s]… [4s-8s]… [8s-12s]` |\n| **Time point** | *\"Quick left sideways transition at the 5-second mark.\"* |\n| **Relative** | *\"After 3 seconds, everyone around him shakes their head.\"* · *\"The frame freezes for 1 second after he presses the shutter.\"* |\n\n**The rules that come with them:**\n\n- **One second is the smallest unit.** Integers only — no `2.5s`, no frames.\n- **No gaps in the timeline.** `0-3s… 5-6s…` leaves 3-5s unspecified and the model fills it however\n it likes. Intervals must abut: `0-3s`, `3-7s`, `7-15s`.\n- **Budget the plot to the seconds.** Too little content in a range and the model improvises to\n fill it; too much and you get extra cuts or dropped beats. This is the actual craft of a 30s take.\n- **Never time-code a high-frequency action.** *\"Shake your head three times per second\"* is\n explicitly called out as a misuse — timestamps schedule beats, they don't choreograph frames.\n- **Transitions want both halves:** the moment AND the method — *\"At the 5-second mark, the camera\n transitions leftward with a left wipe into a natural dissolve.\"*\n- **Timestamps work on an EDIT too**, and that is where they earn the most: they scope a change in\n time as well as in content — *\"Change the man's action from drinking coffee to mopping the floor\n from 4-6 seconds in Video 1, and leave the rest of the content unchanged.\"* Without a range, a\n whole-clip instruction is applied to the whole clip.\n\nDo **not** carry this back to 2.0, and do not carry Veo's `[00:00-00:02]` bracket syntax into\neither — 2.0 ignores time entirely, and the cross-model syntax swap is its own known failure.",
13
18
  "seedance-25-timestamps-short": "Seedance 2.0 ignores timing and answers only to \"Shot 1 / Shot 2\"; 2.5 acts on whole-second timestamps, and that is what makes a 30-second take controllable rather than just long. Three forms work: intervals (\"0-3 seconds…3-7 seconds\"), a point (\"at the 5-second mark\"), or relative (\"after 3 seconds\"). Whole seconds only, no gaps between intervals, and never to choreograph fast repeated motion. They work on edits too, where a range scopes the change: \"…from 4-6 seconds…\".",
19
+ "sheet-tool-defaults": "**What the sheet tools render on** (you do not pick these; omit `model`):\n\n- **Character identity sheet:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n- **Establishing image:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n\nPrice a sheet for that model at 16:9, with resolution and quality left at their defaults. **Never 4K** — no identity gain at sheet scale, wasted spend.",
14
20
  "still-gate": "**A visible defect in the still is already a STOP.** Do not animate it. Fix the frame first, then move to motion — and go to motion only when the crop passes the still scan and you genuinely need movement to confirm an uncertain edge, reflection, or object.\n\nThis is a **cost** rule as much as a craft rule: a 1080p/10s premium video generation costs many multiples of an image re-roll, and video is where a defect stops being fixable. Anything wrong in the still gets worse in motion — soft geometry mushes, broken-but-plausible objects fall apart, oily textures start crawling. **Animating a known-bad frame is the single most expensive mistake in the pipeline.** Re-rolling the image is the cheap move; re-rolling the video is not.",
15
21
  "thresholds": "<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. Below it, announce the cost once and go.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.",
16
22
  };
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
16
16
  /** Footer callout paragraphs. */
17
17
  footer?: string[];
18
18
  }
19
- export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'ltx-2-5' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx' | 'inworld-tts-2';
19
+ export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'ltx-2-5' | 'nano-banana' | 'nano-banana-lite' | 'gpt-image-2-5' | 'seed-audio' | 'eleven-sfx' | 'inworld-tts-2';
20
20
  export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
21
21
  /** Null when no tips exist for the key — callers render an honest fallback. */
22
22
  export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
@@ -137,7 +137,7 @@ const SEEDANCE_25 = {
137
137
  [
138
138
  {
139
139
  heading: '720p is not the cheap one here',
140
- example: '30s \u00b7 720p \u00b7 Face route = 484 credits\n15s \u00b7 1080p \u00b7 Seedance 2.0 Face = 411 credits',
140
+ example: '30s \u00b7 720p \u00b7 Face route = 489 credits\n15s \u00b7 1080p \u00b7 Seedance 2.0 Face = 411 credits',
141
141
  note: 'Length is what moves the price, and 2.5 doubles the length ceiling — so a 30-second 720p clip can cost more than a 15-second 1080p one, against a 1,000-credit starting balance. Explore at short LENGTH rather than low resolution: cut the seconds to 4-8 while you are finding the shot, and stay at the resolution you actually want. A 480p pass does not de-risk a 720p render — generation is stochastic, so the 720p run is a different take, not the same shot rendered better. The Generate button always shows the exact number first.',
142
142
  critical: true,
143
143
  },
@@ -151,7 +151,7 @@ const SEEDANCE_25 = {
151
151
  ],
152
152
  footer: [
153
153
  '30 image references is a budget, not a target — 2-4 strong references still beat both extremes, one per role. ByteDance\'s own ceilings for 2.5: 1-8 subjects bound by image reference stay stable (9-12 works but needs re-rolls), 1-5 subjects bound by video or audio reference, and 5-10 seconds is the sweet spot for a reference clip. Unlike 2.0, a multi-view turnaround sheet can be a single subject reference here — past 5 subjects, go back to one view per image. The larger budget is for long multi-shot takes and for video plus audio references alongside images.',
154
- 'A reference VIDEO bills input seconds PLUS output seconds, and 2.5 accepts references up to 30s combined — so a 20-second reference driving a 20-second output bills 40 seconds. The Generate button shows the total.',
154
+ 'A reference VIDEO bills input seconds PLUS output seconds, and 2.5 accepts references up to 30s combined — so a 20-second reference driving a 20-second output bills 40 seconds. On the AI-face route the reference counts as at least as long as the output: a 5-second reference on a 20-second output bills 40 seconds, not 25. The Generate button shows the total.',
155
155
  ...(SEEDANCE.footer ?? []).slice(0, 2),
156
156
  'Frames and reference images stay mutually exclusive, and on a first/last-frame generation Seedance 2.5 chooses the aspect ratio itself — the ratio control shows "Adaptive" because the start frame decides the shape.',
157
157
  ],
@@ -190,7 +190,7 @@ const SEEDANCE_25_EDIT = {
190
190
  [
191
191
  {
192
192
  heading: 'When to use it instead of the others',
193
- note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner and the cheapest seat, and Kling O3 Edit is the one that takes subject and style reference images.',
193
+ note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner and the cheapest option, and Kling O3 Edit is the one that takes subject and style reference images.',
194
194
  },
195
195
  {
196
196
  heading: 'It edits the audio too',
@@ -347,7 +347,7 @@ const OMNI_FLASH = {
347
347
  note: 'No negative-prompt field — write what to avoid as a direct instruction.',
348
348
  },
349
349
  {
350
- heading: 'Know its seat',
350
+ heading: 'Know its role',
351
351
  note: 'Cheap drafts, iteration volume, and audio-in-one-gen at low cost. For hero shots, Seedance 2.5 (the default) or Seedance 2.0 (4K, cheaper) still win.',
352
352
  },
353
353
  ],
@@ -398,6 +398,45 @@ const OMNI_FLASH_EDIT = {
398
398
  'Ship via segment-splice: edit only the seconds where the change happens (Trim / Split first), then splice back over the original on the timeline with the original audio underneath. Chain edits one change at a time — each edit saves as a new clip linked to its parent.',
399
399
  ],
400
400
  };
401
+ const GPT_IMAGE_25 = {
402
+ label: 'GPT Image 2.5',
403
+ intro: [
404
+ 'Two tiers at the same price: Flare is the faster one; Sunburst is the better-quality one, and slower. Explore on Flare, finish on Sunburst.',
405
+ 'The photoreal front-runner for people, and the most reliable model for readable text, ordered panels and exact placement.',
406
+ ],
407
+ columns: [
408
+ [
409
+ {
410
+ heading: 'Name each reference where it is used',
411
+ example: 'The woman from image 1 cooks on a rocky summit, lit and graded like image 2.',
412
+ note: PARTIALS['reference-tips-short'],
413
+ },
414
+ {
415
+ heading: 'Quote text that must render',
416
+ example: 'the word "SLATES" once in small plain letters on the left chest',
417
+ note: "Quoted strings render most reliably. Describe a font's feel, never its name, and keep on-image text under about 30 words.",
418
+ },
419
+ {
420
+ heading: 'Set the quality tier on purpose',
421
+ example: 'low · medium · high · xhigh · max',
422
+ note: 'High is the everyday tier and max costs about four times as much. Medium is cheap enough to draft on; go past high only when tiny type or a finished frame needs it.',
423
+ },
424
+ ],
425
+ [
426
+ {
427
+ heading: 'The look: describe what the camera sees',
428
+ example: 'She is close to a silhouette: her face falls into deep shadow.\nThe sky around the sun burns out to white.',
429
+ note: PARTIALS['cinematic-tips-short'],
430
+ critical: true,
431
+ },
432
+ {
433
+ heading: 'Describe the frame, or swap into one you own',
434
+ example: 'Dark hiking trousers. The only things on the rock are the stove and the pan.\nTake image 1 and change only the character to the character in image 2.',
435
+ note: PARTIALS['cinematic-routes-short'],
436
+ },
437
+ ],
438
+ ],
439
+ };
401
440
  const NANO_BANANA = {
402
441
  label: 'Nano Banana 2',
403
442
  intro: [
@@ -414,7 +453,7 @@ const NANO_BANANA = {
414
453
  {
415
454
  heading: 'Named lenses + apertures',
416
455
  example: '85mm f/1.4 · 135mm f/2.8 · 50mm f/1.2 · 35mm f/2 · Panavision anamorphic · 400mm telephoto',
417
- note: '135mm f/2.8 is the cheat code for skin texture and intimate compression. Anamorphic for cinematic width + horizontal flares.',
456
+ note: 'Describe the intended perspective, depth of field and texture alongside the lens. A model does not guarantee physical lens simulation.',
418
457
  },
419
458
  {
420
459
  heading: 'Named film stocks (one per prompt)',
@@ -424,7 +463,7 @@ const NANO_BANANA = {
424
463
  {
425
464
  heading: "Don't carry lens + stock into a video prompt",
426
465
  example: '85mm f/1.4, Portra 400\n→ close-up, shallow depth of field, warm natural colors, cinematic texture',
427
- note: 'Lenses, apertures, film stocks and camera bodies are an image-model lever and a video-model anti-pattern — ByteDance\'s Seedance guide never mentions f-stops, lens millimetres, fps or shutter angle. When you animate a frame you made here, translate the look into shot size, depth of field and colour tone instead of pasting the gear list across.',
466
+ note: PARTIALS['lens-video-split'],
428
467
  },
429
468
  {
430
469
  heading: 'Physics-based lighting',
@@ -434,14 +473,19 @@ const NANO_BANANA = {
434
473
  {
435
474
  heading: 'Imperfection vocabulary',
436
475
  example: 'visible pores · peach fuzz · ISO noise · sweat beading · slight hyperpigmentation · unretouched raw photography',
437
- note: 'Forces the model away from AI-clean skin. The default is too smooth — you have to ask for the imperfections that real photos have.',
476
+ note: 'Forces the model away from AI-clean skin. The default is too smooth — you have to ask for the imperfections that real photos have. Lead with the kind of photograph and the conditions on the skin; a bare list of flaw words reads as tokens.',
477
+ },
478
+ {
479
+ heading: 'The look: describe what the camera sees',
480
+ example: 'She is close to a silhouette: her face falls into deep shadow.\nThe sky around the sun burns out to white.',
481
+ note: PARTIALS['cinematic-tips-short'],
438
482
  },
439
483
  ],
440
484
  [
441
485
  {
442
486
  heading: '❌ The anti-list — avoid these',
443
487
  example: '8k · masterpiece · hyperrealistic · ultra-detailed · trending on ArtStation · perfect skin · flawless · airbrushed · cinematic (alone)',
444
- note: 'Tag-soup phrases from the Stable-Diffusion era. Measured success ~60-70% with these vs ~95%+ with positive description. Always specify which cinema — director, lens, era, stock.',
488
+ note: 'Generic quality tags do not specify an observable result. Describe the medium, light, exposure and texture the brief calls for.',
445
489
  critical: true,
446
490
  },
447
491
  {
@@ -537,7 +581,7 @@ const SEED_AUDIO = {
537
581
  note: 'Up to 3 audio clips (max 30s each), referenced as @Audio1–@Audio3 — OR one image to score what is in frame. Never both in the same generation.',
538
582
  },
539
583
  {
540
- heading: 'Know its seat',
584
+ heading: 'Know its role',
541
585
  note: 'Scenes, beds, room tone and dialogue in one pass. For a single effect that has to land on a specific frame, use Sound Effects.',
542
586
  },
543
587
  ],
@@ -579,7 +623,7 @@ const ELEVEN_SFX = {
579
623
  note: 'Higher hugs your wording with less variation between takes; lower explores. Raise it when a re-roll keeps wandering off the brief.',
580
624
  },
581
625
  {
582
- heading: 'Know its seat',
626
+ heading: 'Know its role',
583
627
  note: 'One precise effect on a known frame. Full rooms and layered scenes are cheaper and better in one Seed Audio pass.',
584
628
  },
585
629
  ],
@@ -589,7 +633,7 @@ const MINIMAX_H3 = {
589
633
  label: 'MiniMax H3',
590
634
  intro: [
591
635
  'MiniMax H3 generates picture and sound in one pass — 24fps, 32kHz stereo, 5-15 seconds, 11 stably-supported languages. It is the only video model in Slates where audio is AUTHORED rather than switched on: synchronised dialogue and action sounds go in the body of the prompt, ambience goes in a soundscape section, and audience-only music goes in a score section. Put a sound in the wrong section and it is dropped, doubled, or attributed to the wrong source.',
592
- 'Two seats that differ in LADDER and PRICE, not in what they accept. Base H3 runs 480p / 768p / 2K / 4K; H3 Max is fal\'s faster post-train and runs 480p / 768p / 1080p, dearer than base H3 at the tier they share - a deliberate speed pick, never the cheap one. BOTH read up to 9 reference images plus 3 video and 3 audio clips (12 files total, and audio never travels alone), and both animate a start frame and an end frame. 768p is the default on both because it is the tier the model natively generates; base H3\'s 2K and 4K are upscales of a 768p base. Reference images past the free allowance are billed and the allowances DIFFER: 5 free on base H3, 4 on Max.',
636
+ 'Three models. Base H3 runs 480p / 768p / 2K / 4K; H3 Max is fal\'s faster post-train and runs 480p / 768p / 1080p, dearer than base H3 at the tier they share - a deliberate speed pick, never the cheap one; H3 Max Turbo has Max\'s ladder at half Max\'s rate and takes NO references. Base H3 and Max read up to 9 reference images plus 3 video and 3 audio clips (12 files total, and audio never travels alone); all three animate a start frame and an end frame. 768p is the default on all three because it is the tier the model natively generates; base H3\'s 2K and 4K are upscales of a 768p base, and 1080p on Max and Turbo is a refinement of it. Reference images past the free allowance are billed and the allowances DIFFER: 5 free on base H3, 4 on Max.',
593
637
  ],
594
638
  columns: [
595
639
  [
@@ -624,7 +668,7 @@ const MINIMAX_H3 = {
624
668
  {
625
669
  heading: 'Say how much of a reference survives',
626
670
  example: 'Give the man in image 3 the weathered leather texture of the jacket in image 4.',
627
- note: 'H3 is the only seat that understands transferring a characteristic onto a DIFFERENT subject. State each reference\'s job and how much of it should carry through — kept whole, kept in part, transferred, or a loose echo.',
671
+ note: 'H3 is the only model that understands transferring a characteristic onto a DIFFERENT subject. State each reference\'s job and how much of it should carry through — kept whole, kept in part, transferred, or a loose echo.',
628
672
  },
629
673
  {
630
674
  heading: 'Reference images past the fifth cost extra',
@@ -650,7 +694,7 @@ const LTX_2_5 = {
650
694
  label: 'LTX-2.5',
651
695
  intro: [
652
696
  'LTX-2.5 scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE, not the last. Lightricks ranks the six parts of a prompt in this order: sound, camera, character detail, shot type and scene, then scene dressing — and scene dressing is the first thing to cut when a prompt sprawls. Everything goes in ONE flowing paragraph, not a list of labelled sections.',
653
- 'Two seats. Base LTX-2.5 is the distilled build: 720p / 1080p / 1440p / 4K and clips from 6 to 20 seconds, and it is the cheapest native 1080p second in Slates. LTX-2.5 Pro is the full diffusion build ("Diffusion Fidelity Rendering" spends extra compute on busy frames) but reaches a SHORTER ladder — 1080p and 10 seconds maximum — while costing about a third more. Pro is for a dense final render; base is for iteration, long takes and 4K.',
697
+ 'Two models. Base LTX-2.5 is the distilled build: 720p / 1080p / 1440p / 4K and clips from 6 to 20 seconds, and it is the cheapest native 1080p second in Slates. LTX-2.5 Pro is the full diffusion build ("Diffusion Fidelity Rendering" spends extra compute on busy frames) but reaches a SHORTER ladder — 1080p and 10 seconds maximum — while costing about a third more. Pro is for a dense final render; base is for iteration, long takes and 4K.',
654
698
  ],
655
699
  columns: [
656
700
  [
@@ -706,14 +750,14 @@ const LTX_2_5 = {
706
750
  ],
707
751
  footer: [
708
752
  'Frames, not references. LTX takes a start frame and an optional end frame (which generates a transition between the two) — it has no reference endpoint at all, so identity, style and environment reference images are not available on this model. For character consistency across separate shots, use MiniMax H3 or Kling.',
709
- 'Aspect ratios are 16:9 and 9:16 only, and native audio is included free at every resolution — there is no sound surcharge on either seat.',
753
+ 'Aspect ratios are 16:9 and 9:16 only, and native audio is included free at every resolution — there is no sound surcharge on either model.',
710
754
  'In image-to-video, do not cut away from the opening frame too early: you have paid for that frame, so let it play before the first move.',
711
755
  ],
712
756
  };
713
757
  const INWORLD_TTS = {
714
758
  label: 'Inworld TTS-2',
715
759
  intro: [
716
- 'The voice seat: one named voice saying one line. Unlike every other surface in Slates, the prompt is not a description of what you want — it IS the words that get spoken, verbatim, and its length is what you are billed for.',
760
+ 'The voice model: one named voice saying one line. Unlike every other surface in Slates, the prompt is not a description of what you want — it IS the words that get spoken, verbatim, and its length is what you are billed for.',
717
761
  'A voice belongs to a character, the same way a face does. Build it once from a clip or a description, then send it lines.',
718
762
  ],
719
763
  columns: [
@@ -761,7 +805,7 @@ const INWORLD_TTS = {
761
805
  note: 'Numbers, dates and abbreviations are read literally. Write them as they should sound.',
762
806
  },
763
807
  {
764
- heading: 'Know its seat',
808
+ heading: 'Know its role',
765
809
  note: 'One voice, cleanly. Dialogue mixed with effects and room tone in one pass is Seed Audio; a single non-speech sound is Sound Effects.',
766
810
  },
767
811
  ],
@@ -780,6 +824,7 @@ export const PROMPTING_TIPS = {
780
824
  'ltx-2-5': LTX_2_5,
781
825
  'nano-banana': NANO_BANANA,
782
826
  'nano-banana-lite': NANO_BANANA_LITE,
827
+ 'gpt-image-2-5': GPT_IMAGE_25,
783
828
  'seed-audio': SEED_AUDIO,
784
829
  'eleven-sfx': ELEVEN_SFX,
785
830
  'inworld-tts-2': INWORLD_TTS,
@@ -196,4 +196,6 @@ export declare const KLING_EDIT_MAX_REFS = 4;
196
196
  * trimmed first — subjects are the feature)
197
197
  */
198
198
  export declare function composeKlingEdit(rawPrompt: string, groups: ReferenceGroup[]): KlingEditComposition;
199
+ /** No-reference paths use the same composer and preserve authored prose. */
200
+ export declare function cleanPrompt(userPrompt: string): string;
199
201
  //# sourceMappingURL=reference-composer.d.ts.map
@@ -19,11 +19,12 @@
19
19
  // is each model's own official consistency lever (NB2 "assign a
20
20
  // distinct name", Seedance "Reference Subject_N in Image_N", Kling "reuse a fixed
21
21
  // label verbatim"); the heavy role-essay block was the off-doctrine part.
22
- // Normalize a name/token for matching: drop the sigil, lowercase, strip
23
- // spaces/underscores/hyphens. "@big_red" / "@Big Red" / "#Big-Red" all collapse
24
- // to the same key. Identical to the agent-side resolver's `norm`.
22
+ // A mention's matching key: lowercase, spaces/underscores/hyphens stripped, and
23
+ // the SIGIL KEPT. "@big_red" / "@Big Red" / "@Big-Red" are one key, and "@red"
24
+ // and "#red" are two: names are unique per sigil, so a subject and a look may
25
+ // share one, and a sigil-free key bound both mentions to whichever came last.
25
26
  function normToken(s) {
26
- return s.toLowerCase().replace(/[@#]/g, '').replace(/[\s_-]+/g, '');
27
+ return s.toLowerCase().replace(/[\s_-]+/g, '');
27
28
  }
28
29
  // Free-reference IMAGE kinds get an "image N" number. Frames are transported in
29
30
  // their own dedicated slots (start/last frame) by the per-model adapter and are
@@ -179,7 +180,7 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
179
180
  // ── 2. Inline-name token groups in the prompt body ──
180
181
  // For each character/environment group whose token appears in the prompt, the
181
182
  // FIRST occurrence becomes "Name (image N)"; later ones become just "Name".
182
- // Style tokens are removed (a single trailing clause carries the style). Token
183
+ // Style tokens become image citations at the user's chosen binding site. Token
183
184
  // groups NOT found in the prompt fall through to a key line in step 3.
184
185
  const tokenGroups = numbered.filter((g) => g.token && (g.kind === 'character' || g.kind === 'environment' || g.kind === 'style'));
185
186
  const byNorm = new Map();
@@ -204,20 +205,7 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
204
205
  unresolvedSeen.add(key);
205
206
  unresolvedTokens.push(`${sigil}${tok}`);
206
207
  };
207
- // First strip "in/with the style of #tag" phrases so the style reads as a
208
- // clean trailing clause, not a dangling preposition (legacy cleanPrompt
209
- // behaviour). ONLY when the tag resolves: an unresolved one leaves the whole
210
- // phrase exactly as authored and falls through to the token pass below, which
211
- // reports it and sends it as written.
212
- let body = rawPrompt.replace(/\s+(with|in)\s+the\s+style\s+of\s+([@#])([\w-]+(?:\.[\w-]+)*)/gi, (_full, _prep, sigil, tok) => {
213
- const g = byNorm.get(normToken(`${sigil}${tok}`));
214
- if (g && g.kind === 'style') {
215
- matchedInPrompt.add(normToken(`${sigil}${tok}`));
216
- return '';
217
- }
218
- return _full;
219
- });
220
- body = body.replace(TOKEN_RE, (_full, _sigil, tok) => {
208
+ const body = rawPrompt.replace(TOKEN_RE, (_full, _sigil, tok) => {
221
209
  const key = normToken(`${_sigil}${tok}`);
222
210
  const g = byNorm.get(key);
223
211
  if (!g) {
@@ -227,7 +215,7 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
227
215
  }
228
216
  matchedInPrompt.add(key);
229
217
  if (g.kind === 'style')
230
- return ''; // styles never inline — trailing clause only
218
+ return g.imageNums.length ? citeImages(g.imageNums) : g.name;
231
219
  if (!seenFirst.has(key)) {
232
220
  seenFirst.add(key);
233
221
  // 🚨 ONE BINDING SITE PER ENTITY, CARRYING EVERY MEDIUM SHE OWNS
@@ -257,8 +245,13 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
257
245
  }
258
246
  return g.name;
259
247
  });
260
- // Collapse the whitespace the token removals left behind.
261
- body = body.replace(/[ \t]{2,}/g, ' ').replace(/\s+([,.;:!?])/g, '$1').trim();
248
+ // Literal citations are authored bindings too. Do not add a competing role
249
+ // sentence when the user already describes what that image supplies.
250
+ const citedImages = new Set();
251
+ for (const match of body.matchAll(/\bimages?\s+(\d+(?:\s*(?:,\s*(?:(?:and|&)\s*)?|(?:and|&)\s*)\d+)*)\b/gi)) {
252
+ for (const n of match[1].match(/\d+/g) ?? [])
253
+ citedImages.add(Number(n));
254
+ }
262
255
  // ── 3. Build the key lines for token-less / unmatched-token groups ──
263
256
  // Video sources, pinned/base canvases, and picked subjects that have no token
264
257
  // in the prompt each get ONE short neutral key line (never an essay). The user's
@@ -314,10 +307,11 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
314
307
  for (const g of numbered) {
315
308
  if ((g.kind === 'character' || g.kind === 'environment') && g.imageNums.length > 0) {
316
309
  const tokenWasMatched = g.token && matchedInPrompt.has(normToken(g.token));
317
- if (!tokenWasMatched) {
318
- const noun = g.imageNums.length === 1 ? 'Image' : 'Images';
319
- const verb = g.imageNums.length === 1 ? 'is' : 'are';
320
- topKeys.push(`${noun} ${joinNums(g.imageNums)} ${verb} ${g.name}.`);
310
+ const unmentioned = g.imageNums.filter((n) => !citedImages.has(n));
311
+ if (!tokenWasMatched && unmentioned.length) {
312
+ const noun = unmentioned.length === 1 ? 'Image' : 'Images';
313
+ const verb = unmentioned.length === 1 ? 'is' : 'are';
314
+ topKeys.push(`${noun} ${joinNums(unmentioned)} ${verb} ${g.name}.`);
321
315
  }
322
316
  }
323
317
  }
@@ -419,11 +413,11 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
419
413
  // `Audio 1` sitting mid-sentence among lowercase `image 1`s; moving the
420
414
  // binding inline removed the reason for the exception along with the
421
415
  // exception.
422
- // ── 4. Style trailing clause (one, at the end — style reads best last) ──
416
+ // ── 4. Fallback for style attachments the user has not cited ──
423
417
  const styleNums = [];
424
418
  for (const g of numbered) {
425
419
  if (g.kind === 'style')
426
- styleNums.push(...g.imageNums);
420
+ styleNums.push(...g.imageNums.filter((n) => !citedImages.has(n)));
427
421
  }
428
422
  const styleClauses = [];
429
423
  if (styleNums.length > 0) {
@@ -475,10 +469,12 @@ export function composeVoiceCitations(rawPrompt, voices) {
475
469
  if (v?.token)
476
470
  byNorm.set(normToken(v.token), { n: i + 1, name: v.name });
477
471
  });
472
+ // The one mention grammar (TOKEN_RE), so `joe@sarah.com` and `@sarah.extra`
473
+ // stay prose. A `#` token is a look, never a voice.
478
474
  const seen = new Set();
479
- const body = rawPrompt.replace(/@([\w-]+)/g, (full, tok) => {
480
- const key = normToken(`@${tok}`);
481
- const v = byNorm.get(key);
475
+ const body = rawPrompt.replace(TOKEN_RE, (full, sigil, tok) => {
476
+ const key = normToken(`${sigil}${tok}`);
477
+ const v = sigil === '@' ? byNorm.get(key) : undefined;
482
478
  if (!v)
483
479
  return full;
484
480
  if (seen.has(key))
@@ -525,7 +521,12 @@ export function composeKlingEdit(rawPrompt, groups) {
525
521
  const elements = [];
526
522
  const styleImages = [];
527
523
  const styleNums = [];
528
- let body = rawPrompt;
524
+ // Each mention's citation, by key. The prompt is walked ONCE with the one
525
+ // mention grammar (TOKEN_RE), so `joe@marcus.com` and `@marcus.extra` stay
526
+ // prose, as they do in `composeReferences`.
527
+ const citations = new Map();
528
+ const inPrompt = new Set([...rawPrompt.matchAll(TOKEN_RE)].map((t) => normToken(`${t[1]}${t[2]}`)));
529
+ const keyLines = [];
529
530
  // Subjects first — they own the @ElementN numbering.
530
531
  const subjectGroups = groups.filter((g) => (g.kind === 'character' || g.kind === 'environment') && g.media.some((m) => m.mediaKind === 'image'));
531
532
  for (const g of subjectGroups) {
@@ -536,20 +537,13 @@ export function composeKlingEdit(rawPrompt, groups) {
536
537
  continue;
537
538
  const n = elements.length + 1;
538
539
  elements.push({ frontal: imgs[0], angles: imgs.slice(1, 4), name: g.name });
539
- if (g.token) {
540
- // Replace every @token occurrence with the element citation.
541
- const escaped = g.token.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
542
- const re = new RegExp(`${escaped}\\b`, 'gi');
543
- if (re.test(body)) {
544
- body = body.replace(re, `@Element${n}`);
545
- }
546
- else {
547
- body = `${body}\n@Element${n} is ${g.name}.`;
548
- }
549
- }
550
- else {
551
- body = `${body}\n@Element${n} is ${g.name}.`;
552
- }
540
+ // Every mention of the subject becomes its element citation; a subject the
541
+ // prompt never names gets a key line.
542
+ const key = g.token ? normToken(g.token) : null;
543
+ if (key && inPrompt.has(key) && !citations.has(key))
544
+ citations.set(key, `@Element${n}`);
545
+ else
546
+ keyLines.push(`@Element${n} is ${g.name}.`);
553
547
  }
554
548
  // Style / pinned refs take the remaining slots as @ImageN.
555
549
  for (const g of groups) {
@@ -564,12 +558,15 @@ export function composeKlingEdit(rawPrompt, groups) {
564
558
  const n = styleImages.length;
565
559
  if (g.kind === 'style')
566
560
  styleNums.push(n);
567
- if (g.token) {
568
- const escaped = g.token.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
569
- body = body.replace(new RegExp(`${escaped}\\b`, 'gi'), `@Image${n}`);
570
- }
561
+ // A group's mention cites its FIRST image.
562
+ const key = g.token ? normToken(g.token) : null;
563
+ if (key && !citations.has(key))
564
+ citations.set(key, `@Image${n}`);
571
565
  }
572
566
  }
567
+ let body = rawPrompt.replace(TOKEN_RE, (full, sigil, tok) => citations.get(normToken(`${sigil}${tok}`)) ?? full);
568
+ for (const line of keyLines)
569
+ body = `${body}\n${line}`;
573
570
  if (styleNums.length > 0) {
574
571
  const cites = styleNums.map((n) => `@Image${n}`).join(' and ');
575
572
  body = `${body}\nApply the visual style of ${cites}.`;
@@ -584,4 +581,8 @@ export function composeKlingEdit(rawPrompt, groups) {
584
581
  styleImages,
585
582
  };
586
583
  }
584
+ /** No-reference paths use the same composer and preserve authored prose. */
585
+ export function cleanPrompt(userPrompt) {
586
+ return composeReferences(userPrompt, []).prompt;
587
+ }
587
588
  //# sourceMappingURL=reference-composer.js.map