@slatesvideo/shared 0.5.6 → 0.5.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,8 +6,50 @@ export interface ModelFact {
6
6
  maxRefImages: number | null;
7
7
  /** Max ingredient images (video models) — null if not applicable. */
8
8
  maxIngredients: number | null;
9
+ /** Reference VIDEOS accepted in one generation. null/absent = none. */
10
+ maxReferenceVideos?: number | null;
11
+ /** Reference AUDIO clips accepted in one generation. null/absent = none. */
12
+ maxReferenceAudio?: number | null;
13
+ /** Combined seconds across every reference video. */
14
+ maxReferenceVideoSeconds?: number | null;
15
+ /** Combined seconds across every reference audio clip. */
16
+ maxReferenceAudioSeconds?: number | null;
17
+ /** Ceiling on TOTAL reference files across all modalities. */
18
+ maxReferenceFilesTotal?: number | null;
19
+ /** An audio reference needs at least one image or video reference alongside. */
20
+ audioRefNeedsCompanion?: boolean;
9
21
  notes: string;
10
22
  }
23
+ /**
24
+ * One sentence of multimodal-reference capacity for a model, derived. Returns
25
+ * an empty string for a model that takes none, so a caller can append it
26
+ * unconditionally.
27
+ */
28
+ export declare function multimodalRefSummary(id: string): string;
29
+ /**
30
+ * The prompt words that make Seedance 2.5 reclassify a reference-carrying
31
+ * request as a video EDIT or EXTEND — after which it fails on task-type
32
+ * constraints it never set, ASYNCHRONOUSLY, once the job has queued.
33
+ *
34
+ * 🚨 THIS DRIVES A WARNING THAT NAMES THE WORDS. It must never drive a rewrite:
35
+ * silently mutating the user's prompt to dodge a provider classifier is banned
36
+ * by the prompt-transparency invariant (slate/CLAUDE.md). "Remove the tripod" is
37
+ * the user's sentence; the honest move is to say what will happen.
38
+ *
39
+ * ⚠️ MIRRORED, and the mirror is deliberate. The desktop's copy is
40
+ * `SEEDANCE_EDIT_INTENT_KEYWORDS` + `SEEDANCE_EXTEND_INTENT_KEYWORDS` in
41
+ * `slate/src/shared/pricing.ts`, which cannot import from this package (it is
42
+ * loaded by the renderer through the `@shared/*` alias, with no npm dependency).
43
+ * Same situation as `promptComposition.ts` ↔ `reference-composer.ts`. Change one,
44
+ * change the other in the same pass. Both sides match on a WORD BOUNDARY, so
45
+ * "added" and "readdress" are not hits.
46
+ */
47
+ export declare const SEEDANCE_TASK_INTENT_WORDS: readonly ["add", "insert", "remove", "delete", "modify", "replace", "change", "edit the video", "extend", "continue", "continue the story"];
48
+ /** Which trigger words a prompt actually contains, so a warning can name them.
49
+ * Mirrors `seedanceTaskIntentWords()` in slate/src/shared/pricing.ts. */
50
+ export declare function seedanceTaskIntentWords(prompt: string): string[];
51
+ /** Every model that reads reference video and/or audio, for op descriptions. */
52
+ export declare function multimodalRefModels(): string[];
11
53
  export declare const MODEL_FACTS: ModelFact[];
12
54
  export declare function getModelFact(id: string): ModelFact | undefined;
13
55
  /** The official NB2 / general image prompt formula (subject-first). */
@@ -3,6 +3,74 @@
3
3
  // RUNTIME source of truth for limits is slate/src/shared/pricing.ts
4
4
  // (MODEL_REGISTRY.maxRefImages / maxIngredientImages); these mirror it for
5
5
  // documentation. Code-verified 2026-06-25.
6
+ //
7
+ // Prose that ALSO appears in a skill or the tips card comes from
8
+ // skills/_partials/*.md via PARTIALS — never restated here. A `notes` string is
9
+ // a third rendering of a fact, and a third rendering is a third thing that can
10
+ // survive a doctrine reversal the other two got.
11
+ import { PARTIALS } from './partials.generated.js';
12
+ /**
13
+ * One sentence of multimodal-reference capacity for a model, derived. Returns
14
+ * an empty string for a model that takes none, so a caller can append it
15
+ * unconditionally.
16
+ */
17
+ export function multimodalRefSummary(id) {
18
+ const f = MODEL_FACTS.find((m) => m.id === id);
19
+ if (!f)
20
+ return '';
21
+ const v = f.maxReferenceVideos ?? 0;
22
+ const a = f.maxReferenceAudio ?? 0;
23
+ if (v === 0 && a === 0)
24
+ return '';
25
+ const parts = [];
26
+ if (v > 0)
27
+ parts.push(`${v} reference video${v === 1 ? '' : 's'} (${f.maxReferenceVideoSeconds}s combined)`);
28
+ if (a > 0)
29
+ parts.push(`${a} reference audio clip${a === 1 ? '' : 's'} (${f.maxReferenceAudioSeconds}s combined)`);
30
+ const total = f.maxReferenceFilesTotal ? `, ${f.maxReferenceFilesTotal} files max across all modalities` : '';
31
+ const companion = f.audioRefNeedsCompanion
32
+ ? ' Audio needs at least one image or video reference alongside it.'
33
+ : ' Audio-only references are allowed.';
34
+ return `${f.label}: up to ${parts.join(' and ')}${total}.${companion}`;
35
+ }
36
+ /**
37
+ * The prompt words that make Seedance 2.5 reclassify a reference-carrying
38
+ * request as a video EDIT or EXTEND — after which it fails on task-type
39
+ * constraints it never set, ASYNCHRONOUSLY, once the job has queued.
40
+ *
41
+ * 🚨 THIS DRIVES A WARNING THAT NAMES THE WORDS. It must never drive a rewrite:
42
+ * silently mutating the user's prompt to dodge a provider classifier is banned
43
+ * by the prompt-transparency invariant (slate/CLAUDE.md). "Remove the tripod" is
44
+ * the user's sentence; the honest move is to say what will happen.
45
+ *
46
+ * ⚠️ MIRRORED, and the mirror is deliberate. The desktop's copy is
47
+ * `SEEDANCE_EDIT_INTENT_KEYWORDS` + `SEEDANCE_EXTEND_INTENT_KEYWORDS` in
48
+ * `slate/src/shared/pricing.ts`, which cannot import from this package (it is
49
+ * loaded by the renderer through the `@shared/*` alias, with no npm dependency).
50
+ * Same situation as `promptComposition.ts` ↔ `reference-composer.ts`. Change one,
51
+ * change the other in the same pass. Both sides match on a WORD BOUNDARY, so
52
+ * "added" and "readdress" are not hits.
53
+ */
54
+ export const SEEDANCE_TASK_INTENT_WORDS = [
55
+ // The edit half and the extension half of ByteDance's own trigger lists
56
+ // (Seedance 2.5 prompt guide → "Trigger keywords in prompt"). 'insert',
57
+ // 'delete' and 'modify' are named there and were missing here; 'change to',
58
+ // 'extend forward/backward', 'continue from' and 'extend the story' are all
59
+ // caught by the shorter word already in the list.
60
+ 'add', 'insert', 'remove', 'delete', 'modify', 'replace', 'change', 'edit the video',
61
+ 'extend', 'continue', 'continue the story',
62
+ ];
63
+ /** Which trigger words a prompt actually contains, so a warning can name them.
64
+ * Mirrors `seedanceTaskIntentWords()` in slate/src/shared/pricing.ts. */
65
+ export function seedanceTaskIntentWords(prompt) {
66
+ return SEEDANCE_TASK_INTENT_WORDS.filter((w) => new RegExp(`\\b${w.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'i').test(prompt));
67
+ }
68
+ /** Every model that reads reference video and/or audio, for op descriptions. */
69
+ export function multimodalRefModels() {
70
+ return MODEL_FACTS
71
+ .filter((m) => (m.maxReferenceVideos ?? 0) > 0 || (m.maxReferenceAudio ?? 0) > 0)
72
+ .map((m) => m.id);
73
+ }
6
74
  export const MODEL_FACTS = [
7
75
  {
8
76
  id: 'nano-banana-2',
@@ -61,7 +129,36 @@ export const MODEL_FACTS = [
61
129
  kind: 'video',
62
130
  maxRefImages: null,
63
131
  maxIngredients: 9, // ingredient images per video gen
64
- notes: 'PREMIUM video tier — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). Up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Also the PREMIUM engine inside the Motion Transfer and Lip Sync tools (single-pass: driving video / dialogue are native conditioning signals — better motion fidelity, natural speech, voice cloned from a video source; video references bill input+output seconds).',
132
+ maxReferenceVideos: 3,
133
+ maxReferenceAudio: 3,
134
+ maxReferenceVideoSeconds: 15,
135
+ maxReferenceAudioSeconds: 15,
136
+ maxReferenceFilesTotal: 12,
137
+ audioRefNeedsCompanion: true,
138
+ notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds. 2.0 STAYS THE DEFAULT over 2.5 because it is the only Seedance with 1080p and 4K.',
139
+ },
140
+ {
141
+ id: 'seedance-2.5',
142
+ label: 'Seedance 2.5',
143
+ kind: 'video',
144
+ maxRefImages: null,
145
+ maxIngredients: 30, // 30 image refs; the model also takes 10 video + 10 audio (50 total)
146
+ maxReferenceVideos: 10,
147
+ maxReferenceAudio: 10,
148
+ maxReferenceVideoSeconds: 30,
149
+ maxReferenceAudioSeconds: 30,
150
+ maxReferenceFilesTotal: 50,
151
+ // No companion requirement — audio-only references are one of the things
152
+ // the second seat actually buys.
153
+ notes: `A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is 480p/720p ONLY. No 1080p, no 4K, on any provider. Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), TIMED BEATS, or tighter prompt adherence. Pick 2.0 when resolution matters at all. VIDEO-ONLY. TIMESTAMPS: ${PARTIALS['seedance-25-timestamps-short']} Multi-view subject reference images are also supported on 2.5 (up to 5 subjects) where 2.0 wanted one view per subject. 🚨 COST DISCIPLINE: 720p STOPS READING AS "THE CHEAP ONE" HERE. A 30s 720p clip on the real-face route is 710 credits and on the AI-face route 484 — more than a 15s 1080p Seedance 2.0 face generation (411), against a 1,000-credit welcome grant. Always quote with slates_estimate_generation_cost before a long take, and draft at 480p/4-8s. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.`,
154
+ },
155
+ {
156
+ id: 'seedance-2.5-edit',
157
+ label: 'Seedance 2.5 Edit',
158
+ kind: 'video',
159
+ maxRefImages: null,
160
+ maxIngredients: 0, // prompt + source clip only on slates_edit_video
161
+ notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s), though ByteDance recommends staying inside 20s for quality. That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p output, native audio. Prompt + source clip only on this op — no reference images (the MODEL takes 1-5 reference images on an edit; Slates has not wired that path). Phrase the change as "from A to B", and TIMESTAMP a partial edit ("…from 4-6 seconds…") — 2.5 reads whole-second timestamps on edits, and without a range the instruction applies to the whole clip. AUDIO is editable on this same row: change a line, change an accent, translate dialogue with re-fitted lips, strip or replace BGM and sound effects. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
65
162
  },
66
163
  {
67
164
  id: 'kling-v3',
@@ -69,7 +166,7 @@ export const MODEL_FACTS = [
69
166
  kind: 'video',
70
167
  maxRefImages: null,
71
168
  maxIngredients: 4,
72
- notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. In the Motion Transfer / Lip Sync tools, Kling (MC / lip-sync / avatar) is the cheap utility lane; Seedance is the premium single-pass lane.',
169
+ notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools (MC std/pro, lip-sync, avatar) — those two tools are Kling-only.',
73
170
  },
74
171
  {
75
172
  id: 'kling-v3-edit',
@@ -111,14 +208,6 @@ export const MODEL_FACTS = [
111
208
  maxIngredients: null,
112
209
  notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
113
210
  },
114
- {
115
- id: 'eleven-v3',
116
- label: 'ElevenLabs Eleven v3 (TTS)',
117
- kind: 'audio',
118
- maxRefImages: null,
119
- maxIngredients: null,
120
- notes: 'CONTROLLED, REPEATABLE named-voice VOICEOVER — route here whenever the exact words matter and must be re-renderable in the same voice (ad reads, narration, character lines to lip-sync against). AUDIO-ONLY. The text field IS the script: it is spoken verbatim, so never put stage directions in it. 1-5000 characters, billed per 100-character bucket, so trimming a sentence genuinely saves credits. 20 preset voices (Rachel default) — pick one and keep it for the whole piece. stability 0-1 trades consistency against expressiveness (low = more emotive and more variable). No voice cloning on this route. Word-level timestamps come back free and are what a future caption pass consumes. For scene ambience or SFX rather than speech, use seed-audio / eleven-sfx.',
121
- },
122
211
  {
123
212
  id: 'eleven-sfx',
124
213
  label: 'ElevenLabs Sound Effects v2',
@@ -127,14 +216,6 @@ export const MODEL_FACTS = [
127
216
  maxIngredients: null,
128
217
  notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
129
218
  },
130
- {
131
- id: 'suno',
132
- label: 'Suno',
133
- kind: 'audio',
134
- maxRefImages: null,
135
- maxIngredients: null,
136
- notes: 'FULL MUSIC TRACKS with structure — route here for anything a listener would call a song or a score. AUDIO-ONLY. EVERY call returns TWO variations for one flat price, and duration is FREE up to 360s (measured 2026-07-31: a 240s track costs the same as a default one), which makes it the cheapest way to get a bed that outlasts a cut. Two modes: DESCRIPTION mode (customMode=false, prompt is a <=500-char description and the lyrics get written for you) and CUSTOM mode (customMode=true, needs style + title; prompt then holds the EXACT LYRICS, sung as written — never put a description there). instrumental=true scores a scene with no vocals. Steer with style/negativeTags rather than piling adjectives into the prompt. duration 10-360s is V5_5 + custom mode only. Unofficial API wrapper, so treat availability as best-effort.',
137
- },
138
219
  ];
139
220
  const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
140
221
  export function getModelFact(id) {
@@ -9,6 +9,8 @@ export const PARTIALS = {
9
9
  "reference-rules-core": "Identity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates composes the naming for you from your `@mentions` / `#tags` — you never hand-write role labels.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.",
10
10
  "reference-tips-short": "Name each reference inline; never write role essays. Slates does this for you: `@mention` a subject or environment and it composes `Marcus (image 1) in the cafe (image 2)`, citing them in the exact order it sends them. One canonical identity image avoids competing facial renderings; a \"Reference Image Instructions\" block drags reference lighting into your scene. Start with 2-3 focused refs.",
11
11
  "references-read-literally": "> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.",
12
+ "seedance-25-timestamps": "**2.0 does not respond to timestamps and answers only to shot numbers. 2.5 responds to\ninteger-second timestamps.** That is ByteDance's own first line under \"Differences from Seedance\n2.0\", and it is why a 30-second take is usable at all: the length is only worth buying if you can\nsay *when* things happen inside it.\n\nBoth formats are valid on 2.5, and you can mix them — `Shot N` blocks for a storyboard whose\npacing you are happy to leave to the model, timestamps when a beat has to land at a moment.\n\n**Three ways to control time, all first-party:**\n\n| Form | Write it like |\n|---|---|\n| **Interval** | `0-3 seconds… 3-7 seconds… 7-15 seconds` or `[1s-4s]… [4s-8s]… [8s-12s]` |\n| **Time point** | *\"Quick left sideways transition at the 5-second mark.\"* |\n| **Relative** | *\"After 3 seconds, everyone around him shakes their head.\"* · *\"The frame freezes for 1 second after he presses the shutter.\"* |\n\n**The rules that come with them:**\n\n- **One second is the smallest unit.** Integers only — no `2.5s`, no frames.\n- **No gaps in the timeline.** `0-3s… 5-6s…` leaves 3-5s unspecified and the model fills it however\n it likes. Intervals must abut: `0-3s`, `3-7s`, `7-15s`.\n- **Budget the plot to the seconds.** Too little content in a range and the model improvises to\n fill it; too much and you get extra cuts or dropped beats. This is the actual craft of a 30s take.\n- **Never time-code a high-frequency action.** *\"Shake your head three times per second\"* is\n explicitly called out as a misuse — timestamps schedule beats, they don't choreograph frames.\n- **Transitions want both halves:** the moment AND the method — *\"At the 5-second mark, the camera\n transitions leftward with a left wipe into a natural dissolve.\"*\n- **Timestamps work on an EDIT too**, and that is where they earn the most: they scope a change in\n time as well as in content — *\"Change the man's action from drinking coffee to mopping the floor\n from 4-6 seconds in Video 1, and leave the rest of the content unchanged.\"* Without a range, a\n whole-clip instruction is applied to the whole clip.\n\nDo **not** carry this back to 2.0, and do not carry Veo's `[00:00-00:02]` bracket syntax into\neither — 2.0 ignores time entirely, and the cross-model syntax swap is its own known failure.",
13
+ "seedance-25-timestamps-short": "Seedance 2.0 ignores timing and answers only to \"Shot 1 / Shot 2\"; 2.5 acts on whole-second timestamps, and that is what makes a 30-second take controllable rather than just long. Three forms work: intervals (\"0-3 seconds…3-7 seconds\"), a point (\"at the 5-second mark\"), or relative (\"after 3 seconds\"). Whole seconds only, no gaps between intervals, and never to choreograph fast repeated motion. They work on edits too, where a range scopes the change: \"…from 4-6 seconds…\".",
12
14
  "still-gate": "**A visible defect in the still is already a STOP.** Do not animate it. Fix the frame first, then move to motion — and go to motion only when the crop passes the still scan and you genuinely need movement to confirm an uncertain edge, reflection, or object.\n\nThis is a **cost** rule as much as a craft rule: a 1080p/10s premium video generation costs many multiples of an image re-roll, and video is where a defect stops being fixable. Anything wrong in the still gets worse in motion — soft geometry mushes, broken-but-plausible objects fall apart, oily textures start crawling. **Animating a known-bad frame is the single most expensive mistake in the pipeline.** Re-rolling the image is the cheap move; re-rolling the video is not.",
13
15
  };
14
16
  //# sourceMappingURL=partials.generated.js.map
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
16
16
  /** Footer callout paragraphs. */
17
17
  footer?: string[];
18
18
  }
19
- export type PromptingTipsKey = 'seedance' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-v3' | 'eleven-sfx' | 'suno';
19
+ export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
20
20
  export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
21
21
  /** Null when no tips exist for the key — callers render an honest fallback. */
22
22
  export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
@@ -1,7 +1,20 @@
1
- // Per-model PROMPTING TIPS — the user-facing card content rendered by the
2
- // desktop app's "See prompting tips" modals. SINGLE SOURCE OF TRUTH: this
3
- // file. The desktop renders whatever this exports (no hand-written tips JSX
4
- // in slate — that's how the Omni Flash / Veo / Kling chimera modal shipped).
1
+ // Per-model PROMPTING TIPS — the short, curated, FREE prompting guidance.
2
+ // SINGLE SOURCE OF TRUTH: this file. Nothing downstream authors tips copy (no
3
+ // hand-written tips JSX or prose anywhere — that's how the Omni Flash / Veo /
4
+ // Kling chimera modal shipped).
5
+ //
6
+ // WHERE IT RENDERS (changed 2026-08-10): exactly one place, the generated page
7
+ // https://slates.video/docs/prompting, emitted by slates-web
8
+ // scripts/build-llm-docs.mjs via scripts/llm-docs/extract-tips.ts. It used to
9
+ // render inside the desktop app's Settings modal — 92 cards, 12 families,
10
+ // two-up, in a 448px drawer. It is documentation, so it lives on the web; the
11
+ // app links to it. Do not add an in-app renderer back (slate/CLAUDE.md →
12
+ // Prompting-tips SSOT).
13
+ //
14
+ // A NEW ENTRY NEEDS A READING GROUP. extract-tips.ts owns the order the page
15
+ // lists families in; a key that appears in no group ships in this package and
16
+ // renders on no page. The generator warns, it does not fail — check the
17
+ // `npm run build:llm-docs` output.
5
18
  //
6
19
  // Relationship to the skills: packages/shared/skills/slates-prompting-*.md
7
20
  // are the LONG-FORM agent guidance; these tips are the curated end-user
@@ -33,7 +46,7 @@ const SEEDANCE = {
33
46
  {
34
47
  heading: 'Shot 1 / Shot 2 / Shot 3 — never time stamps',
35
48
  example: 'Shot 1: Side shot of the alley; the man slowly starts running.\nShot 2: He knocks over a fruit stand; the camera shakes and cuts to his face.\nShot 3: He climbs a low wall; the camera pulls back onto the empty street.',
36
- note: 'ByteDance: write a "Shot 1 / Shot 2 / Shot 3" storyboard in the order events occur, then merge it into one prompt. Do NOT write "At 4 seconds" or "0:00–0:03" and do not set per-shot durations — official docs say precise timing is unstable and forcing it "may lead to abnormal generation results." Let the plot set the pacing.',
49
+ note: 'ByteDance: write a "Shot 1 / Shot 2 / Shot 3" storyboard in the order events occur, then merge it into one prompt. Do NOT write "At 4 seconds" or "0:00–0:03" and do not set per-shot durations — Seedance 2.0 does not respond to timestamps at all, and forcing them "may lead to abnormal generation results." Let the plot set the pacing. (Seedance 2.5 is the exception: it does read integer-second timestamps.)',
37
50
  critical: true,
38
51
  },
39
52
  {
@@ -68,6 +81,12 @@ const SEEDANCE = {
68
81
  example: 'The earbud rises smoothly. The camera tracks upward.',
69
82
  note: 'Two different sentences. Mixing them ("the camera speed ramps as the earbud rises") is a common cause of shaky, glitchy output.',
70
83
  },
84
+ {
85
+ heading: 'Images, clips and audio in ONE generation',
86
+ example: 'Marcus (image 1) performs the motion from video 1, speaking the line in audio 1.',
87
+ note: 'Attaching a clip does NOT mean "edit this clip". A video or audio attachment is a REFERENCE, numbered in the rail exactly like an image, and it sits alongside your images in the same generation — the composer cites them as "image N", "video N", "audio N", in rail order, and shows you the exact sentence before you press Generate. Reorder the tiles to change what those numbers mean. To actually rewrite a clip, use Edit with AI instead — that is a different, deliberate choice.',
88
+ critical: true,
89
+ },
71
90
  {
72
91
  heading: 'Multi-character shots — forbid twins',
73
92
  example: 'Throughout the video, characters with completely identical appearance, clothing, and accessories are prohibited. Do not generate duplicate avatars or a twin effect.',
@@ -78,10 +97,111 @@ const SEEDANCE = {
78
97
  footer: [
79
98
  'Quality and constraint slots have their own official vocabulary: ask for "HD, rich details, cinematic texture, natural colors, soft lighting" — not "8K / masterpiece / trending on artstation." Seedance has no negative-prompt field, so constraints go inline: "keep it subtitle-free", "do not generate a logo", "do not generate a watermark".',
80
99
  'Style block at the end: one primary anchor plus 2-3 supporting details. End with "Single continuous take" if you want one shot with no cuts. Never write "no cut" or "seamless transition" — those aren\'t in the training vocabulary.',
81
- 'Multi-modal: up to 9 images, 3 videos and 3 audio references. Cite them by type and index — "Zhang San@Image 1", or the "Marcus (image 1)" form Slates composes from your @mentions. Never cite an asset ID instead of the image number; the model can\'t associate the two. Max length: 4,000 characters.',
100
+ 'Multi-modal: up to 9 images, 3 videos and 3 audio references — 12 files in total, with the reference video capped at 15 seconds combined and the audio at 15. An audio reference on 2.0 needs at least one image or video alongside it (2.5 accepts audio on its own). Cite them by type and index — "Zhang San@Image 1", or the "Marcus (image 1)" form Slates composes from your @mentions. Never cite an asset ID instead of the image number; the model can\'t associate the two. Max length: 4,000 characters.',
101
+ 'A reference VIDEO changes the price: it bills input seconds PLUS output seconds, summed across every clip attached. Two 5-second references on an 8-second generation bills 18 seconds, not 8. The Generate button and the duration menu both show that total before you commit. Over the cap is refused rather than trimmed, precisely so you are never charged for a clip the model never saw.',
82
102
  'Don\'t cross-pollinate image-model syntax: named lenses, apertures and film stocks ("85mm f/1.4", "Kodak Portra 400") are a Nano Banana lever and a Seedance anti-pattern. Translate them into shot size, depth of field and colour tone instead.',
83
103
  ],
84
104
  };
105
+ /** The 2.0 tip that 2.5 REVERSES — 2.0 ignores timestamps, 2.5 acts on them.
106
+ * Matched by heading so the 2.5 entry drops it rather than contradicting it. */
107
+ const SEEDANCE_NO_TIMESTAMPS_HEADING = 'Shot 1 / Shot 2 / Shot 3 — never time stamps';
108
+ const SEEDANCE_25 = {
109
+ ...SEEDANCE,
110
+ label: 'Seedance 2.5',
111
+ intro: [
112
+ 'Seedance 2.5 is a SECOND SEAT next to 2.0, not an upgrade of it. It buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), audio-only references and integer-second timestamps — and it gives up 1080p and 4K entirely. It is 480p or 720p, on every route.',
113
+ "ByteDance's official advanced formula has 8 slots: precise subject + action details + scene/environment + lighting & color tone + camera movement + visual style + image quality + constraints. Sweet spot 60-150 words for a single shot, longer for multi-shot.",
114
+ ],
115
+ columns: [
116
+ [
117
+ {
118
+ heading: 'Do not write edit instructions here',
119
+ example: '\u274c a wide shot of the workshop, remove the tripod\n\u2705 the workshop bench, clear and uncluttered',
120
+ note: 'With references attached, "add", "insert", "remove", "delete", "modify", "replace", "change", "extend" and "continue" make Seedance 2.5 treat the request as a video EDIT, and it then fails on constraints it never set — after the job has queued. Describe the finished frame instead. To actually edit a clip, attach it and pick Seedance 2.5 Edit.',
121
+ critical: true,
122
+ },
123
+ {
124
+ heading: 'Timestamps work here — they do not on 2.0',
125
+ example: '0-3 seconds: he steps off the curb, rain starting.\n3-7 seconds: headlights sweep across him; he turns.\n7-15 seconds: he runs; the camera falls behind.',
126
+ note: PARTIALS['seedance-25-timestamps-short'],
127
+ critical: true,
128
+ },
129
+ ...SEEDANCE.columns[0].filter((t) => t.heading !== SEEDANCE_NO_TIMESTAMPS_HEADING),
130
+ ],
131
+ [
132
+ {
133
+ heading: '720p is not the cheap one here',
134
+ example: '30s \u00b7 720p \u00b7 Face route = 484 credits\n15s \u00b7 1080p \u00b7 Seedance 2.0 Face = 411 credits',
135
+ note: 'Length is what moves the price, and 2.5 doubles the length ceiling — so a 30-second 720p clip can cost more than a 15-second 1080p one, against a 1,000-credit starting balance. Draft at 480p and 4-8 seconds; spend the length only on a take you already know works. The Generate button always shows the exact number first.',
136
+ critical: true,
137
+ },
138
+ {
139
+ heading: 'Audio-only references',
140
+ example: 'Reference the timbre in audio 1 to generate...',
141
+ note: '2.5 accepts an audio reference on its own — a voice line, a music bed, a room tone — with no image or video alongside it. 2.0 could not. Audio references never cost extra on any Seedance route.',
142
+ },
143
+ ...SEEDANCE.columns[1],
144
+ ],
145
+ ],
146
+ footer: [
147
+ '30 image references is a budget, not a target — 2-4 strong references still beat both extremes, one per role. ByteDance\'s own ceilings for 2.5: 1-8 subjects bound by image reference stay stable (9-12 works but needs re-rolls), 1-5 subjects bound by video or audio reference, and 5-10 seconds is the sweet spot for a reference clip. Unlike 2.0, a multi-view turnaround sheet can be a single subject reference here — past 5 subjects, go back to one view per image. The larger budget is for long multi-shot takes and for video plus audio references alongside images.',
148
+ 'A reference VIDEO bills input seconds PLUS output seconds, and 2.5 accepts references up to 30s combined — so a 20-second reference driving a 20-second output bills 40 seconds. The Generate button shows the total.',
149
+ ...(SEEDANCE.footer ?? []).slice(0, 2),
150
+ 'Frames and reference images stay mutually exclusive, and on a first/last-frame generation Seedance 2.5 chooses the aspect ratio itself — the ratio control shows "Adaptive" because the start frame decides the shape.',
151
+ ],
152
+ };
153
+ const SEEDANCE_25_EDIT = {
154
+ ...SEEDANCE_25,
155
+ label: 'Seedance 2.5 Edit',
156
+ intro: [
157
+ 'Seedance 2.5 Edit changes an existing clip: attach the clip, describe only what should be different, and the original motion, framing and timing are kept. It is the only editor in Slates that takes a clip longer than 15 seconds — 4 to 30s, against Kling O3 Edit\'s 3-15s and Omni Flash Edit\'s 3-10s.',
158
+ 'Output length and aspect ratio follow the SOURCE clip, so there is no duration or ratio control — the clip you attach is the quote. Output is 480p or 720p with native audio.',
159
+ ],
160
+ columns: [
161
+ [
162
+ {
163
+ heading: 'Name the change, keep the rest',
164
+ example: 'Strictly edit the clip, and change the blue jacket from navy to red.',
165
+ note: 'The clip already carries its composition, motion, timing and performance — re-describing them fights the model. Say the change as "from A to B" rather than as an outcome: naming what it currently is tells the model what to overwrite. One change per pass; chain passes for compound edits. Never write "reference the video" in an edit: that phrasing gets the request re-read as a fresh generation inspired by your clip instead of an edit of it.',
166
+ critical: true,
167
+ },
168
+ {
169
+ heading: 'Scope the edit in time',
170
+ example: 'Change the man\'s action from drinking coffee to mopping the floor from 4-6 seconds, and leave the rest of the clip unchanged.',
171
+ note: PARTIALS['seedance-25-timestamps-short'],
172
+ critical: true,
173
+ },
174
+ {
175
+ heading: 'Turn Face on when a face is visible',
176
+ note: 'The default provider blocks character faces outright — this is not a price optimisation, it is whether the job runs at all. There is no consented-real-face route for editing; real-person footage the Face route rejects has to go to Kling O3 Edit.',
177
+ critical: true,
178
+ },
179
+ {
180
+ heading: 'An edit costs about double a generation',
181
+ note: 'Every provider bills an edit on the input clip AND the output, so a 20-second edit is priced like 40 seconds of generation. Read the number on the Generate button rather than reasoning from the generation rate.',
182
+ },
183
+ ],
184
+ [
185
+ {
186
+ heading: 'When to use it instead of the others',
187
+ note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner and the cheapest seat, and Kling O3 Edit is the one that takes subject and style reference images.',
188
+ },
189
+ {
190
+ heading: 'It edits the audio too',
191
+ example: 'Only edit the man\'s dialogue: change it to "Don\'t come over here," in an American accent. Keep everything else unchanged.',
192
+ note: 'The same engine rewrites what is heard while the picture stays put — change a line, change an accent, translate the dialogue and re-fit the lip movement, or strip and replace music and sound effects. Priced like any other edit, on the source clip\'s length.',
193
+ },
194
+ {
195
+ heading: 'Prompt and clip only',
196
+ note: 'No character or style reference images on this engine in Slates today. If the edit needs a reference image to lock an identity, that is Kling O3 Edit\'s job.',
197
+ },
198
+ ],
199
+ ],
200
+ footer: [
201
+ 'Trim before you edit, not after: the bill is the source clip\'s length rounded up, so a 30-second clip you only needed 8 seconds of costs nearly four times what it had to.',
202
+ 'Clips under 20 seconds edit most reliably. Up to 30 is accepted, and the returned clip can land within about a third of a second of the source length — only transition frames are compressed, nothing is cut.',
203
+ ],
204
+ };
85
205
  const KLING = {
86
206
  label: 'Kling 3.0',
87
207
  intro: [
@@ -360,7 +480,7 @@ const NANO_BANANA_LITE = {
360
480
  ],
361
481
  };
362
482
  // ── Audio lane ──────────────────────────────────────────────────
363
- // The four audio surfaces prompt NOTHING like the video models. The single
483
+ // The two audio surfaces prompt NOTHING like the video models. The single
364
484
  // most expensive mistake is bringing Kling's "SFX:" / "Ambient noise:" syntax
365
485
  // to Seed Audio, which reads it as literal text. Every entry below leads with
366
486
  // what the surface actually wants.
@@ -412,55 +532,10 @@ const SEED_AUDIO = {
412
532
  },
413
533
  {
414
534
  heading: 'Know its seat',
415
- note: 'Scenes, beds and room tone in one pass. For an exact script in a repeatable voice use Eleven v3; for a single effect on a specific frame use Sound Effects; for a song use Suno.',
416
- },
417
- ],
418
- ],
419
- };
420
- const ELEVEN_V3 = {
421
- label: 'ElevenLabs Eleven v3',
422
- intro: [
423
- 'Eleven v3 speaks your text verbatim in a named voice. It is the controlled, repeatable lane: the same text and the same voice give you a read you can regenerate after a script tweak without the performance drifting.',
424
- 'Billing is per 100 characters of text, rounded up — so tightening a sentence genuinely costs less, and a stray pasted paragraph genuinely costs more.',
425
- ],
426
- columns: [
427
- [
428
- {
429
- heading: 'The text field is the script',
430
- example: '✗ (excited) Say this fast: Grab yours today!\n✓ Grab yours today!',
431
- note: 'Everything you type gets spoken. Stage directions, character names and bracketed notes will be read out loud.',
432
- critical: true,
433
- },
434
- {
435
- heading: 'Punctuate for pace',
436
- example: 'It works. Every time. · It works — every time…',
437
- note: 'Full stops, commas, dashes and ellipses are your only timing controls. Rewrite the punctuation before you touch the settings.',
438
- },
439
- {
440
- heading: 'Pick a voice and stay',
441
- note: '20 preset voices. Choosing one per character or per piece is what makes a series sound deliberate; swapping voices mid-piece reads as a mistake.',
442
- },
443
- ],
444
- [
445
- {
446
- heading: 'Stability',
447
- example: '0.3 emotive · 0.5 default · 0.8 steady',
448
- note: 'Lower is more expressive and more variable take-to-take. Higher is flatter and more repeatable. Raise it for long narration, lower it for a single dramatic line.',
449
- },
450
- {
451
- heading: 'Spell out the tricky bits',
452
- example: 'SKU → "ess kay you" · 2026 → "twenty twenty six"',
453
- note: 'Acronyms, product names, prices and years are where TTS embarrasses itself. Write the pronunciation you want.',
454
- },
455
- {
456
- heading: 'Know its seat',
457
- note: 'Exact words, repeatable voice, lines you will lip-sync against. Ambience, crowds and rooms belong to Seed Audio; one-shot effects to Sound Effects.',
535
+ note: 'Scenes, beds, room tone and dialogue in one pass. For a single effect that has to land on a specific frame, use Sound Effects.',
458
536
  },
459
537
  ],
460
538
  ],
461
- footer: [
462
- 'Word-level timestamps come back with every generation at no extra cost — that is what a future caption pass will read, so there is no reason to turn them off.',
463
- ],
464
539
  };
465
540
  const ELEVEN_SFX = {
466
541
  label: 'ElevenLabs Sound Effects',
@@ -504,53 +579,10 @@ const ELEVEN_SFX = {
504
579
  ],
505
580
  ],
506
581
  };
507
- const SUNO = {
508
- label: 'Suno',
509
- intro: [
510
- 'Suno writes full music. Every generation returns TWO variations for one flat price, and length is free up to six minutes — so there is never a reason to generate a bed that is shorter than your edit.',
511
- 'The one thing to get right is which mode you are in, because it changes what the prompt field means.',
512
- ],
513
- columns: [
514
- [
515
- {
516
- heading: 'Description mode',
517
- example: 'brooding synthwave for a night drive, analog bass, no vocals',
518
- note: 'The prompt is a description (max 500 characters) and the lyrics get written for you. This is the fast path when you just need a mood.',
519
- },
520
- {
521
- heading: 'Custom mode — the prompt IS the lyrics',
522
- example: 'Style: dream pop, hazy\nTitle: Blue Hour\nPrompt: [Verse 1] The lights come on…',
523
- note: 'In custom mode the prompt is sung exactly as written. Putting a description there gets your description sung back at you.',
524
- critical: true,
525
- },
526
- {
527
- heading: 'Instrumental',
528
- note: 'Turn instrumental on to score a scene with no vocals — style and title still steer it, and the prompt field is ignored.',
529
- },
530
- ],
531
- [
532
- {
533
- heading: 'Steer with style, not adjectives',
534
- example: 'Style: 90s trip-hop, dusty breakbeat, Rhodes\nAvoid: brass, EDM drops',
535
- note: 'Genre, era, instrumentation and tempo belong in the style field. Negative tags remove what keeps creeping in.',
536
- },
537
- {
538
- heading: 'Length is free',
539
- example: '10–360 seconds',
540
- note: 'A six-minute track costs exactly what a default one does. Ask for longer than the cut needs and trim on the timeline.',
541
- },
542
- {
543
- heading: 'Two songs, both yours',
544
- note: 'Both variations land in the gallery. They are genuinely different takes on the same brief — audition both before re-rolling.',
545
- },
546
- ],
547
- ],
548
- footer: [
549
- 'Files are hosted by the provider for a limited window — Slates downloads and stores them on your machine as soon as the track finishes, so nothing expires out from under a project.',
550
- ],
551
- };
552
582
  export const PROMPTING_TIPS = {
553
583
  seedance: SEEDANCE,
584
+ 'seedance-2-5': SEEDANCE_25,
585
+ 'seedance-2-5-edit': SEEDANCE_25_EDIT,
554
586
  kling: KLING,
555
587
  'kling-edit': KLING_EDIT,
556
588
  veo: VEO,
@@ -559,9 +591,7 @@ export const PROMPTING_TIPS = {
559
591
  'nano-banana': NANO_BANANA,
560
592
  'nano-banana-lite': NANO_BANANA_LITE,
561
593
  'seed-audio': SEED_AUDIO,
562
- 'eleven-v3': ELEVEN_V3,
563
594
  'eleven-sfx': ELEVEN_SFX,
564
- suno: SUNO,
565
595
  };
566
596
  /** Null when no tips exist for the key — callers render an honest fallback. */
567
597
  export function getPromptingTips(key) {
@@ -1,7 +1,7 @@
1
- export type ReferenceKind = 'character' | 'environment' | 'style' | 'pinned' | 'first-frame' | 'last-frame' | 'video';
1
+ export type ReferenceKind = 'character' | 'environment' | 'style' | 'pinned' | 'first-frame' | 'last-frame' | 'video' | 'video-ref' | 'audio-ref';
2
2
  export interface ReferenceMedia {
3
3
  path: string;
4
- mediaKind: 'image' | 'video';
4
+ mediaKind: 'image' | 'video' | 'audio';
5
5
  }
6
6
  /**
7
7
  * A named bucket of reference media that can be @mentioned. The whole definition
@@ -22,8 +22,25 @@ export interface ComposedReferences {
22
22
  prompt: string;
23
23
  /** Free-reference image paths in cited order — flatten yields "image 1..N". */
24
24
  orderedImagePaths: string[];
25
- /** Video reference paths in cited order — "video 1..M". */
25
+ /** Video paths in cited order — "video 1..M". Edit sources and reference
26
+ * clips SHARE this list and one counter, so a mixed state can never emit
27
+ * two "Video 1"s. */
26
28
  orderedVideoPaths: string[];
29
+ /** Reference audio paths in cited order — "audio 1..K". */
30
+ orderedAudioPaths: string[];
31
+ /**
32
+ * Tokens written in the prompt that matched NO reference group, as authored
33
+ * (`'#noir'`, `'@bob'`), first-appearance order, deduped case-insensitively.
34
+ *
35
+ * 🚨 THIS FIELD EXISTS BECAUSE THE ALTERNATIVE IS A SILENT EDIT. A `#tag` with
36
+ * no matching style is DELETED from the text — a raw tag confuses every model,
37
+ * so removing it is right, but removing it without saying so changes what the
38
+ * user asked for behind their back. Prompt-transparency doctrine (slate
39
+ * `CLAUDE.md`) requires the surface to be able to say "this went nowhere", and
40
+ * this is the only record that the token was ever there. Callers that render a
41
+ * composed-prompt preview MUST surface it.
42
+ */
43
+ unresolvedTokens: string[];
27
44
  }
28
45
  /**
29
46
  * Compose the raw prompt (mentions intact) + an ORDERED list of reference groups
@@ -41,6 +58,7 @@ export interface ComposedReferences {
41
58
  export interface ComposeOptions {
42
59
  startImageNumber?: number;
43
60
  startVideoNumber?: number;
61
+ startAudioNumber?: number;
44
62
  }
45
63
  export declare function composeReferences(rawPrompt: string, groups: ReferenceGroup[], opts?: ComposeOptions): ComposedReferences;
46
64
  export interface KlingEditElement {