@slatesvideo/shared 0.5.5 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. package/dist/index.d.ts +2 -2
  2. package/dist/index.js +6 -3
  3. package/dist/operations/index.d.ts +106 -19
  4. package/dist/operations/index.js +624 -172
  5. package/dist/prompts/character-sheet.d.ts +2 -1
  6. package/dist/prompts/character-sheet.js +82 -13
  7. package/dist/prompts/model-facts.d.ts +43 -1
  8. package/dist/prompts/model-facts.js +104 -2
  9. package/dist/prompts/prompting-tips.d.ts +1 -1
  10. package/dist/prompts/prompting-tips.js +208 -5
  11. package/dist/prompts/reference-composer.d.ts +21 -3
  12. package/dist/prompts/reference-composer.js +80 -10
  13. package/dist/prompts/reference-rules.d.ts +19 -2
  14. package/dist/prompts/reference-rules.js +18 -1
  15. package/dist/skills/content.js +8 -5
  16. package/exports/slates-prompt-builder/generated/SKILL.md +2 -2
  17. package/exports/slates-prompt-builder/generated/reference-character.md +9 -4
  18. package/exports/slates-prompt-builder/generated/reference-seedance.md +12 -1
  19. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +11 -11
  20. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  21. package/package.json +1 -1
  22. package/skills/slates-character-identity.md +9 -4
  23. package/skills/slates-model-selection.md +39 -11
  24. package/skills/slates-prompting-elevenlabs.md +69 -0
  25. package/skills/slates-prompting-lip-sync.md +12 -14
  26. package/skills/slates-prompting-motion-transfer.md +18 -14
  27. package/skills/slates-prompting-seed-audio.md +110 -0
  28. package/skills/slates-prompting-seedance-2-5.md +215 -0
  29. package/skills/slates-prompting-seedance.md +17 -1
@@ -9,7 +9,8 @@
9
9
  * that cannot match the portrait's, so the sheet would carry two competing
10
10
  * identities and the model averages them. A back view has no face to compete
11
11
  * with, and it is the only panel where hair fall reads. See the header comment
12
- * for the receipt and for why the phrasing must stay framing, not removal.
12
+ * for the receipt, and for why the phrasing must stay FRAMING rather than
13
+ * removal and must be SCOPED TO THE FACE rather than to the whole body.
13
14
  */
14
15
  export declare const CHARACTER_SHEET_PANELS_DESC: string;
15
16
  /** Panel identifiers, in sheet order. */
@@ -11,14 +11,45 @@
11
11
  // 2. One authoritative face avoids averaging competing renderings.
12
12
  // 3. One generation means one asset to inspect and bind.
13
13
  //
14
- // KNOWN COST, accepted for v1: a neutral chest-up portrait carries no dental
15
- // information, so a character who smiles in a shot gets invented teeth. The
16
- // 2026-06-26 doctrine already holds that the user's prompt owns expression.
17
- // Revisit only if a receipt shows invented teeth.
14
+ // EXPRESSION IS A SLIGHT SMILE WITH THE TEETH JUST VISIBLE (Eric, 2026-07-30,
15
+ // replacing the v1 neutral default). The v1 note called neutral's missing
16
+ // dental information a "known cost, revisit only if a receipt shows invented
17
+ // teeth" — this is that revisit, and it arrived from the other direction:
18
+ // Eric hand-edited a sheet to smiling and the result "worked really well".
19
+ //
20
+ // WHY: a closed-mouth portrait carries ZERO dental information, so every
21
+ // downstream shot where the character smiles has to invent teeth, and teeth are
22
+ // person-specific and stable — inventing them is a highly visible identity
23
+ // break. The smile also records the nasolabial fold, the eye crinkle and where
24
+ // the cheeks sit raised, none of which a neutral mouth shows.
25
+ //
26
+ // THE COST, NAMED: `references-read-literally.md` says a baked-in property is
27
+ // read as a property of the SUBJECT, so a smile risks a character who smiles
28
+ // through a beat that asked for grief. Survivable because the 2026-06-26
29
+ // doctrine still holds — the user's prompt owns expression, and downstream
30
+ // prompts name an emotional register on every delivered line.
31
+ //
32
+ // SLIGHT, never a grin: a broad smile deforms eyes, cheeks and mouth enough
33
+ // that the model has to un-deform it for any neutral shot.
34
+ //
35
+ // NON-HUMANS ARE CARVED OUT (Eric, 2026-07-30). A smile clause on a horse, a
36
+ // dragon or a robot produces bared teeth. Non-human characters get a natural
37
+ // neutral expression instead. NOTE THE SCOPE DIFFERS from the A-pose carve-out
38
+ // beside it: that one is anatomical (quadruped / non-bipedal), this one is
39
+ // about having a human mouth — so a BIPEDAL robot or humanoid alien is covered
40
+ // by this carve-out and not by that one. Two carve-outs, deliberately, because
41
+ // one predicate does not fit both.
42
+ //
43
+ // Receipt strength: N=1, and it is a "looked right" judgement, not a scored
44
+ // identity-hold comparison against the neutral sheets. HOW YOU'D KNOW THIS IS
45
+ // BEATEN: a character smiling through a beat prompted angry or grieving, or
46
+ // identity holding measurably worse than neutral did. Reverting is one line.
18
47
  //
19
48
  // THE FRONT PANEL IS HEADLESS (shipped 2026-07-22, receipt-gated). The face is
20
- // cropped off the front body panel — the "ghost mannequin" treatment — taking
21
- // competing face renderings 2 → 1. The BACK panel keeps its head: it has no
49
+ // cropped off the front body panel, taking competing face renderings 2 → 1.
50
+ // ONLY the face — the body, neck, arms and hands render normally; see the
51
+ // 2026-07-30 (b) note below for what happens when that isn't said. The BACK
52
+ // panel keeps its head: it has no
22
53
  // face to compete with the portrait, and it is the only panel where hair fall
23
54
  // reads. The rule is "kill every competing rendering of the FACE", not "kill
24
55
  // every head".
@@ -27,9 +58,45 @@
27
58
  // real generations (research/model-prompting-research.md, "Head-crop receipt"):
28
59
  // 1. It is prompt-reachable. NB2 renders a clean invisible-mannequin panel
29
60
  // with no refusal. THIS DEPENDS ON THE PHRASING: it is framed as FRAMING
30
- // ("cropped at the collarbone, head not shown, invisible-mannequin
31
- // presentation"), a standard e-commerce genre with deep training data.
32
- // Never phrase it as removal or decapitation.
61
+ // ("cropped at the collarbone, invisible-mannequin presentation"), a
62
+ // standard e-commerce genre with deep training data. Never phrase it as
63
+ // removal or decapitation.
64
+ //
65
+ // 2026-07-30 (a) — THE PHRASING NARROWED, and the receipt above was
66
+ // MODEL-SCOPED. "framed from the collarbone down with the head not shown"
67
+ // passed NB2 and is a HARD 422 on gpt-image-2: fal returns
68
+ // content_policy_violation with loc ["body","prompt"], so the text is
69
+ // rejected before any image is read. Diagnosis: "the head not shown"
70
+ // states an anatomical ABSENCE — a headless human body — which reads as
71
+ // gore to OpenAI's classifier. "cropped at the collarbone" states a
72
+ // CAMERA fact and carries the same instruction.
73
+ // RULE: describe an exclusion as a framing choice, never as a missing
74
+ // body part. The "invisible-mannequin" genre anchor is KEPT because the
75
+ // NB2 receipt says it is what makes the panel reachable there.
76
+ // HOW YOU'D KNOW THIS IS BEATEN: a front panel that comes back with a
77
+ // head on it, meaning "cropped at the collarbone" alone is too weak
78
+ // without the absence clause. If that happens, the fix is a
79
+ // model-conditional phrasing, not restoring the 422.
80
+ //
81
+ // 2026-07-30 (b) — THE GENRE ANCHOR HAS TO BE SCOPED TO THE FACE, and
82
+ // this one cost real generations. "an invisible-mannequin presentation
83
+ // WHERE THE CLOTHING HOLDS ITS OWN SHAPE" is the e-commerce genre stated
84
+ // in full — and the full genre means NO BODY AT ALL, an empty outfit
85
+ // photographed on nothing. Eric's sheets came back with the skin removed:
86
+ // no neck, no hands, no forearms, a floating garment. The genre anchor
87
+ // was doing exactly what it says.
88
+ // Eric's replacement, verbatim, and the default since: "an invisible-
89
+ // mannequin presentation with just the face cropped out." Same anchor
90
+ // (still what makes the panel reachable on NB2), bounded so the only
91
+ // thing missing is the face.
92
+ // RULE: a genre anchor imports the WHOLE genre unless you bound it —
93
+ // name what STAYS, not just what goes. This is the same failure shape as
94
+ // (a) from the opposite side: (a) was an exclusion phrased too
95
+ // anatomically, (b) was an exclusion scoped too widely.
96
+ // HOW YOU'D KNOW THIS IS BEATEN: front panels returning with a head on
97
+ // them — "just the face cropped out" would then be reading as a face
98
+ // edit rather than a crop, leaving the collarbone clause to carry it
99
+ // alone.
33
100
  // 2. The literal-reading law does NOT fire on it. This was the real risk and
34
101
  // it is ours, not the source corpus's: `references-read-literally.md` says
35
102
  // a baked-in property is read as a property of the SUBJECT, and this panel
@@ -65,11 +132,12 @@ import { renderStyleInstruction } from './style-library.js';
65
132
  * that cannot match the portrait's, so the sheet would carry two competing
66
133
  * identities and the model averages them. A back view has no face to compete
67
134
  * with, and it is the only panel where hair fall reads. See the header comment
68
- * for the receipt and for why the phrasing must stay framing, not removal.
135
+ * for the receipt, and for why the phrasing must stay FRAMING rather than
136
+ * removal and must be SCOPED TO THE FACE rather than to the whole body.
69
137
  */
70
138
  export const CHARACTER_SHEET_PANELS_DESC = 'a large chest-up portrait on the left at a three-quarter angle (never dead-on), ' +
71
- 'a full-body front view in a relaxed A-pose in the centre, framed from the collarbone down with the head not shown — ' +
72
- 'an invisible-mannequin presentation where the clothing holds its own shape, ' +
139
+ 'a full-body front view in a relaxed A-pose in the centre, cropped at the collarbone — ' +
140
+ 'an invisible-mannequin presentation with just the face cropped out, ' +
73
141
  'and a full-body back view on the right with the head and hair fully visible';
74
142
  /** Panel identifiers, in sheet order. */
75
143
  export const BODY_POSE_LABELS = ['portrait', 'front', 'back'];
@@ -89,8 +157,9 @@ export function buildCharacterIdentityPrompt(userStyle) {
89
157
  `${CHARACTER_SHEET_PANELS_DESC}. ` +
90
158
  `The portrait is the largest panel and occupies roughly a quarter to a third of the sheet — it is the sole authority for the face, so render it at maximum facial detail. ` +
91
159
  `No second rendering of the face anywhere on the sheet. ` +
92
- `Neutral expression and identical appearance, wardrobe and hair across all three panels. ` +
160
+ `A slight natural smile with the teeth just visible, and identical appearance, wardrobe and hair across all three panels. ` +
93
161
  `${styleDirective(userStyle)} ${IDENTITY_LIGHTING_CLAUSE} ${IDENTITY_CRAFT_CLAUSE} ` +
162
+ `For non-human characters, use a natural neutral expression instead of a smile. ` +
94
163
  `For quadruped or non-bipedal characters, replace the A-pose with a natural standing stance, show the whole animal including the head on both body panels, and keep the same three-panel layout. ` +
95
164
  `No text, no labels, no captions, no panel borders.`);
96
165
  }
@@ -1,13 +1,55 @@
1
1
  export interface ModelFact {
2
2
  id: string;
3
3
  label: string;
4
- kind: 'image' | 'video';
4
+ kind: 'image' | 'video' | 'audio';
5
5
  /** Max reference images (image models) — null if not applicable. */
6
6
  maxRefImages: number | null;
7
7
  /** Max ingredient images (video models) — null if not applicable. */
8
8
  maxIngredients: number | null;
9
+ /** Reference VIDEOS accepted in one generation. null/absent = none. */
10
+ maxReferenceVideos?: number | null;
11
+ /** Reference AUDIO clips accepted in one generation. null/absent = none. */
12
+ maxReferenceAudio?: number | null;
13
+ /** Combined seconds across every reference video. */
14
+ maxReferenceVideoSeconds?: number | null;
15
+ /** Combined seconds across every reference audio clip. */
16
+ maxReferenceAudioSeconds?: number | null;
17
+ /** Ceiling on TOTAL reference files across all modalities. */
18
+ maxReferenceFilesTotal?: number | null;
19
+ /** An audio reference needs at least one image or video reference alongside. */
20
+ audioRefNeedsCompanion?: boolean;
9
21
  notes: string;
10
22
  }
23
+ /**
24
+ * One sentence of multimodal-reference capacity for a model, derived. Returns
25
+ * an empty string for a model that takes none, so a caller can append it
26
+ * unconditionally.
27
+ */
28
+ export declare function multimodalRefSummary(id: string): string;
29
+ /**
30
+ * The prompt words that make Seedance 2.5 reclassify a reference-carrying
31
+ * request as a video EDIT or EXTEND — after which it fails on task-type
32
+ * constraints it never set, ASYNCHRONOUSLY, once the job has queued.
33
+ *
34
+ * 🚨 THIS DRIVES A WARNING THAT NAMES THE WORDS. It must never drive a rewrite:
35
+ * silently mutating the user's prompt to dodge a provider classifier is banned
36
+ * by the prompt-transparency invariant (slate/CLAUDE.md). "Remove the tripod" is
37
+ * the user's sentence; the honest move is to say what will happen.
38
+ *
39
+ * ⚠️ MIRRORED, and the mirror is deliberate. The desktop's copy is
40
+ * `SEEDANCE_EDIT_INTENT_KEYWORDS` + `SEEDANCE_EXTEND_INTENT_KEYWORDS` in
41
+ * `slate/src/shared/pricing.ts`, which cannot import from this package (it is
42
+ * loaded by the renderer through the `@shared/*` alias, with no npm dependency).
43
+ * Same situation as `promptComposition.ts` ↔ `reference-composer.ts`. Change one,
44
+ * change the other in the same pass. Both sides match on a WORD BOUNDARY, so
45
+ * "added" and "readdress" are not hits.
46
+ */
47
+ export declare const SEEDANCE_TASK_INTENT_WORDS: readonly ["add", "remove", "replace", "change", "edit the video", "extend", "continue", "continue the story"];
48
+ /** Which trigger words a prompt actually contains, so a warning can name them.
49
+ * Mirrors `seedanceTaskIntentWords()` in slate/src/shared/pricing.ts. */
50
+ export declare function seedanceTaskIntentWords(prompt: string): string[];
51
+ /** Every model that reads reference video and/or audio, for op descriptions. */
52
+ export declare function multimodalRefModels(): string[];
11
53
  export declare const MODEL_FACTS: ModelFact[];
12
54
  export declare function getModelFact(id: string): ModelFact | undefined;
13
55
  /** The official NB2 / general image prompt formula (subject-first). */
@@ -3,6 +3,63 @@
3
3
  // RUNTIME source of truth for limits is slate/src/shared/pricing.ts
4
4
  // (MODEL_REGISTRY.maxRefImages / maxIngredientImages); these mirror it for
5
5
  // documentation. Code-verified 2026-06-25.
6
+ /**
7
+ * One sentence of multimodal-reference capacity for a model, derived. Returns
8
+ * an empty string for a model that takes none, so a caller can append it
9
+ * unconditionally.
10
+ */
11
+ export function multimodalRefSummary(id) {
12
+ const f = MODEL_FACTS.find((m) => m.id === id);
13
+ if (!f)
14
+ return '';
15
+ const v = f.maxReferenceVideos ?? 0;
16
+ const a = f.maxReferenceAudio ?? 0;
17
+ if (v === 0 && a === 0)
18
+ return '';
19
+ const parts = [];
20
+ if (v > 0)
21
+ parts.push(`${v} reference video${v === 1 ? '' : 's'} (${f.maxReferenceVideoSeconds}s combined)`);
22
+ if (a > 0)
23
+ parts.push(`${a} reference audio clip${a === 1 ? '' : 's'} (${f.maxReferenceAudioSeconds}s combined)`);
24
+ const total = f.maxReferenceFilesTotal ? `, ${f.maxReferenceFilesTotal} files max across all modalities` : '';
25
+ const companion = f.audioRefNeedsCompanion
26
+ ? ' Audio needs at least one image or video reference alongside it.'
27
+ : ' Audio-only references are allowed.';
28
+ return `${f.label}: up to ${parts.join(' and ')}${total}.${companion}`;
29
+ }
30
+ /**
31
+ * The prompt words that make Seedance 2.5 reclassify a reference-carrying
32
+ * request as a video EDIT or EXTEND — after which it fails on task-type
33
+ * constraints it never set, ASYNCHRONOUSLY, once the job has queued.
34
+ *
35
+ * 🚨 THIS DRIVES A WARNING THAT NAMES THE WORDS. It must never drive a rewrite:
36
+ * silently mutating the user's prompt to dodge a provider classifier is banned
37
+ * by the prompt-transparency invariant (slate/CLAUDE.md). "Remove the tripod" is
38
+ * the user's sentence; the honest move is to say what will happen.
39
+ *
40
+ * ⚠️ MIRRORED, and the mirror is deliberate. The desktop's copy is
41
+ * `SEEDANCE_EDIT_INTENT_KEYWORDS` + `SEEDANCE_EXTEND_INTENT_KEYWORDS` in
42
+ * `slate/src/shared/pricing.ts`, which cannot import from this package (it is
43
+ * loaded by the renderer through the `@shared/*` alias, with no npm dependency).
44
+ * Same situation as `promptComposition.ts` ↔ `reference-composer.ts`. Change one,
45
+ * change the other in the same pass. Both sides match on a WORD BOUNDARY, so
46
+ * "added" and "readdress" are not hits.
47
+ */
48
+ export const SEEDANCE_TASK_INTENT_WORDS = [
49
+ 'add', 'remove', 'replace', 'change', 'edit the video',
50
+ 'extend', 'continue', 'continue the story',
51
+ ];
52
+ /** Which trigger words a prompt actually contains, so a warning can name them.
53
+ * Mirrors `seedanceTaskIntentWords()` in slate/src/shared/pricing.ts. */
54
+ export function seedanceTaskIntentWords(prompt) {
55
+ return SEEDANCE_TASK_INTENT_WORDS.filter((w) => new RegExp(`\\b${w.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'i').test(prompt));
56
+ }
57
+ /** Every model that reads reference video and/or audio, for op descriptions. */
58
+ export function multimodalRefModels() {
59
+ return MODEL_FACTS
60
+ .filter((m) => (m.maxReferenceVideos ?? 0) > 0 || (m.maxReferenceAudio ?? 0) > 0)
61
+ .map((m) => m.id);
62
+ }
6
63
  export const MODEL_FACTS = [
7
64
  {
8
65
  id: 'nano-banana-2',
@@ -61,7 +118,36 @@ export const MODEL_FACTS = [
61
118
  kind: 'video',
62
119
  maxRefImages: null,
63
120
  maxIngredients: 9, // ingredient images per video gen
64
- notes: 'PREMIUM video tier — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). Up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Also the PREMIUM engine inside the Motion Transfer and Lip Sync tools (single-pass: driving video / dialogue are native conditioning signals — better motion fidelity, natural speech, voice cloned from a video source; video references bill input+output seconds).',
121
+ maxReferenceVideos: 3,
122
+ maxReferenceAudio: 3,
123
+ maxReferenceVideoSeconds: 15,
124
+ maxReferenceAudioSeconds: 15,
125
+ maxReferenceFilesTotal: 12,
126
+ audioRefNeedsCompanion: true,
127
+ notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds. 2.0 STAYS THE DEFAULT over 2.5 because it is the only Seedance with 1080p and 4K.',
128
+ },
129
+ {
130
+ id: 'seedance-2.5',
131
+ label: 'Seedance 2.5',
132
+ kind: 'video',
133
+ maxRefImages: null,
134
+ maxIngredients: 30, // 30 image refs; the model also takes 10 video + 10 audio (50 total)
135
+ maxReferenceVideos: 10,
136
+ maxReferenceAudio: 10,
137
+ maxReferenceVideoSeconds: 30,
138
+ maxReferenceAudioSeconds: 30,
139
+ maxReferenceFilesTotal: 50,
140
+ // No companion requirement — audio-only references are one of the things
141
+ // the second seat actually buys.
142
+ notes: 'A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is 480p/720p ONLY. No 1080p, no 4K, on any provider. Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), or tighter prompt adherence. Pick 2.0 when resolution matters at all. VIDEO-ONLY. 🚨 COST DISCIPLINE: 720p STOPS READING AS "THE CHEAP ONE" HERE. A 30s 720p clip on the real-face route is 710 credits and on the AI-face route 484 — more than a 15s 1080p Seedance 2.0 face generation (411), against a 1,000-credit welcome grant. Always quote with slates_estimate_generation_cost before a long take, and draft at 480p/4-8s. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.',
143
+ },
144
+ {
145
+ id: 'seedance-2.5-edit',
146
+ label: 'Seedance 2.5 Edit',
147
+ kind: 'video',
148
+ maxRefImages: null,
149
+ maxIngredients: 0, // prompt + source clip only on slates_edit_video
150
+ notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s). That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p output, native audio. Prompt + source clip only on this op — no reference images. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
65
151
  },
66
152
  {
67
153
  id: 'kling-v3',
@@ -69,7 +155,7 @@ export const MODEL_FACTS = [
69
155
  kind: 'video',
70
156
  maxRefImages: null,
71
157
  maxIngredients: 4,
72
- notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. In the Motion Transfer / Lip Sync tools, Kling (MC / lip-sync / avatar) is the cheap utility lane; Seedance is the premium single-pass lane.',
158
+ notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools (MC std/pro, lip-sync, avatar) — those two tools are Kling-only.',
73
159
  },
74
160
  {
75
161
  id: 'kling-v3-edit',
@@ -103,6 +189,22 @@ export const MODEL_FACTS = [
103
189
  maxIngredients: 0, // prompt + source clip ONLY — no element/style refs on this endpoint
104
190
  notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (7/09 head-to-head vs Kling edit on real talking footage: lips held perfectly, audio near-identical, both action beats landed). Takes an EXISTING 3-10s clip and changes what the prompt names, footage-synced (prop/effect/environment/lighting swaps). Fidelity is EARNED by prompt discipline: ONE short instruction + "Keep everything else the same." — long descriptive prompts DESTROY it (Google-documented + 7/09 receipt). Never name objects as metaphors ("candle-like" → literal candle). Quirk: occasional tail jitter/doubled last speech beat — trim the tail. NO reference images (identity swaps needing refs → Kling edit); bit-exact audio needs → Kling keep_audio or segment-splice. 720p output, cheapest edit seat (~2/3 of Kling edit Std).',
105
191
  },
192
+ {
193
+ id: 'seed-audio',
194
+ label: 'Seed Audio 1.0',
195
+ kind: 'audio',
196
+ maxRefImages: 1, // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
197
+ maxIngredients: null,
198
+ notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
199
+ },
200
+ {
201
+ id: 'eleven-sfx',
202
+ label: 'ElevenLabs Sound Effects v2',
203
+ kind: 'audio',
204
+ maxRefImages: null,
205
+ maxIngredients: null,
206
+ notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
207
+ },
106
208
  ];
107
209
  const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
108
210
  export function getModelFact(id) {
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
16
16
  /** Footer callout paragraphs. */
17
17
  footer?: string[];
18
18
  }
19
- export type PromptingTipsKey = 'seedance' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite';
19
+ export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
20
20
  export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
21
21
  /** Null when no tips exist for the key — callers render an honest fallback. */
22
22
  export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
@@ -1,7 +1,20 @@
1
- // Per-model PROMPTING TIPS — the user-facing card content rendered by the
2
- // desktop app's "See prompting tips" modals. SINGLE SOURCE OF TRUTH: this
3
- // file. The desktop renders whatever this exports (no hand-written tips JSX
4
- // in slate — that's how the Omni Flash / Veo / Kling chimera modal shipped).
1
+ // Per-model PROMPTING TIPS — the short, curated, FREE prompting guidance.
2
+ // SINGLE SOURCE OF TRUTH: this file. Nothing downstream authors tips copy (no
3
+ // hand-written tips JSX or prose anywhere — that's how the Omni Flash / Veo /
4
+ // Kling chimera modal shipped).
5
+ //
6
+ // WHERE IT RENDERS (changed 2026-08-10): exactly one place, the generated page
7
+ // https://slates.video/docs/prompting, emitted by slates-web
8
+ // scripts/build-llm-docs.mjs via scripts/llm-docs/extract-tips.ts. It used to
9
+ // render inside the desktop app's Settings modal — 92 cards, 12 families,
10
+ // two-up, in a 448px drawer. It is documentation, so it lives on the web; the
11
+ // app links to it. Do not add an in-app renderer back (slate/CLAUDE.md →
12
+ // Prompting-tips SSOT).
13
+ //
14
+ // A NEW ENTRY NEEDS A READING GROUP. extract-tips.ts owns the order the page
15
+ // lists families in; a key that appears in no group ships in this package and
16
+ // renders on no page. The generator warns, it does not fail — check the
17
+ // `npm run build:llm-docs` output.
5
18
  //
6
19
  // Relationship to the skills: packages/shared/skills/slates-prompting-*.md
7
20
  // are the LONG-FORM agent guidance; these tips are the curated end-user
@@ -68,6 +81,12 @@ const SEEDANCE = {
68
81
  example: 'The earbud rises smoothly. The camera tracks upward.',
69
82
  note: 'Two different sentences. Mixing them ("the camera speed ramps as the earbud rises") is a common cause of shaky, glitchy output.',
70
83
  },
84
+ {
85
+ heading: 'Images, clips and audio in ONE generation',
86
+ example: 'Marcus (image 1) performs the motion from video 1, speaking the line in audio 1.',
87
+ note: 'Attaching a clip does NOT mean "edit this clip". A video or audio attachment is a REFERENCE, numbered in the rail exactly like an image, and it sits alongside your images in the same generation — the composer cites them as "image N", "video N", "audio N", in rail order, and shows you the exact sentence before you press Generate. Reorder the tiles to change what those numbers mean. To actually rewrite a clip, use Edit with AI instead — that is a different, deliberate choice.',
88
+ critical: true,
89
+ },
71
90
  {
72
91
  heading: 'Multi-character shots — forbid twins',
73
92
  example: 'Throughout the video, characters with completely identical appearance, clothing, and accessories are prohibited. Do not generate duplicate avatars or a twin effect.',
@@ -78,10 +97,90 @@ const SEEDANCE = {
78
97
  footer: [
79
98
  'Quality and constraint slots have their own official vocabulary: ask for "HD, rich details, cinematic texture, natural colors, soft lighting" — not "8K / masterpiece / trending on artstation." Seedance has no negative-prompt field, so constraints go inline: "keep it subtitle-free", "do not generate a logo", "do not generate a watermark".',
80
99
  'Style block at the end: one primary anchor plus 2-3 supporting details. End with "Single continuous take" if you want one shot with no cuts. Never write "no cut" or "seamless transition" — those aren\'t in the training vocabulary.',
81
- 'Multi-modal: up to 9 images, 3 videos and 3 audio references. Cite them by type and index — "Zhang San@Image 1", or the "Marcus (image 1)" form Slates composes from your @mentions. Never cite an asset ID instead of the image number; the model can\'t associate the two. Max length: 4,000 characters.',
100
+ 'Multi-modal: up to 9 images, 3 videos and 3 audio references — 12 files in total, with the reference video capped at 15 seconds combined and the audio at 15. An audio reference on 2.0 needs at least one image or video alongside it (2.5 accepts audio on its own). Cite them by type and index — "Zhang San@Image 1", or the "Marcus (image 1)" form Slates composes from your @mentions. Never cite an asset ID instead of the image number; the model can\'t associate the two. Max length: 4,000 characters.',
101
+ 'A reference VIDEO changes the price: it bills input seconds PLUS output seconds, summed across every clip attached. Two 5-second references on an 8-second generation bills 18 seconds, not 8. The Generate button and the duration menu both show that total before you commit. Over the cap is refused rather than trimmed, precisely so you are never charged for a clip the model never saw.',
82
102
  'Don\'t cross-pollinate image-model syntax: named lenses, apertures and film stocks ("85mm f/1.4", "Kodak Portra 400") are a Nano Banana lever and a Seedance anti-pattern. Translate them into shot size, depth of field and colour tone instead.',
83
103
  ],
84
104
  };
105
+ const SEEDANCE_25 = {
106
+ ...SEEDANCE,
107
+ label: 'Seedance 2.5',
108
+ intro: [
109
+ 'Seedance 2.5 is a SECOND SEAT next to 2.0, not an upgrade of it. It buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), and audio-only references — and it gives up 1080p and 4K entirely. It is 480p or 720p, on every route. Everything below about writing the prompt is the same as 2.0.',
110
+ "ByteDance's official advanced formula has 8 slots: precise subject + action details + scene/environment + lighting & color tone + camera movement + visual style + image quality + constraints. Sweet spot 60-150 words for a single shot, longer for multi-shot.",
111
+ ],
112
+ columns: [
113
+ [
114
+ {
115
+ heading: 'Do not write edit instructions here',
116
+ example: '\u274c a wide shot of the workshop, remove the tripod\n\u2705 the workshop bench, clear and uncluttered',
117
+ note: 'With references attached, "add", "remove", "replace", "change", "extend" and "continue" make Seedance 2.5 treat the request as a video EDIT, and it then fails on constraints it never set — after the job has queued. Describe the finished frame instead. To actually edit a clip, attach it and pick Seedance 2.5 Edit.',
118
+ critical: true,
119
+ },
120
+ ...SEEDANCE.columns[0],
121
+ ],
122
+ [
123
+ {
124
+ heading: '720p is not the cheap one here',
125
+ example: '30s \u00b7 720p \u00b7 Face route = 484 credits\n15s \u00b7 1080p \u00b7 Seedance 2.0 Face = 411 credits',
126
+ note: 'Length is what moves the price, and 2.5 doubles the length ceiling — so a 30-second 720p clip can cost more than a 15-second 1080p one, against a 1,000-credit starting balance. Draft at 480p and 4-8 seconds; spend the length only on a take you already know works. The Generate button always shows the exact number first.',
127
+ critical: true,
128
+ },
129
+ {
130
+ heading: 'Audio-only references',
131
+ example: 'Reference the timbre in audio 1 to generate...',
132
+ note: '2.5 accepts an audio reference on its own — a voice line, a music bed, a room tone — with no image or video alongside it. 2.0 could not. Audio references never cost extra on any Seedance route.',
133
+ },
134
+ ...SEEDANCE.columns[1],
135
+ ],
136
+ ],
137
+ footer: [
138
+ '30 image references is a budget, not a target. Every reference rule still holds: 2-4 strong references beat both extremes, one reference per role, one authoritative rendering per subject — and past 4 reference PEOPLE, output stability drops regardless of the cap. The larger budget is for long multi-shot takes and for video plus audio references alongside images.',
139
+ 'A reference VIDEO bills input seconds PLUS output seconds, and 2.5 accepts references up to 30s combined — so a 20-second reference driving a 20-second output bills 40 seconds. The Generate button shows the total.',
140
+ ...(SEEDANCE.footer ?? []).slice(0, 2),
141
+ 'Frames and reference images stay mutually exclusive, and on a first/last-frame generation Seedance 2.5 chooses the aspect ratio itself — the ratio control shows "Adaptive" because the start frame decides the shape.',
142
+ ],
143
+ };
144
+ const SEEDANCE_25_EDIT = {
145
+ ...SEEDANCE_25,
146
+ label: 'Seedance 2.5 Edit',
147
+ intro: [
148
+ 'Seedance 2.5 Edit changes an existing clip: attach the clip, describe only what should be different, and the original motion, framing and timing are kept. It is the only editor in Slates that takes a clip longer than 15 seconds — 4 to 30s, against Kling O3 Edit\'s 3-15s and Omni Flash Edit\'s 3-10s.',
149
+ 'Output length and aspect ratio follow the SOURCE clip, so there is no duration or ratio control — the clip you attach is the quote. Output is 480p or 720p with native audio.',
150
+ ],
151
+ columns: [
152
+ [
153
+ {
154
+ heading: 'Name the change, keep the rest',
155
+ example: 'Strictly edit the clip, and change the blue jacket to a red one.',
156
+ note: 'The clip already carries its composition, motion, timing and performance — re-describing them fights the model. One change per pass; chain passes for compound edits. Never write "reference the video" in an edit: that phrasing gets the request re-read as a fresh generation inspired by your clip instead of an edit of it.',
157
+ critical: true,
158
+ },
159
+ {
160
+ heading: 'Turn Face on when a face is visible',
161
+ note: 'The default provider blocks character faces outright — this is not a price optimisation, it is whether the job runs at all. There is no consented-real-face route for editing; real-person footage the Face route rejects has to go to Kling O3 Edit.',
162
+ critical: true,
163
+ },
164
+ {
165
+ heading: 'An edit costs about double a generation',
166
+ note: 'Every provider bills an edit on the input clip AND the output, so a 20-second edit is priced like 40 seconds of generation. Read the number on the Generate button rather than reasoning from the generation rate.',
167
+ },
168
+ ],
169
+ [
170
+ {
171
+ heading: 'When to use it instead of the others',
172
+ note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner and the cheapest seat, and Kling O3 Edit is the one that takes subject and style reference images.',
173
+ },
174
+ {
175
+ heading: 'Prompt and clip only',
176
+ note: 'No character or style reference images on this engine. If the edit needs a reference image to lock an identity, that is Kling O3 Edit\'s job.',
177
+ },
178
+ ],
179
+ ],
180
+ footer: [
181
+ 'Trim before you edit, not after: the bill is the source clip\'s length rounded up, so a 30-second clip you only needed 8 seconds of costs nearly four times what it had to.',
182
+ ],
183
+ };
85
184
  const KLING = {
86
185
  label: 'Kling 3.0',
87
186
  intro: [
@@ -359,8 +458,110 @@ const NANO_BANANA_LITE = {
359
458
  : card),
360
459
  ],
361
460
  };
461
+ // ── Audio lane ──────────────────────────────────────────────────
462
+ // The two audio surfaces prompt NOTHING like the video models. The single
463
+ // most expensive mistake is bringing Kling's "SFX:" / "Ambient noise:" syntax
464
+ // to Seed Audio, which reads it as literal text. Every entry below leads with
465
+ // what the surface actually wants.
466
+ const SEED_AUDIO = {
467
+ label: 'Seed Audio 1.0',
468
+ intro: [
469
+ 'Seed Audio builds a whole audio scene — dialogue, effects and ambience together — from one plain sentence. Write it the way you would describe the moment to a person standing next to you, not the way you would write a video prompt.',
470
+ 'It has no duration setting. Length comes from the words, so Slates appends your chosen duration to the prompt ("… 15 seconds") and bills exactly that. Set the duration control to what you actually want and let the sentence stay clean.',
471
+ ],
472
+ columns: [
473
+ [
474
+ {
475
+ heading: 'One plain sentence',
476
+ example: 'nature soundscape, wide open field cicadas and birds and a loon.',
477
+ note: 'No shot language, no production jargon, no formatting. Plain description outperforms anything that reads like a spec sheet.',
478
+ },
479
+ {
480
+ heading: 'Duration lives in the prompt',
481
+ example: 'tiny applause of 2 or 3 people at an open mic. 15 seconds',
482
+ note: 'The duration control writes this for you. Do not also type a different length into your sentence — the two will fight and you pay for the one you selected.',
483
+ critical: true,
484
+ },
485
+ {
486
+ heading: 'Say the crowd size',
487
+ example: 'tiny applause of 2 or 3 people · a packed arena roaring',
488
+ note: '"Applause" alone returns a full room. Scale words are the single highest-leverage edit on any crowd, traffic or nature bed.',
489
+ },
490
+ {
491
+ heading: 'Cut beds longer than the shot',
492
+ note: 'Ask for a few seconds more than the clip needs so the edit has handles to fade in and out of. Beds that end exactly on the cut always sound clipped.',
493
+ },
494
+ ],
495
+ [
496
+ {
497
+ heading: 'No Kling syntax here',
498
+ example: '✗ SFX: heavy boots\n✓ heavy boots on wet pavement, a siren far off',
499
+ note: 'The "SFX:" and "Ambient noise:" prefixes belong to Kling video prompts. Seed Audio treats them as words in the scene and the result gets worse.',
500
+ critical: true,
501
+ },
502
+ {
503
+ heading: 'Dialogue in quotes',
504
+ example: 'a tired bartender says, "we closed twenty minutes ago", glasses clinking behind him',
505
+ note: 'Speech goes in quotes inside the same sentence as the room. Pick a preset voice for a specific speaker, or leave it unset and let the scene cast itself.',
506
+ },
507
+ {
508
+ heading: 'References',
509
+ example: 'match the room tone of @Audio1',
510
+ note: 'Up to 3 audio clips (max 30s each), referenced as @Audio1–@Audio3 — OR one image to score what is in frame. Never both in the same generation.',
511
+ },
512
+ {
513
+ heading: 'Know its seat',
514
+ note: 'Scenes, beds, room tone and dialogue in one pass. For a single effect that has to land on a specific frame, use Sound Effects.',
515
+ },
516
+ ],
517
+ ],
518
+ };
519
+ const ELEVEN_SFX = {
520
+ label: 'ElevenLabs Sound Effects',
521
+ intro: [
522
+ 'Sound Effects makes one short sound with an exact length — the lane for a hit that has to land on a specific frame, or a seamless loop you can lay under a whole scene.',
523
+ 'Duration is always sent explicitly (0.5–22s). Billing is per second, so the length you pick is the price you pay.',
524
+ ],
525
+ columns: [
526
+ [
527
+ {
528
+ heading: 'Describe the cause, not the label',
529
+ example: '✗ door sound\n✓ heavy oak door slams shut in a stone hallway',
530
+ note: 'Material, weight and room are what separate a usable effect from a stock-library shrug. Name all three.',
531
+ critical: true,
532
+ },
533
+ {
534
+ heading: 'One sound per generation',
535
+ note: 'This surface makes a single event. A door, then footsteps, then a siren is three generations layered on the timeline — or one Seed Audio scene.',
536
+ },
537
+ {
538
+ heading: 'Duration is the edit',
539
+ example: '0.8s for an impact · 4s for a whoosh · 22s for a bed',
540
+ note: 'Ask for roughly the length you need. A 4-second request for a door slam pads the tail with room tone you then have to trim.',
541
+ },
542
+ ],
543
+ [
544
+ {
545
+ heading: 'Loops',
546
+ example: 'steady rain on a canvas tent (loop on, 12s)',
547
+ note: 'Turn loop on for anything continuous — rain, engine hum, crowd murmur — and it will tile without a seam.',
548
+ },
549
+ {
550
+ heading: 'Prompt influence',
551
+ example: '0.3 default · 0.7 literal',
552
+ note: 'Higher hugs your wording with less variation between takes; lower explores. Raise it when a re-roll keeps wandering off the brief.',
553
+ },
554
+ {
555
+ heading: 'Know its seat',
556
+ note: 'One precise effect on a known frame. Full rooms and layered scenes are cheaper and better in one Seed Audio pass.',
557
+ },
558
+ ],
559
+ ],
560
+ };
362
561
  export const PROMPTING_TIPS = {
363
562
  seedance: SEEDANCE,
563
+ 'seedance-2-5': SEEDANCE_25,
564
+ 'seedance-2-5-edit': SEEDANCE_25_EDIT,
364
565
  kling: KLING,
365
566
  'kling-edit': KLING_EDIT,
366
567
  veo: VEO,
@@ -368,6 +569,8 @@ export const PROMPTING_TIPS = {
368
569
  'omni-flash-edit': OMNI_FLASH_EDIT,
369
570
  'nano-banana': NANO_BANANA,
370
571
  'nano-banana-lite': NANO_BANANA_LITE,
572
+ 'seed-audio': SEED_AUDIO,
573
+ 'eleven-sfx': ELEVEN_SFX,
371
574
  };
372
575
  /** Null when no tips exist for the key — callers render an honest fallback. */
373
576
  export function getPromptingTips(key) {