@slatesvideo/shared 0.5.6 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +4 -1
- package/dist/operations/index.d.ts +43 -36
- package/dist/operations/index.js +289 -262
- package/dist/prompts/model-facts.d.ts +42 -0
- package/dist/prompts/model-facts.js +88 -18
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +108 -99
- package/dist/prompts/reference-composer.d.ts +21 -3
- package/dist/prompts/reference-composer.js +80 -10
- package/dist/skills/content.js +7 -7
- package/exports/slates-prompt-builder/generated/SKILL.md +2 -2
- package/exports/slates-prompt-builder/generated/reference-seedance.md +12 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +8 -8
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +1 -1
- package/skills/slates-model-selection.md +21 -19
- package/skills/slates-prompting-elevenlabs.md +16 -78
- package/skills/slates-prompting-lip-sync.md +12 -14
- package/skills/slates-prompting-motion-transfer.md +18 -14
- package/skills/slates-prompting-seed-audio.md +2 -2
- package/skills/slates-prompting-seedance-2-5.md +215 -0
- package/skills/slates-prompting-seedance.md +17 -1
- package/skills/slates-prompting-suno.md +0 -110
|
@@ -6,8 +6,50 @@ export interface ModelFact {
|
|
|
6
6
|
maxRefImages: number | null;
|
|
7
7
|
/** Max ingredient images (video models) — null if not applicable. */
|
|
8
8
|
maxIngredients: number | null;
|
|
9
|
+
/** Reference VIDEOS accepted in one generation. null/absent = none. */
|
|
10
|
+
maxReferenceVideos?: number | null;
|
|
11
|
+
/** Reference AUDIO clips accepted in one generation. null/absent = none. */
|
|
12
|
+
maxReferenceAudio?: number | null;
|
|
13
|
+
/** Combined seconds across every reference video. */
|
|
14
|
+
maxReferenceVideoSeconds?: number | null;
|
|
15
|
+
/** Combined seconds across every reference audio clip. */
|
|
16
|
+
maxReferenceAudioSeconds?: number | null;
|
|
17
|
+
/** Ceiling on TOTAL reference files across all modalities. */
|
|
18
|
+
maxReferenceFilesTotal?: number | null;
|
|
19
|
+
/** An audio reference needs at least one image or video reference alongside. */
|
|
20
|
+
audioRefNeedsCompanion?: boolean;
|
|
9
21
|
notes: string;
|
|
10
22
|
}
|
|
23
|
+
/**
|
|
24
|
+
* One sentence of multimodal-reference capacity for a model, derived. Returns
|
|
25
|
+
* an empty string for a model that takes none, so a caller can append it
|
|
26
|
+
* unconditionally.
|
|
27
|
+
*/
|
|
28
|
+
export declare function multimodalRefSummary(id: string): string;
|
|
29
|
+
/**
|
|
30
|
+
* The prompt words that make Seedance 2.5 reclassify a reference-carrying
|
|
31
|
+
* request as a video EDIT or EXTEND — after which it fails on task-type
|
|
32
|
+
* constraints it never set, ASYNCHRONOUSLY, once the job has queued.
|
|
33
|
+
*
|
|
34
|
+
* 🚨 THIS DRIVES A WARNING THAT NAMES THE WORDS. It must never drive a rewrite:
|
|
35
|
+
* silently mutating the user's prompt to dodge a provider classifier is banned
|
|
36
|
+
* by the prompt-transparency invariant (slate/CLAUDE.md). "Remove the tripod" is
|
|
37
|
+
* the user's sentence; the honest move is to say what will happen.
|
|
38
|
+
*
|
|
39
|
+
* ⚠️ MIRRORED, and the mirror is deliberate. The desktop's copy is
|
|
40
|
+
* `SEEDANCE_EDIT_INTENT_KEYWORDS` + `SEEDANCE_EXTEND_INTENT_KEYWORDS` in
|
|
41
|
+
* `slate/src/shared/pricing.ts`, which cannot import from this package (it is
|
|
42
|
+
* loaded by the renderer through the `@shared/*` alias, with no npm dependency).
|
|
43
|
+
* Same situation as `promptComposition.ts` ↔ `reference-composer.ts`. Change one,
|
|
44
|
+
* change the other in the same pass. Both sides match on a WORD BOUNDARY, so
|
|
45
|
+
* "added" and "readdress" are not hits.
|
|
46
|
+
*/
|
|
47
|
+
export declare const SEEDANCE_TASK_INTENT_WORDS: readonly ["add", "remove", "replace", "change", "edit the video", "extend", "continue", "continue the story"];
|
|
48
|
+
/** Which trigger words a prompt actually contains, so a warning can name them.
|
|
49
|
+
* Mirrors `seedanceTaskIntentWords()` in slate/src/shared/pricing.ts. */
|
|
50
|
+
export declare function seedanceTaskIntentWords(prompt: string): string[];
|
|
51
|
+
/** Every model that reads reference video and/or audio, for op descriptions. */
|
|
52
|
+
export declare function multimodalRefModels(): string[];
|
|
11
53
|
export declare const MODEL_FACTS: ModelFact[];
|
|
12
54
|
export declare function getModelFact(id: string): ModelFact | undefined;
|
|
13
55
|
/** The official NB2 / general image prompt formula (subject-first). */
|
|
@@ -3,6 +3,63 @@
|
|
|
3
3
|
// RUNTIME source of truth for limits is slate/src/shared/pricing.ts
|
|
4
4
|
// (MODEL_REGISTRY.maxRefImages / maxIngredientImages); these mirror it for
|
|
5
5
|
// documentation. Code-verified 2026-06-25.
|
|
6
|
+
/**
|
|
7
|
+
* One sentence of multimodal-reference capacity for a model, derived. Returns
|
|
8
|
+
* an empty string for a model that takes none, so a caller can append it
|
|
9
|
+
* unconditionally.
|
|
10
|
+
*/
|
|
11
|
+
export function multimodalRefSummary(id) {
|
|
12
|
+
const f = MODEL_FACTS.find((m) => m.id === id);
|
|
13
|
+
if (!f)
|
|
14
|
+
return '';
|
|
15
|
+
const v = f.maxReferenceVideos ?? 0;
|
|
16
|
+
const a = f.maxReferenceAudio ?? 0;
|
|
17
|
+
if (v === 0 && a === 0)
|
|
18
|
+
return '';
|
|
19
|
+
const parts = [];
|
|
20
|
+
if (v > 0)
|
|
21
|
+
parts.push(`${v} reference video${v === 1 ? '' : 's'} (${f.maxReferenceVideoSeconds}s combined)`);
|
|
22
|
+
if (a > 0)
|
|
23
|
+
parts.push(`${a} reference audio clip${a === 1 ? '' : 's'} (${f.maxReferenceAudioSeconds}s combined)`);
|
|
24
|
+
const total = f.maxReferenceFilesTotal ? `, ${f.maxReferenceFilesTotal} files max across all modalities` : '';
|
|
25
|
+
const companion = f.audioRefNeedsCompanion
|
|
26
|
+
? ' Audio needs at least one image or video reference alongside it.'
|
|
27
|
+
: ' Audio-only references are allowed.';
|
|
28
|
+
return `${f.label}: up to ${parts.join(' and ')}${total}.${companion}`;
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
31
|
+
* The prompt words that make Seedance 2.5 reclassify a reference-carrying
|
|
32
|
+
* request as a video EDIT or EXTEND — after which it fails on task-type
|
|
33
|
+
* constraints it never set, ASYNCHRONOUSLY, once the job has queued.
|
|
34
|
+
*
|
|
35
|
+
* 🚨 THIS DRIVES A WARNING THAT NAMES THE WORDS. It must never drive a rewrite:
|
|
36
|
+
* silently mutating the user's prompt to dodge a provider classifier is banned
|
|
37
|
+
* by the prompt-transparency invariant (slate/CLAUDE.md). "Remove the tripod" is
|
|
38
|
+
* the user's sentence; the honest move is to say what will happen.
|
|
39
|
+
*
|
|
40
|
+
* ⚠️ MIRRORED, and the mirror is deliberate. The desktop's copy is
|
|
41
|
+
* `SEEDANCE_EDIT_INTENT_KEYWORDS` + `SEEDANCE_EXTEND_INTENT_KEYWORDS` in
|
|
42
|
+
* `slate/src/shared/pricing.ts`, which cannot import from this package (it is
|
|
43
|
+
* loaded by the renderer through the `@shared/*` alias, with no npm dependency).
|
|
44
|
+
* Same situation as `promptComposition.ts` ↔ `reference-composer.ts`. Change one,
|
|
45
|
+
* change the other in the same pass. Both sides match on a WORD BOUNDARY, so
|
|
46
|
+
* "added" and "readdress" are not hits.
|
|
47
|
+
*/
|
|
48
|
+
export const SEEDANCE_TASK_INTENT_WORDS = [
|
|
49
|
+
'add', 'remove', 'replace', 'change', 'edit the video',
|
|
50
|
+
'extend', 'continue', 'continue the story',
|
|
51
|
+
];
|
|
52
|
+
/** Which trigger words a prompt actually contains, so a warning can name them.
|
|
53
|
+
* Mirrors `seedanceTaskIntentWords()` in slate/src/shared/pricing.ts. */
|
|
54
|
+
export function seedanceTaskIntentWords(prompt) {
|
|
55
|
+
return SEEDANCE_TASK_INTENT_WORDS.filter((w) => new RegExp(`\\b${w.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'i').test(prompt));
|
|
56
|
+
}
|
|
57
|
+
/** Every model that reads reference video and/or audio, for op descriptions. */
|
|
58
|
+
export function multimodalRefModels() {
|
|
59
|
+
return MODEL_FACTS
|
|
60
|
+
.filter((m) => (m.maxReferenceVideos ?? 0) > 0 || (m.maxReferenceAudio ?? 0) > 0)
|
|
61
|
+
.map((m) => m.id);
|
|
62
|
+
}
|
|
6
63
|
export const MODEL_FACTS = [
|
|
7
64
|
{
|
|
8
65
|
id: 'nano-banana-2',
|
|
@@ -61,7 +118,36 @@ export const MODEL_FACTS = [
|
|
|
61
118
|
kind: 'video',
|
|
62
119
|
maxRefImages: null,
|
|
63
120
|
maxIngredients: 9, // ingredient images per video gen
|
|
64
|
-
|
|
121
|
+
maxReferenceVideos: 3,
|
|
122
|
+
maxReferenceAudio: 3,
|
|
123
|
+
maxReferenceVideoSeconds: 15,
|
|
124
|
+
maxReferenceAudioSeconds: 15,
|
|
125
|
+
maxReferenceFilesTotal: 12,
|
|
126
|
+
audioRefNeedsCompanion: true,
|
|
127
|
+
notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds. 2.0 STAYS THE DEFAULT over 2.5 because it is the only Seedance with 1080p and 4K.',
|
|
128
|
+
},
|
|
129
|
+
{
|
|
130
|
+
id: 'seedance-2.5',
|
|
131
|
+
label: 'Seedance 2.5',
|
|
132
|
+
kind: 'video',
|
|
133
|
+
maxRefImages: null,
|
|
134
|
+
maxIngredients: 30, // 30 image refs; the model also takes 10 video + 10 audio (50 total)
|
|
135
|
+
maxReferenceVideos: 10,
|
|
136
|
+
maxReferenceAudio: 10,
|
|
137
|
+
maxReferenceVideoSeconds: 30,
|
|
138
|
+
maxReferenceAudioSeconds: 30,
|
|
139
|
+
maxReferenceFilesTotal: 50,
|
|
140
|
+
// No companion requirement — audio-only references are one of the things
|
|
141
|
+
// the second seat actually buys.
|
|
142
|
+
notes: 'A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is 480p/720p ONLY. No 1080p, no 4K, on any provider. Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), or tighter prompt adherence. Pick 2.0 when resolution matters at all. VIDEO-ONLY. 🚨 COST DISCIPLINE: 720p STOPS READING AS "THE CHEAP ONE" HERE. A 30s 720p clip on the real-face route is 710 credits and on the AI-face route 484 — more than a 15s 1080p Seedance 2.0 face generation (411), against a 1,000-credit welcome grant. Always quote with slates_estimate_generation_cost before a long take, and draft at 480p/4-8s. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.',
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
id: 'seedance-2.5-edit',
|
|
146
|
+
label: 'Seedance 2.5 Edit',
|
|
147
|
+
kind: 'video',
|
|
148
|
+
maxRefImages: null,
|
|
149
|
+
maxIngredients: 0, // prompt + source clip only on slates_edit_video
|
|
150
|
+
notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s). That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p output, native audio. Prompt + source clip only on this op — no reference images. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
|
|
65
151
|
},
|
|
66
152
|
{
|
|
67
153
|
id: 'kling-v3',
|
|
@@ -69,7 +155,7 @@ export const MODEL_FACTS = [
|
|
|
69
155
|
kind: 'video',
|
|
70
156
|
maxRefImages: null,
|
|
71
157
|
maxIngredients: 4,
|
|
72
|
-
notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics.
|
|
158
|
+
notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools (MC std/pro, lip-sync, avatar) — those two tools are Kling-only.',
|
|
73
159
|
},
|
|
74
160
|
{
|
|
75
161
|
id: 'kling-v3-edit',
|
|
@@ -111,14 +197,6 @@ export const MODEL_FACTS = [
|
|
|
111
197
|
maxIngredients: null,
|
|
112
198
|
notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
|
|
113
199
|
},
|
|
114
|
-
{
|
|
115
|
-
id: 'eleven-v3',
|
|
116
|
-
label: 'ElevenLabs Eleven v3 (TTS)',
|
|
117
|
-
kind: 'audio',
|
|
118
|
-
maxRefImages: null,
|
|
119
|
-
maxIngredients: null,
|
|
120
|
-
notes: 'CONTROLLED, REPEATABLE named-voice VOICEOVER — route here whenever the exact words matter and must be re-renderable in the same voice (ad reads, narration, character lines to lip-sync against). AUDIO-ONLY. The text field IS the script: it is spoken verbatim, so never put stage directions in it. 1-5000 characters, billed per 100-character bucket, so trimming a sentence genuinely saves credits. 20 preset voices (Rachel default) — pick one and keep it for the whole piece. stability 0-1 trades consistency against expressiveness (low = more emotive and more variable). No voice cloning on this route. Word-level timestamps come back free and are what a future caption pass consumes. For scene ambience or SFX rather than speech, use seed-audio / eleven-sfx.',
|
|
121
|
-
},
|
|
122
200
|
{
|
|
123
201
|
id: 'eleven-sfx',
|
|
124
202
|
label: 'ElevenLabs Sound Effects v2',
|
|
@@ -127,14 +205,6 @@ export const MODEL_FACTS = [
|
|
|
127
205
|
maxIngredients: null,
|
|
128
206
|
notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
|
|
129
207
|
},
|
|
130
|
-
{
|
|
131
|
-
id: 'suno',
|
|
132
|
-
label: 'Suno',
|
|
133
|
-
kind: 'audio',
|
|
134
|
-
maxRefImages: null,
|
|
135
|
-
maxIngredients: null,
|
|
136
|
-
notes: 'FULL MUSIC TRACKS with structure — route here for anything a listener would call a song or a score. AUDIO-ONLY. EVERY call returns TWO variations for one flat price, and duration is FREE up to 360s (measured 2026-07-31: a 240s track costs the same as a default one), which makes it the cheapest way to get a bed that outlasts a cut. Two modes: DESCRIPTION mode (customMode=false, prompt is a <=500-char description and the lyrics get written for you) and CUSTOM mode (customMode=true, needs style + title; prompt then holds the EXACT LYRICS, sung as written — never put a description there). instrumental=true scores a scene with no vocals. Steer with style/negativeTags rather than piling adjectives into the prompt. duration 10-360s is V5_5 + custom mode only. Unofficial API wrapper, so treat availability as best-effort.',
|
|
137
|
-
},
|
|
138
208
|
];
|
|
139
209
|
const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
|
|
140
210
|
export function getModelFact(id) {
|
|
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
|
|
|
16
16
|
/** Footer callout paragraphs. */
|
|
17
17
|
footer?: string[];
|
|
18
18
|
}
|
|
19
|
-
export type PromptingTipsKey = 'seedance' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-
|
|
19
|
+
export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
|
|
20
20
|
export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
|
|
21
21
|
/** Null when no tips exist for the key — callers render an honest fallback. */
|
|
22
22
|
export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
|
|
@@ -1,7 +1,20 @@
|
|
|
1
|
-
// Per-model PROMPTING TIPS — the
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
1
|
+
// Per-model PROMPTING TIPS — the short, curated, FREE prompting guidance.
|
|
2
|
+
// SINGLE SOURCE OF TRUTH: this file. Nothing downstream authors tips copy (no
|
|
3
|
+
// hand-written tips JSX or prose anywhere — that's how the Omni Flash / Veo /
|
|
4
|
+
// Kling chimera modal shipped).
|
|
5
|
+
//
|
|
6
|
+
// WHERE IT RENDERS (changed 2026-08-10): exactly one place, the generated page
|
|
7
|
+
// https://slates.video/docs/prompting, emitted by slates-web
|
|
8
|
+
// scripts/build-llm-docs.mjs via scripts/llm-docs/extract-tips.ts. It used to
|
|
9
|
+
// render inside the desktop app's Settings modal — 92 cards, 12 families,
|
|
10
|
+
// two-up, in a 448px drawer. It is documentation, so it lives on the web; the
|
|
11
|
+
// app links to it. Do not add an in-app renderer back (slate/CLAUDE.md →
|
|
12
|
+
// Prompting-tips SSOT).
|
|
13
|
+
//
|
|
14
|
+
// A NEW ENTRY NEEDS A READING GROUP. extract-tips.ts owns the order the page
|
|
15
|
+
// lists families in; a key that appears in no group ships in this package and
|
|
16
|
+
// renders on no page. The generator warns, it does not fail — check the
|
|
17
|
+
// `npm run build:llm-docs` output.
|
|
5
18
|
//
|
|
6
19
|
// Relationship to the skills: packages/shared/skills/slates-prompting-*.md
|
|
7
20
|
// are the LONG-FORM agent guidance; these tips are the curated end-user
|
|
@@ -68,6 +81,12 @@ const SEEDANCE = {
|
|
|
68
81
|
example: 'The earbud rises smoothly. The camera tracks upward.',
|
|
69
82
|
note: 'Two different sentences. Mixing them ("the camera speed ramps as the earbud rises") is a common cause of shaky, glitchy output.',
|
|
70
83
|
},
|
|
84
|
+
{
|
|
85
|
+
heading: 'Images, clips and audio in ONE generation',
|
|
86
|
+
example: 'Marcus (image 1) performs the motion from video 1, speaking the line in audio 1.',
|
|
87
|
+
note: 'Attaching a clip does NOT mean "edit this clip". A video or audio attachment is a REFERENCE, numbered in the rail exactly like an image, and it sits alongside your images in the same generation — the composer cites them as "image N", "video N", "audio N", in rail order, and shows you the exact sentence before you press Generate. Reorder the tiles to change what those numbers mean. To actually rewrite a clip, use Edit with AI instead — that is a different, deliberate choice.',
|
|
88
|
+
critical: true,
|
|
89
|
+
},
|
|
71
90
|
{
|
|
72
91
|
heading: 'Multi-character shots — forbid twins',
|
|
73
92
|
example: 'Throughout the video, characters with completely identical appearance, clothing, and accessories are prohibited. Do not generate duplicate avatars or a twin effect.',
|
|
@@ -78,10 +97,90 @@ const SEEDANCE = {
|
|
|
78
97
|
footer: [
|
|
79
98
|
'Quality and constraint slots have their own official vocabulary: ask for "HD, rich details, cinematic texture, natural colors, soft lighting" — not "8K / masterpiece / trending on artstation." Seedance has no negative-prompt field, so constraints go inline: "keep it subtitle-free", "do not generate a logo", "do not generate a watermark".',
|
|
80
99
|
'Style block at the end: one primary anchor plus 2-3 supporting details. End with "Single continuous take" if you want one shot with no cuts. Never write "no cut" or "seamless transition" — those aren\'t in the training vocabulary.',
|
|
81
|
-
'Multi-modal: up to 9 images, 3 videos and 3 audio references. Cite them by type and index — "Zhang San@Image 1", or the "Marcus (image 1)" form Slates composes from your @mentions. Never cite an asset ID instead of the image number; the model can\'t associate the two. Max length: 4,000 characters.',
|
|
100
|
+
'Multi-modal: up to 9 images, 3 videos and 3 audio references — 12 files in total, with the reference video capped at 15 seconds combined and the audio at 15. An audio reference on 2.0 needs at least one image or video alongside it (2.5 accepts audio on its own). Cite them by type and index — "Zhang San@Image 1", or the "Marcus (image 1)" form Slates composes from your @mentions. Never cite an asset ID instead of the image number; the model can\'t associate the two. Max length: 4,000 characters.',
|
|
101
|
+
'A reference VIDEO changes the price: it bills input seconds PLUS output seconds, summed across every clip attached. Two 5-second references on an 8-second generation bills 18 seconds, not 8. The Generate button and the duration menu both show that total before you commit. Over the cap is refused rather than trimmed, precisely so you are never charged for a clip the model never saw.',
|
|
82
102
|
'Don\'t cross-pollinate image-model syntax: named lenses, apertures and film stocks ("85mm f/1.4", "Kodak Portra 400") are a Nano Banana lever and a Seedance anti-pattern. Translate them into shot size, depth of field and colour tone instead.',
|
|
83
103
|
],
|
|
84
104
|
};
|
|
105
|
+
const SEEDANCE_25 = {
|
|
106
|
+
...SEEDANCE,
|
|
107
|
+
label: 'Seedance 2.5',
|
|
108
|
+
intro: [
|
|
109
|
+
'Seedance 2.5 is a SECOND SEAT next to 2.0, not an upgrade of it. It buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), and audio-only references — and it gives up 1080p and 4K entirely. It is 480p or 720p, on every route. Everything below about writing the prompt is the same as 2.0.',
|
|
110
|
+
"ByteDance's official advanced formula has 8 slots: precise subject + action details + scene/environment + lighting & color tone + camera movement + visual style + image quality + constraints. Sweet spot 60-150 words for a single shot, longer for multi-shot.",
|
|
111
|
+
],
|
|
112
|
+
columns: [
|
|
113
|
+
[
|
|
114
|
+
{
|
|
115
|
+
heading: 'Do not write edit instructions here',
|
|
116
|
+
example: '\u274c a wide shot of the workshop, remove the tripod\n\u2705 the workshop bench, clear and uncluttered',
|
|
117
|
+
note: 'With references attached, "add", "remove", "replace", "change", "extend" and "continue" make Seedance 2.5 treat the request as a video EDIT, and it then fails on constraints it never set — after the job has queued. Describe the finished frame instead. To actually edit a clip, attach it and pick Seedance 2.5 Edit.',
|
|
118
|
+
critical: true,
|
|
119
|
+
},
|
|
120
|
+
...SEEDANCE.columns[0],
|
|
121
|
+
],
|
|
122
|
+
[
|
|
123
|
+
{
|
|
124
|
+
heading: '720p is not the cheap one here',
|
|
125
|
+
example: '30s \u00b7 720p \u00b7 Face route = 484 credits\n15s \u00b7 1080p \u00b7 Seedance 2.0 Face = 411 credits',
|
|
126
|
+
note: 'Length is what moves the price, and 2.5 doubles the length ceiling — so a 30-second 720p clip can cost more than a 15-second 1080p one, against a 1,000-credit starting balance. Draft at 480p and 4-8 seconds; spend the length only on a take you already know works. The Generate button always shows the exact number first.',
|
|
127
|
+
critical: true,
|
|
128
|
+
},
|
|
129
|
+
{
|
|
130
|
+
heading: 'Audio-only references',
|
|
131
|
+
example: 'Reference the timbre in audio 1 to generate...',
|
|
132
|
+
note: '2.5 accepts an audio reference on its own — a voice line, a music bed, a room tone — with no image or video alongside it. 2.0 could not. Audio references never cost extra on any Seedance route.',
|
|
133
|
+
},
|
|
134
|
+
...SEEDANCE.columns[1],
|
|
135
|
+
],
|
|
136
|
+
],
|
|
137
|
+
footer: [
|
|
138
|
+
'30 image references is a budget, not a target. Every reference rule still holds: 2-4 strong references beat both extremes, one reference per role, one authoritative rendering per subject — and past 4 reference PEOPLE, output stability drops regardless of the cap. The larger budget is for long multi-shot takes and for video plus audio references alongside images.',
|
|
139
|
+
'A reference VIDEO bills input seconds PLUS output seconds, and 2.5 accepts references up to 30s combined — so a 20-second reference driving a 20-second output bills 40 seconds. The Generate button shows the total.',
|
|
140
|
+
...(SEEDANCE.footer ?? []).slice(0, 2),
|
|
141
|
+
'Frames and reference images stay mutually exclusive, and on a first/last-frame generation Seedance 2.5 chooses the aspect ratio itself — the ratio control shows "Adaptive" because the start frame decides the shape.',
|
|
142
|
+
],
|
|
143
|
+
};
|
|
144
|
+
const SEEDANCE_25_EDIT = {
|
|
145
|
+
...SEEDANCE_25,
|
|
146
|
+
label: 'Seedance 2.5 Edit',
|
|
147
|
+
intro: [
|
|
148
|
+
'Seedance 2.5 Edit changes an existing clip: attach the clip, describe only what should be different, and the original motion, framing and timing are kept. It is the only editor in Slates that takes a clip longer than 15 seconds — 4 to 30s, against Kling O3 Edit\'s 3-15s and Omni Flash Edit\'s 3-10s.',
|
|
149
|
+
'Output length and aspect ratio follow the SOURCE clip, so there is no duration or ratio control — the clip you attach is the quote. Output is 480p or 720p with native audio.',
|
|
150
|
+
],
|
|
151
|
+
columns: [
|
|
152
|
+
[
|
|
153
|
+
{
|
|
154
|
+
heading: 'Name the change, keep the rest',
|
|
155
|
+
example: 'Strictly edit the clip, and change the blue jacket to a red one.',
|
|
156
|
+
note: 'The clip already carries its composition, motion, timing and performance — re-describing them fights the model. One change per pass; chain passes for compound edits. Never write "reference the video" in an edit: that phrasing gets the request re-read as a fresh generation inspired by your clip instead of an edit of it.',
|
|
157
|
+
critical: true,
|
|
158
|
+
},
|
|
159
|
+
{
|
|
160
|
+
heading: 'Turn Face on when a face is visible',
|
|
161
|
+
note: 'The default provider blocks character faces outright — this is not a price optimisation, it is whether the job runs at all. There is no consented-real-face route for editing; real-person footage the Face route rejects has to go to Kling O3 Edit.',
|
|
162
|
+
critical: true,
|
|
163
|
+
},
|
|
164
|
+
{
|
|
165
|
+
heading: 'An edit costs about double a generation',
|
|
166
|
+
note: 'Every provider bills an edit on the input clip AND the output, so a 20-second edit is priced like 40 seconds of generation. Read the number on the Generate button rather than reasoning from the generation rate.',
|
|
167
|
+
},
|
|
168
|
+
],
|
|
169
|
+
[
|
|
170
|
+
{
|
|
171
|
+
heading: 'When to use it instead of the others',
|
|
172
|
+
note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner and the cheapest seat, and Kling O3 Edit is the one that takes subject and style reference images.',
|
|
173
|
+
},
|
|
174
|
+
{
|
|
175
|
+
heading: 'Prompt and clip only',
|
|
176
|
+
note: 'No character or style reference images on this engine. If the edit needs a reference image to lock an identity, that is Kling O3 Edit\'s job.',
|
|
177
|
+
},
|
|
178
|
+
],
|
|
179
|
+
],
|
|
180
|
+
footer: [
|
|
181
|
+
'Trim before you edit, not after: the bill is the source clip\'s length rounded up, so a 30-second clip you only needed 8 seconds of costs nearly four times what it had to.',
|
|
182
|
+
],
|
|
183
|
+
};
|
|
85
184
|
const KLING = {
|
|
86
185
|
label: 'Kling 3.0',
|
|
87
186
|
intro: [
|
|
@@ -360,7 +459,7 @@ const NANO_BANANA_LITE = {
|
|
|
360
459
|
],
|
|
361
460
|
};
|
|
362
461
|
// ── Audio lane ──────────────────────────────────────────────────
|
|
363
|
-
// The
|
|
462
|
+
// The two audio surfaces prompt NOTHING like the video models. The single
|
|
364
463
|
// most expensive mistake is bringing Kling's "SFX:" / "Ambient noise:" syntax
|
|
365
464
|
// to Seed Audio, which reads it as literal text. Every entry below leads with
|
|
366
465
|
// what the surface actually wants.
|
|
@@ -412,56 +511,11 @@ const SEED_AUDIO = {
|
|
|
412
511
|
},
|
|
413
512
|
{
|
|
414
513
|
heading: 'Know its seat',
|
|
415
|
-
note: 'Scenes, beds
|
|
514
|
+
note: 'Scenes, beds, room tone and dialogue in one pass. For a single effect that has to land on a specific frame, use Sound Effects.',
|
|
416
515
|
},
|
|
417
516
|
],
|
|
418
517
|
],
|
|
419
518
|
};
|
|
420
|
-
const ELEVEN_V3 = {
|
|
421
|
-
label: 'ElevenLabs Eleven v3',
|
|
422
|
-
intro: [
|
|
423
|
-
'Eleven v3 speaks your text verbatim in a named voice. It is the controlled, repeatable lane: the same text and the same voice give you a read you can regenerate after a script tweak without the performance drifting.',
|
|
424
|
-
'Billing is per 100 characters of text, rounded up — so tightening a sentence genuinely costs less, and a stray pasted paragraph genuinely costs more.',
|
|
425
|
-
],
|
|
426
|
-
columns: [
|
|
427
|
-
[
|
|
428
|
-
{
|
|
429
|
-
heading: 'The text field is the script',
|
|
430
|
-
example: '✗ (excited) Say this fast: Grab yours today!\n✓ Grab yours today!',
|
|
431
|
-
note: 'Everything you type gets spoken. Stage directions, character names and bracketed notes will be read out loud.',
|
|
432
|
-
critical: true,
|
|
433
|
-
},
|
|
434
|
-
{
|
|
435
|
-
heading: 'Punctuate for pace',
|
|
436
|
-
example: 'It works. Every time. · It works — every time…',
|
|
437
|
-
note: 'Full stops, commas, dashes and ellipses are your only timing controls. Rewrite the punctuation before you touch the settings.',
|
|
438
|
-
},
|
|
439
|
-
{
|
|
440
|
-
heading: 'Pick a voice and stay',
|
|
441
|
-
note: '20 preset voices. Choosing one per character or per piece is what makes a series sound deliberate; swapping voices mid-piece reads as a mistake.',
|
|
442
|
-
},
|
|
443
|
-
],
|
|
444
|
-
[
|
|
445
|
-
{
|
|
446
|
-
heading: 'Stability',
|
|
447
|
-
example: '0.3 emotive · 0.5 default · 0.8 steady',
|
|
448
|
-
note: 'Lower is more expressive and more variable take-to-take. Higher is flatter and more repeatable. Raise it for long narration, lower it for a single dramatic line.',
|
|
449
|
-
},
|
|
450
|
-
{
|
|
451
|
-
heading: 'Spell out the tricky bits',
|
|
452
|
-
example: 'SKU → "ess kay you" · 2026 → "twenty twenty six"',
|
|
453
|
-
note: 'Acronyms, product names, prices and years are where TTS embarrasses itself. Write the pronunciation you want.',
|
|
454
|
-
},
|
|
455
|
-
{
|
|
456
|
-
heading: 'Know its seat',
|
|
457
|
-
note: 'Exact words, repeatable voice, lines you will lip-sync against. Ambience, crowds and rooms belong to Seed Audio; one-shot effects to Sound Effects.',
|
|
458
|
-
},
|
|
459
|
-
],
|
|
460
|
-
],
|
|
461
|
-
footer: [
|
|
462
|
-
'Word-level timestamps come back with every generation at no extra cost — that is what a future caption pass will read, so there is no reason to turn them off.',
|
|
463
|
-
],
|
|
464
|
-
};
|
|
465
519
|
const ELEVEN_SFX = {
|
|
466
520
|
label: 'ElevenLabs Sound Effects',
|
|
467
521
|
intro: [
|
|
@@ -504,53 +558,10 @@ const ELEVEN_SFX = {
|
|
|
504
558
|
],
|
|
505
559
|
],
|
|
506
560
|
};
|
|
507
|
-
const SUNO = {
|
|
508
|
-
label: 'Suno',
|
|
509
|
-
intro: [
|
|
510
|
-
'Suno writes full music. Every generation returns TWO variations for one flat price, and length is free up to six minutes — so there is never a reason to generate a bed that is shorter than your edit.',
|
|
511
|
-
'The one thing to get right is which mode you are in, because it changes what the prompt field means.',
|
|
512
|
-
],
|
|
513
|
-
columns: [
|
|
514
|
-
[
|
|
515
|
-
{
|
|
516
|
-
heading: 'Description mode',
|
|
517
|
-
example: 'brooding synthwave for a night drive, analog bass, no vocals',
|
|
518
|
-
note: 'The prompt is a description (max 500 characters) and the lyrics get written for you. This is the fast path when you just need a mood.',
|
|
519
|
-
},
|
|
520
|
-
{
|
|
521
|
-
heading: 'Custom mode — the prompt IS the lyrics',
|
|
522
|
-
example: 'Style: dream pop, hazy\nTitle: Blue Hour\nPrompt: [Verse 1] The lights come on…',
|
|
523
|
-
note: 'In custom mode the prompt is sung exactly as written. Putting a description there gets your description sung back at you.',
|
|
524
|
-
critical: true,
|
|
525
|
-
},
|
|
526
|
-
{
|
|
527
|
-
heading: 'Instrumental',
|
|
528
|
-
note: 'Turn instrumental on to score a scene with no vocals — style and title still steer it, and the prompt field is ignored.',
|
|
529
|
-
},
|
|
530
|
-
],
|
|
531
|
-
[
|
|
532
|
-
{
|
|
533
|
-
heading: 'Steer with style, not adjectives',
|
|
534
|
-
example: 'Style: 90s trip-hop, dusty breakbeat, Rhodes\nAvoid: brass, EDM drops',
|
|
535
|
-
note: 'Genre, era, instrumentation and tempo belong in the style field. Negative tags remove what keeps creeping in.',
|
|
536
|
-
},
|
|
537
|
-
{
|
|
538
|
-
heading: 'Length is free',
|
|
539
|
-
example: '10–360 seconds',
|
|
540
|
-
note: 'A six-minute track costs exactly what a default one does. Ask for longer than the cut needs and trim on the timeline.',
|
|
541
|
-
},
|
|
542
|
-
{
|
|
543
|
-
heading: 'Two songs, both yours',
|
|
544
|
-
note: 'Both variations land in the gallery. They are genuinely different takes on the same brief — audition both before re-rolling.',
|
|
545
|
-
},
|
|
546
|
-
],
|
|
547
|
-
],
|
|
548
|
-
footer: [
|
|
549
|
-
'Files are hosted by the provider for a limited window — Slates downloads and stores them on your machine as soon as the track finishes, so nothing expires out from under a project.',
|
|
550
|
-
],
|
|
551
|
-
};
|
|
552
561
|
export const PROMPTING_TIPS = {
|
|
553
562
|
seedance: SEEDANCE,
|
|
563
|
+
'seedance-2-5': SEEDANCE_25,
|
|
564
|
+
'seedance-2-5-edit': SEEDANCE_25_EDIT,
|
|
554
565
|
kling: KLING,
|
|
555
566
|
'kling-edit': KLING_EDIT,
|
|
556
567
|
veo: VEO,
|
|
@@ -559,9 +570,7 @@ export const PROMPTING_TIPS = {
|
|
|
559
570
|
'nano-banana': NANO_BANANA,
|
|
560
571
|
'nano-banana-lite': NANO_BANANA_LITE,
|
|
561
572
|
'seed-audio': SEED_AUDIO,
|
|
562
|
-
'eleven-v3': ELEVEN_V3,
|
|
563
573
|
'eleven-sfx': ELEVEN_SFX,
|
|
564
|
-
suno: SUNO,
|
|
565
574
|
};
|
|
566
575
|
/** Null when no tips exist for the key — callers render an honest fallback. */
|
|
567
576
|
export function getPromptingTips(key) {
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
export type ReferenceKind = 'character' | 'environment' | 'style' | 'pinned' | 'first-frame' | 'last-frame' | 'video';
|
|
1
|
+
export type ReferenceKind = 'character' | 'environment' | 'style' | 'pinned' | 'first-frame' | 'last-frame' | 'video' | 'video-ref' | 'audio-ref';
|
|
2
2
|
export interface ReferenceMedia {
|
|
3
3
|
path: string;
|
|
4
|
-
mediaKind: 'image' | 'video';
|
|
4
|
+
mediaKind: 'image' | 'video' | 'audio';
|
|
5
5
|
}
|
|
6
6
|
/**
|
|
7
7
|
* A named bucket of reference media that can be @mentioned. The whole definition
|
|
@@ -22,8 +22,25 @@ export interface ComposedReferences {
|
|
|
22
22
|
prompt: string;
|
|
23
23
|
/** Free-reference image paths in cited order — flatten yields "image 1..N". */
|
|
24
24
|
orderedImagePaths: string[];
|
|
25
|
-
/** Video
|
|
25
|
+
/** Video paths in cited order — "video 1..M". Edit sources and reference
|
|
26
|
+
* clips SHARE this list and one counter, so a mixed state can never emit
|
|
27
|
+
* two "Video 1"s. */
|
|
26
28
|
orderedVideoPaths: string[];
|
|
29
|
+
/** Reference audio paths in cited order — "audio 1..K". */
|
|
30
|
+
orderedAudioPaths: string[];
|
|
31
|
+
/**
|
|
32
|
+
* Tokens written in the prompt that matched NO reference group, as authored
|
|
33
|
+
* (`'#noir'`, `'@bob'`), first-appearance order, deduped case-insensitively.
|
|
34
|
+
*
|
|
35
|
+
* 🚨 THIS FIELD EXISTS BECAUSE THE ALTERNATIVE IS A SILENT EDIT. A `#tag` with
|
|
36
|
+
* no matching style is DELETED from the text — a raw tag confuses every model,
|
|
37
|
+
* so removing it is right, but removing it without saying so changes what the
|
|
38
|
+
* user asked for behind their back. Prompt-transparency doctrine (slate
|
|
39
|
+
* `CLAUDE.md`) requires the surface to be able to say "this went nowhere", and
|
|
40
|
+
* this is the only record that the token was ever there. Callers that render a
|
|
41
|
+
* composed-prompt preview MUST surface it.
|
|
42
|
+
*/
|
|
43
|
+
unresolvedTokens: string[];
|
|
27
44
|
}
|
|
28
45
|
/**
|
|
29
46
|
* Compose the raw prompt (mentions intact) + an ORDERED list of reference groups
|
|
@@ -41,6 +58,7 @@ export interface ComposedReferences {
|
|
|
41
58
|
export interface ComposeOptions {
|
|
42
59
|
startImageNumber?: number;
|
|
43
60
|
startVideoNumber?: number;
|
|
61
|
+
startAudioNumber?: number;
|
|
44
62
|
}
|
|
45
63
|
export declare function composeReferences(rawPrompt: string, groups: ReferenceGroup[], opts?: ComposeOptions): ComposedReferences;
|
|
46
64
|
export interface KlingEditElement {
|