@slatesvideo/shared 0.6.10 → 0.6.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.js +3 -0
- package/dist/manual/content.d.ts +1 -1
- package/dist/manual/content.js +1 -1
- package/dist/operations/index.js +13 -12
- package/dist/prompts/model-facts.d.ts +27 -0
- package/dist/prompts/model-facts.js +58 -13
- package/dist/prompts/prompting-tips.js +2 -2
- package/dist/skills/content.js +3 -3
- package/dist/update-check.d.ts +22 -0
- package/dist/update-check.js +109 -0
- package/exports/slates-prompt-builder/generated/SKILL.md +2 -2
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +5 -5
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +2 -2
- package/skills/slates-model-selection.md +133 -133
- package/skills/slates-one-prompt-film.md +96 -95
- package/skills/slates-prompting-seedance-2-5.md +5 -6
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export declare const UPDATE_CACHE_FILE: string;
|
|
2
|
+
/** Numeric semver compare on the release segments; a prerelease suffix is ignored. */
|
|
3
|
+
export declare function compareVersions(a: string, b: string): number;
|
|
4
|
+
/** The newest version the last registry lookup recorded for `pkgName`. Sync, disk only. */
|
|
5
|
+
export declare function cachedLatestVersion(pkgName: string, cacheFile?: string): string | null;
|
|
6
|
+
/**
|
|
7
|
+
* Refresh the cached latest version from the npm registry when the entry is
|
|
8
|
+
* older than an hour. Bounded by FETCH_TIMEOUT_MS, never throws. Returns the
|
|
9
|
+
* latest version known after the call (fresh or cached), or null.
|
|
10
|
+
*/
|
|
11
|
+
export declare function refreshLatestVersion(pkgName: string, options?: {
|
|
12
|
+
cacheFile?: string;
|
|
13
|
+
registryUrl?: string;
|
|
14
|
+
now?: number;
|
|
15
|
+
}): Promise<string | null>;
|
|
16
|
+
/**
|
|
17
|
+
* One sentence when `current` is behind `latest`, else null. `howToUpdate` is
|
|
18
|
+
* the surface-specific action: the MCP server's is a client restart, the CLI's
|
|
19
|
+
* is a reinstall.
|
|
20
|
+
*/
|
|
21
|
+
export declare function updateNotice(pkgName: string, current: string, latest: string | null, howToUpdate: string): string | null;
|
|
22
|
+
//# sourceMappingURL=update-check.d.ts.map
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
// Version handshake for the two published entry points (@slatesvideo/mcp-server
|
|
2
|
+
// and @slatesvideo/cli).
|
|
3
|
+
//
|
|
4
|
+
// WHY THIS EXISTS (2026-09-13): a Discord user's agent reported "H3 isn't in
|
|
5
|
+
// the API tool, only Kling, Veo and Seedance are exposed" — a server older
|
|
6
|
+
// than 0.5.10 (2026-08-28). Nothing told him, and nothing told his agent. The
|
|
7
|
+
// install path is `npx -y @slatesvideo/mcp-server` with no version pinned, so
|
|
8
|
+
// a client RESTART is the whole update; the failure is a client that never
|
|
9
|
+
// restarts, a global install nobody re-runs, or skill files copied months ago.
|
|
10
|
+
//
|
|
11
|
+
// HOW IT WORKS. One small registry GET (`/<pkg>/latest`), cached on disk in
|
|
12
|
+
// ~/.slates/update-check.json and refreshed at most once an hour. Readers are
|
|
13
|
+
// SYNCHRONOUS and disk-only so the MCP server can put the notice into its
|
|
14
|
+
// `instructions` at construction time without delaying startup; the refresh
|
|
15
|
+
// runs in the background for the NEXT launch. The CLI awaits the refresh
|
|
16
|
+
// (bounded by FETCH_TIMEOUT_MS) because a CLI turn already pays for a network
|
|
17
|
+
// call. Every path is fail-silent: no network, no home dir, no registry —
|
|
18
|
+
// no notice, never an error.
|
|
19
|
+
import { mkdirSync, readFileSync, writeFileSync } from 'node:fs';
|
|
20
|
+
import { join } from 'node:path';
|
|
21
|
+
import { AGENT_DIR } from './auth.js';
|
|
22
|
+
export const UPDATE_CACHE_FILE = join(AGENT_DIR, 'update-check.json');
|
|
23
|
+
const REFRESH_AFTER_MS = 60 * 60 * 1000;
|
|
24
|
+
const FETCH_TIMEOUT_MS = 2500;
|
|
25
|
+
/** Numeric semver compare on the release segments; a prerelease suffix is ignored. */
|
|
26
|
+
export function compareVersions(a, b) {
|
|
27
|
+
const parse = (v) => v.replace(/^v/, '').split('-')[0].split('.').map((n) => Number.parseInt(n, 10) || 0);
|
|
28
|
+
const pa = parse(a);
|
|
29
|
+
const pb = parse(b);
|
|
30
|
+
for (let i = 0; i < Math.max(pa.length, pb.length); i++) {
|
|
31
|
+
const d = (pa[i] ?? 0) - (pb[i] ?? 0);
|
|
32
|
+
if (d !== 0)
|
|
33
|
+
return d;
|
|
34
|
+
}
|
|
35
|
+
return 0;
|
|
36
|
+
}
|
|
37
|
+
function readCache(file) {
|
|
38
|
+
try {
|
|
39
|
+
const parsed = JSON.parse(readFileSync(file, 'utf8'));
|
|
40
|
+
return parsed && typeof parsed === 'object' ? parsed : {};
|
|
41
|
+
}
|
|
42
|
+
catch {
|
|
43
|
+
return {};
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
function writeCache(file, cache) {
|
|
47
|
+
try {
|
|
48
|
+
mkdirSync(join(file, '..'), { recursive: true });
|
|
49
|
+
writeFileSync(file, JSON.stringify(cache, null, 2) + '\n', 'utf8');
|
|
50
|
+
}
|
|
51
|
+
catch {
|
|
52
|
+
// A read-only home is not a reason to fail a generation.
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
/** The newest version the last registry lookup recorded for `pkgName`. Sync, disk only. */
|
|
56
|
+
export function cachedLatestVersion(pkgName, cacheFile = UPDATE_CACHE_FILE) {
|
|
57
|
+
const entry = readCache(cacheFile)[pkgName];
|
|
58
|
+
return entry && typeof entry.latest === 'string' ? entry.latest : null;
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Refresh the cached latest version from the npm registry when the entry is
|
|
62
|
+
* older than an hour. Bounded by FETCH_TIMEOUT_MS, never throws. Returns the
|
|
63
|
+
* latest version known after the call (fresh or cached), or null.
|
|
64
|
+
*/
|
|
65
|
+
export async function refreshLatestVersion(pkgName, options = {}) {
|
|
66
|
+
const cacheFile = options.cacheFile ?? UPDATE_CACHE_FILE;
|
|
67
|
+
const now = options.now ?? Date.now();
|
|
68
|
+
const cache = readCache(cacheFile);
|
|
69
|
+
const entry = cache[pkgName];
|
|
70
|
+
if (entry && now - entry.checkedAt < REFRESH_AFTER_MS)
|
|
71
|
+
return entry.latest;
|
|
72
|
+
const registryUrl = options.registryUrl ?? 'https://registry.npmjs.org';
|
|
73
|
+
const controller = new AbortController();
|
|
74
|
+
const timer = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS);
|
|
75
|
+
// A stdio server exits when its client closes; a pending timer must not
|
|
76
|
+
// hold the process open for the rest of the timeout.
|
|
77
|
+
timer.unref?.();
|
|
78
|
+
try {
|
|
79
|
+
const res = await fetch(`${registryUrl}/${pkgName}/latest`, {
|
|
80
|
+
signal: controller.signal,
|
|
81
|
+
headers: { accept: 'application/json' },
|
|
82
|
+
});
|
|
83
|
+
if (!res.ok)
|
|
84
|
+
return entry?.latest ?? null;
|
|
85
|
+
const body = (await res.json());
|
|
86
|
+
if (typeof body.version !== 'string')
|
|
87
|
+
return entry?.latest ?? null;
|
|
88
|
+
cache[pkgName] = { latest: body.version, checkedAt: now };
|
|
89
|
+
writeCache(cacheFile, cache);
|
|
90
|
+
return body.version;
|
|
91
|
+
}
|
|
92
|
+
catch {
|
|
93
|
+
return entry?.latest ?? null;
|
|
94
|
+
}
|
|
95
|
+
finally {
|
|
96
|
+
clearTimeout(timer);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
/**
|
|
100
|
+
* One sentence when `current` is behind `latest`, else null. `howToUpdate` is
|
|
101
|
+
* the surface-specific action: the MCP server's is a client restart, the CLI's
|
|
102
|
+
* is a reinstall.
|
|
103
|
+
*/
|
|
104
|
+
export function updateNotice(pkgName, current, latest, howToUpdate) {
|
|
105
|
+
if (!latest || compareVersions(current, latest) >= 0)
|
|
106
|
+
return null;
|
|
107
|
+
return `UPDATE AVAILABLE: ${pkgName} v${current} is running; v${latest} is published. ${howToUpdate}`;
|
|
108
|
+
}
|
|
109
|
+
//# sourceMappingURL=update-check.js.map
|
|
@@ -16,8 +16,8 @@ This portable skill is deliberately thin. Its reference files are generated dire
|
|
|
16
16
|
<!-- @generated:model-routing -->
|
|
17
17
|
| Model | Canonical route | Guide |
|
|
18
18
|
|---|---|---|
|
|
19
|
-
| **Kling 3.0** |
|
|
20
|
-
| **Seedance 2.0** |
|
|
19
|
+
| **Kling 3.0** | THE COST-EFFECTIVE SEAT — strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync and the widest aspect-ratio set; pick it when the budget matters and the shot is a performance or a start-frame animation. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools. | `reference-kling.md` |
|
|
20
|
+
| **Seedance 2.0** | THE 4K AND VALUE SEAT beside the 2.5 default — the only Seedance with native 4K (Pro-gated; base accounts get PRO_REQUIRED) and cheaper than 2.5 at every resolution they share, with the same physics, effects and scale strengths; shorter takes, fewer references, no timestamps. VIDEO-ONLY. A bare "seedance" still resolves here for older CLIs that expect 4K. | `reference-seedance.md` |
|
|
21
21
|
| **Nano Banana 2 (Gemini 3.1 Flash Image)** | DEFAULT image model and the all-rounder — route here unless another seat's speciality is the point. Best start-frame for legible in-scene text. Knowledge cutoff Jan 2025: anything later needs reference images. | `reference-nano-banana.md` |
|
|
22
22
|
<!-- @end:model-routing -->
|
|
23
23
|
|
|
@@ -28,14 +28,14 @@
|
|
|
28
28
|
},
|
|
29
29
|
{
|
|
30
30
|
"path": "src/prompts/model-facts.ts",
|
|
31
|
-
"sha256": "
|
|
31
|
+
"sha256": "4df3ac4b003ed29c2e1992a34519e01957e5292211a294bb2af0045daf9d4194"
|
|
32
32
|
}
|
|
33
33
|
],
|
|
34
34
|
"outputs": [
|
|
35
35
|
{
|
|
36
36
|
"path": "SKILL.md",
|
|
37
|
-
"bytes":
|
|
38
|
-
"sha256": "
|
|
37
|
+
"bytes": 4622,
|
|
38
|
+
"sha256": "0c3f1982796fe668824067b06e8d76070aaae2d00d47a978ef5bd8b489b2a3da"
|
|
39
39
|
},
|
|
40
40
|
{
|
|
41
41
|
"path": "reference-character.md",
|
|
@@ -65,8 +65,8 @@
|
|
|
65
65
|
],
|
|
66
66
|
"archive": {
|
|
67
67
|
"path": "slates-prompt-builder.skill",
|
|
68
|
-
"bytes":
|
|
69
|
-
"sha256": "
|
|
68
|
+
"bytes": 40546,
|
|
69
|
+
"sha256": "0760799f957aac866590768dd00bc74cd4c32525364455c0eab96dba334c4de4",
|
|
70
70
|
"entries": [
|
|
71
71
|
"SKILL.md",
|
|
72
72
|
"reference-character.md",
|
|
Binary file
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@slatesvideo/shared",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.11",
|
|
4
4
|
"description": "Shared operations layer for the Slates MCP server and CLI: auth, cloud/desktop clients, and the single tool surface both consume. Most users want @slatesvideo/mcp-server or @slatesvideo/cli instead.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"sync-partials": "node scripts/sync-partials.mjs",
|
|
44
44
|
"build-prompt-builder": "node scripts/build-prompt-builder.mjs",
|
|
45
45
|
"check-prompt-builder": "node scripts/build-prompt-builder.mjs --check",
|
|
46
|
-
"build": "node scripts/sync-partials.mjs --check && node scripts/build-prompt-builder.mjs --check && node scripts/embed-skills.mjs && tsc && node scripts/render-capability-partials.mjs --check",
|
|
46
|
+
"build": "node scripts/sync-partials.mjs --check && node scripts/build-prompt-builder.mjs --check && node scripts/embed-skills.mjs && tsc && node scripts/render-capability-partials.mjs --check && node scripts/update-check-check.mjs",
|
|
47
47
|
"typecheck": "node scripts/sync-partials.mjs --check && node scripts/build-prompt-builder.mjs --check && node scripts/embed-skills.mjs && tsc --noEmit",
|
|
48
48
|
"prepublishOnly": "npm run build",
|
|
49
49
|
"render-partials": "node scripts/render-capability-partials.mjs"
|
|
@@ -1,133 +1,133 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: slates-model-selection
|
|
3
|
-
description: Which model to pick for a given job — the routing doctrine. Read BEFORE choosing any video or image model, before quoting a plan, and before defaulting anywhere.
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Model selection — the routing doctrine
|
|
7
|
-
|
|
8
|
-
Pick the model FIRST, deliberately, before writing a prompt or quoting a plan. Model routing is a core part of the intelligence users are paying for: the agent knows what each model is good at and which ones underperform for a job — defaulting to the wrong model burns the user's credits on a weaker result.
|
|
9
|
-
|
|
10
|
-
## 🔑 The meta-rule — above the table
|
|
11
|
-
|
|
12
|
-
The tables below are a snapshot. This roster churns constantly (NB2 Lite, Omni Flash, Seedream 5 Lite, GPT Image 2.5 all landed recently) — **a table rots; a rule doesn't.** When the tables and this rule disagree, or when a model appears that the tables don't cover, run the rule:
|
|
13
|
-
|
|
14
|
-
> **Name ONE must-preserve requirement for the shot.** Not a vibe — the single thing that, if it breaks, makes the shot unusable: this face stays this face · the fluid behaves like fluid · the text stays legible · the take stays one unbroken move.
|
|
15
|
-
>
|
|
16
|
-
> **Inspect the output at its intended crop.** A frame that holds up as a thumbnail can fall apart at the size it will actually be watched. For a location, look at atmosphere, material texture, and anchor objects; for a character, identity, skin, pose, and gradients.
|
|
17
|
-
>
|
|
18
|
-
> **Choose the model that PROVES that requirement** and leaves only failures you can afford to rerun or mask.
|
|
19
|
-
>
|
|
20
|
-
> **When the roster changes, repeat the evidence test.** Do not carry today's ranking forward on reputation.
|
|
21
|
-
|
|
22
|
-
## Video routing
|
|
23
|
-
|
|
24
|
-
| Job | Model | Why |
|
|
25
|
-
|---|---|---|
|
|
26
|
-
| **General-purpose — the default for most shots** | **
|
|
27
|
-
|
|
|
28
|
-
|
|
|
29
|
-
|
|
|
30
|
-
|
|
|
31
|
-
| **One take longer than 15 seconds**,
|
|
32
|
-
| **The SOUND has to be directed, not just present** — a specific line delivered a specific way, scene sound that has to sit under it, and score that must stay out of the characters' world | **MiniMax H3** | The only seat where audio is authored in three separate layers in ONE pass (synchronised events in the body, ambience in a soundscape section, audience-only score in its own) rather than toggled on. 5–15s, 480p / 768p / 2K / 4K, 24fps, 32kHz stereo, 11 languages. Rules in `slates-prompting-minimax-h3`. |
|
|
33
|
-
| **A reference has to keep a DECLARED amount of itself** — especially moving one subject's characteristic onto a *different* subject | **MiniMax H3** | The only seat that understands a stated retention relationship (kept whole / kept in part / transferred onto another subject / loose echo). 9 images + 3 video + 3 audio, 12 files total. 🚨 The first 5 reference images are free and every one after that costs 4 credits — pass `referenceImages` to `slates_estimate_generation_cost` before a reference-heavy job. |
|
|
34
|
-
| **Turnaround is the requirement** on a text-to-video or start-frame shot at 480p/768p | **MiniMax H3 Max** | fal's self-hosted post-train of H3. **Measured 2026-08-27: a 5s 768p clip finished in 4.8s against 57s on base H3 — about 12x faster**, same prompt, queue to file. When turnaround is the requirement this is not a marginal win. 🚨 It is the PREMIUM seat, not a cheap H3 — $0.080/s at 768p against base H3's $0.060/s, 33% more, and it tops out at 768p. It still animates a start frame and an end frame — image-to-video is one of the two things it is for — and since 2026-09-09 it takes the full omni-reference set too (9 images + 3 video + 3 audio), so the seats now differ on ladder and price rather than on what they accept. Never the default; never reach for it to save money. |
|
|
35
|
-
| Native synchronized audio (dialogue + SFX generated WITH the video in one gen), 16:9, ≤8s | Veo 3.1 | Narrow, and now narrower: if the sound needs DIRECTING rather than merely existing, MiniMax H3 is the better seat. |
|
|
36
|
-
|
|
37
|
-
### Named Seedance escalation triggers
|
|
38
|
-
|
|
39
|
-
"Physics matter" is an abstract category and it under-fires. These are the beats Seedance is **observably** good at — if the shot contains one, escalate without deliberating:
|
|
40
|
-
|
|
41
|
-
- **Real-time → slow-motion contrast.** The signature beat; nearly every strong clip rides it.
|
|
42
|
-
- **The camera moving while debris, meteors, sparks or particles crash around the subject.** Distinctly a feature of this model, not just a thing it survives.
|
|
43
|
-
- **Massive scale that has to read as genuinely huge** — not "a big thing", a thing whose size is the point of the shot.
|
|
44
|
-
- **One continuous unbroken take.**
|
|
45
|
-
|
|
46
|
-
Concrete beats route better than an abstract category. Cost stays a tiebreaker, never the router (see below).
|
|
47
|
-
|
|
48
|
-
## Video EDIT routing (changing an existing clip)
|
|
49
|
-
|
|
50
|
-
| Job | Tool | Why |
|
|
51
|
-
|---|---|---|
|
|
52
|
-
| **Footage-synced VFX on real footage** — add/remove an effect, prop, or lighting change while the take stays the take (incl. talking heads) | **Omni Flash Edit** (`slates_edit_video`, `omni-flash-edit`) | **The edit-fidelity winner** (head-to-head receipt 2026-07-09, WITH a short prompt): lip movement held perfectly, audio near-identical, effect landed and released on cue — where Kling missed an action beat and drifted lips. Prompt-only, 3–10s clips, 720p out, ~6.4 cr/s (cheapest). Quirk: occasional tail jitter / doubled final speech beat — trim the tail on the timeline. Fidelity is EARNED by prompt discipline: one short line + "Keep everything else the same"; long prompts destroy it (see below). |
|
|
53
|
-
| **Identity swap needing reference images** — put @marcus into the clip, lock a style from refs | **Kling O3 Edit** (`slates_edit_video`) | The only edit engine that takes element/style reference images (frontal + angles lock identity). ~19¢/s. |
|
|
54
|
-
| **Spoken words must be bit-exact** (VO, legal copy, music) | **Kling O3 Edit** with `keepAudio` (default true) — or segment-splice | Kling keeps the ORIGINAL audio track verbatim — but re-synthesizes the video, so lips can drift slightly against it (7/09 receipt). Omni Flash regenerates audio (voice editing unsupported): on the 7/09 receipt it came back near-identical with perfect lips, but "near-identical" is not a guarantee. Zero-risk path for critical audio: segment-splice — edit only the non-talking seconds and keep the original track under the cut. |
|
|
55
|
-
| Style-transfer-heavy re-imagining, full relocate of the scene, or edit quality worth a premium at 1080p+ | Seedance edit/relocate (`videoReferenceAssetId` on `slates_generate_video`) | Seedance's strength is transfer intensity; it re-generates rather than surgically edits. Head-to-head receipt 2026-07-09 (photoreal-insert job, same clip): at 720p it LOST to Omni Flash edit on result while costing ~3× (vref bills input+output seconds; face-lane rates when people are in frame). Route here for its strengths or at 1080p/4K where its ceiling is higher — never as the cheap default. (2.5's relocate lane reaches 1080p too as of 2026-08-24, at $0.2457/s of combined input+output.) Takes long descriptive prompts fine (no Omni-style hard-fail on timing phrasing). |
|
|
56
|
-
| **A clip LONGER THAN 15 SECONDS** | **Seedance 2.5 Edit** (`slates_edit_video`, `seedance-2.5-edit`) | The only edit engine that takes a 4–30s clip — length is the whole reason to route here. 480p/720p/1080p out, native audio, prompt + clip only (no reference images). Output length AND aspect ratio follow the source, so the billed key is the ceiled source length; an edit bills roughly DOUBLE a plain 2.5 generation of the same length because every provider charges an edit on input + output seconds. Set `seedanceFace: true` when a face is visible — the faceless provider blocks faces outright. No consented-real-face route for editing. Inside 15s, choose on fidelity instead. |
|
|
57
|
-
| AI-edit the user's OWN footage | Omni Flash Edit (3–10s), Kling O3 Edit (3–15s, 720–3840px) or Seedance 2.5 Edit (4–30s) | Both take any MP4/MOV — not just Slates gens. Phone footage MUST be rotation-normalized first (players honor the rotation flag; models don't — raw portrait phone clips come back SIDEWAYS). |
|
|
58
|
-
|
|
59
|
-
- **Edit before re-roll.** A re-roll gambles away the parts the user already likes; an edit changes only what the prompt names. Quote the edit first when a clip is mostly right.
|
|
60
|
-
- **Ship via segment-splice.** Every edit model re-synthesizes the whole clip, so fidelity risk scales with clip length. For real deliverables: trim out ONLY the seconds where the change happens, edit that segment, splice it back over the original on the timeline with the ORIGINAL audio underneath. Most of the final video stays the untouched original — that's how the polished split-screen demos going around actually work, plus gesture-only beats with voiceover laid over in post.
|
|
61
|
-
- **One change per pass, short prompts.** On Omni Flash this is documented law ("overly descriptive prompts can lead to unintended changes" — long identity-lock preambles make drift WORSE, receipt 7/09); on Kling multi-beat instructions get dropped. Chain passes instead.
|
|
62
|
-
- Edited clips are themselves editable clips — chain passes; lineage links each output to its parent.
|
|
63
|
-
|
|
64
|
-
## Motion Transfer & Lip Sync routing (Kling-only tools)
|
|
65
|
-
|
|
66
|
-
Both tools are **Kling-only**. Every entry in them is a real Kling endpoint that bolts motion or lip movement onto a finished source as a dedicated post-process.
|
|
67
|
-
|
|
68
|
-
| Job | Tool | Why |
|
|
69
|
-
|---|---|---|
|
|
70
|
-
| Motion retarget onto a still character | Kling MC std/pro (`slates_generate_motion_transfer`) | Structured skeleton/depth retarget, ~32–42 credits / 5s, takes up to 30s driving clips. |
|
|
71
|
-
| Re-voice a clip, or animate a still portrait | Kling lip-sync / avatar (`slates_generate_lip_sync`) | ~4–29 credits / 5s blocks. |
|
|
72
|
-
|
|
73
|
-
**Want the Seedance version of either?** It is not a switch on these tools — it is a normal `slates_generate_video` on `seedance-2` with the clip attached as a **video reference** and the motion or dialogue written into the prompt ("the character from image 1 performs the exact motion from video 1"). That routes to the same endpoint the tool would have called, with the prompt visible and editable instead of ghost-written. Single-pass conditioning genuinely beats post-hoc retargeting on fast choreography, contact, cloth and hair — and it carries native audio — so escalate there whenever fidelity matters.
|
|
74
|
-
|
|
75
|
-
- Seedance video-reference gens bill COMBINED input+output seconds (`seedance-2*-vref-*` keys) — pass the clip duration and quote before confirming. Driving clips must be 2–15s on Seedance 2.0 and up to 30s on 2.5; past that it is Kling MC's lane.
|
|
76
|
-
- Faces on that route go through the normal cascade: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → `seedanceRealFace` + `realFaceConsent` for a real person (premium realface pricing).
|
|
77
|
-
|
|
78
|
-
**Rules:**
|
|
79
|
-
|
|
80
|
-
- **Default video = Kling 3.0 std
|
|
81
|
-
- **Veo is never the default.** 16:9 or 9:16 only, 4/6/8s only (and 8s only at 1080p/4K, or with reference images), and it is not the quality pick — treat it as a single-purpose tool for native-synced-audio shots. If audio can be added after (Kling lip-sync, edit stage), prefer Kling or Seedance + audio in post.
|
|
82
|
-
- **9:16 vertical → Kling or Seedance by preference**, not by necessity: Veo does take 9:16 on the route Slates uses. Route away from it because it is the niche seat, not because it can't.
|
|
83
|
-
- **Ratios and durations are enforced before submit.** `slates_generate_video` validates the aspect ratio, resolution and duration against the model you picked and refuses out-of-set values with the legal list — it will not silently ignore or downgrade them. The authoritative per-model sets are in the op's own param descriptions, which are generated from the capability SSOT; prefer those over any list written in prose here.
|
|
84
|
-
- **Image-to-video from an NB2 start frame** (the standard pipeline) →
|
|
85
|
-
- **User names a model explicitly → use it.** But if it's a mismatch for the job (crazy physics on Kling std, a 30s take on anything but Seedance 2.5, 4K on Seedance 2.5 which has none), say so in one line and offer the right route before generating.
|
|
86
|
-
|
|
87
|
-
## Image routing
|
|
88
|
-
|
|
89
|
-
**Video models (Kling, Seedance, Veo) cannot generate standalone images — ever.** A "premium hero reference image" is still an image job: it routes to an image model below, never to Seedance.
|
|
90
|
-
|
|
91
|
-
- **Default: Nano Banana 2** — strongest reference HANDLING (14 refs; GPT Image now takes more, at 16, but Banana is still the one that holds many subjects coherently), best legible text, the standard start-frame generator.
|
|
92
|
-
- **NB2 Lite** — the fast/draft seat: ~half NB2's price, ~2.7× faster, 1K only. Route iteration volume and drafts here; finals go back to NB2 full (2K/4K).
|
|
93
|
-
- **Nano Banana Pro** — the hero-frame/typography ceiling (~2× NB2). NB2 ≈ 95% of Pro; escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element frames must be perfect. Up to 14 refs — feed it a full subject library.
|
|
94
|
-
- **GPT Image 2.5** — two seats, `gpt-image-2-5-flare` and `gpt-image-2-5-sunburst`, **same price**. Readable text / panels / UI king: character sheets, shot grids, diagrams, text-bearing panels. **Also the photoreal front-runner (Eric, 2026-08-24)** — it beat both Nano Banana rails head-to-head on skin realism, which is why the AI-influencer ad lane generates every plate on this line. **The seat split is SPEED vs QUALITY, not generate vs edit** (OpenAI's own rule): Flare is the small, fast model with quality *comparable to* GPT Image 2 — drafts, exploration, volume; Sunburst is OpenAI's *most capable* image model, higher quality than GPT Image 2, deliberately slower — finals, hero frames, photoreal, and multi-reference edits, where its lead is widest. **Explore on Flare, finish on Sunburst.** Five quality tiers, cheapest first — `low` (layout checks only), `medium` (drafts), **`high` (the default)**, `xhigh`, `max` (the top). Uneven: `max` is 4× `high`, `xhigh` only ~1.8× it. **16 reference images**, the schema ceiling. **Transparent backgrounds** via `backgroundMode` — free, and the only image family that offers them.
|
|
95
|
-
|
|
96
|
-
🚨 **The tier names moved when 2.5 replaced GPT Image 2, and the strings did not.** GPT Image 2's `medium` is 2.5's `high`; its `high` is 2.5's `max` — same money, one rung of renaming. The 2026-08-24 photoreal result was measured at GPT Image 2 `high`, so **the tier that reproduces it is `max`**. Nobody has re-run it on 2.5; the ranking is inherited, not re-measured.
|
|
97
|
-
- **FLUX.2 Max** — photoreal texture, hex-color binding, typography, less censored.
|
|
98
|
-
- **Seedream 5 Lite** — uncensored + any-resolution flat price; volume exploration when the Gemini filter is in the way.
|
|
99
|
-
|
|
100
|
-
**Split rule of thumb:** readable text / panels / UI → GPT Image 2.5 (Flare to explore, Sunburst to finish); **photoreal people, finals and hero frames → Sunburst at `max`** — the 2026-08-24 result was measured at GPT Image 2's `high`, which is `max` here, and Flare only *matches* GPT Image 2 while Sunburst exceeds it; multi-reference edits where several references must all survive into one frame → Sunburst; edit-heavy work → the Banana line; drafts → GPT Image 2.5 Flare at `medium`, which now undercuts NB2 Lite on both price and resolution; uncensored or odd resolutions → Seedream/FLUX.
|
|
101
|
-
|
|
102
|
-
⚠️ **This line said the opposite until 2026-08-24** — it sent photoreal *away* from GPT Image on reputation, which is the exact failure § The meta-rule above warns about. Re-run the evidence test when the roster moves. It moved again on 2026-09-09, and the ranking was carried across rather than re-measured — exactly what the meta-rule says not to trust. Treat it as a starting hypothesis for 2.5, not a receipt. **The seat choice above is likewise reasoned from OpenAI's positioning, not measured:** run Flare-`max` against Sunburst-`max` on one plate and write the answer into `slates-prompting-gpt-image-2-5`.
|
|
103
|
-
|
|
104
|
-
## Audio routing
|
|
105
|
-
|
|
106
|
-
**Image and video models cannot generate standalone audio, and neither audio model can generate images or video.** A shot that needs synced audio generated WITH the picture is still a video job (Kling omni / Veo / Omni Flash / Seedance all carry native audio); the models below produce audio *as its own asset*, to lay on the timeline.
|
|
107
|
-
|
|
108
|
-
| Job | Model | Why |
|
|
109
|
-
|---|---|---|
|
|
110
|
-
| **Default — a whole audio scene in one pass**: room tone, ambience beds, crowds, nature, layered dialogue + effects, spoken lines inside a scene | **Seed Audio 1.0** (`seed-audio`) | One plain sentence in, a complete scene out. The continuity-bed workhorse; dialogue is performed inside the room, not cast. |
|
|
111
|
-
| **One named voice saying one line** — a character's own voice, a narrator, a clean VO to lip-sync against | **Inworld TTS-2** (`inworld-tts-2`) | The prompt IS the words, spoken verbatim and billed per character. Voice = the character's clip (cloned for the take), a description, or a preset. No room tone — mix it on the timeline. |
|
|
112
|
-
| **One effect that lands on a known frame**, or a seamless loop | **Sound Effects v2** (`eleven-sfx`) | The only surface with an exact duration control and a real loop mode. |
|
|
113
|
-
|
|
114
|
-
**There is no music model.** A song is imported (Slates reads audio files and puts them on the timeline), not generated. A line that has to be spoken in a SPECIFIC voice is generated on Inworld TTS-2 and lip-synced against; a line that belongs to a scene is performed by Seed Audio inside it.
|
|
115
|
-
|
|
116
|
-
### Named audio escalation triggers
|
|
117
|
-
|
|
118
|
-
- **"It needs to sound like a place"** → Seed Audio. Three separate SFX generations layered on the timeline is the wrong shape and costs more.
|
|
119
|
-
- **"Read this line"** → Seed Audio, with the line in quotes inside the scene sentence. Re-roll until the take is right, then lip-sync against it.
|
|
120
|
-
- **"That needs a thump right there"** → Sound Effects, with the duration set to roughly the length of the event.
|
|
121
|
-
- **"Give it a track"** → there is no music generation. Say so and offer to lay an imported track on an audio track.
|
|
122
|
-
|
|
123
|
-
**Rules:**
|
|
124
|
-
|
|
125
|
-
- **🚨 Seed Audio has NO duration parameter.** Length comes from the prompt text, so Slates writes the requested duration into the prompt and **bills what you asked for**. Choose the duration deliberately and never write a second, different length into the sentence. Full doctrine: `slates-prompting-seed-audio`.
|
|
126
|
-
- **Kling's audio syntax does not transfer.** `SFX:` / `Ambient noise:` / `Background music:` prefixes are Kling 3.0 *video* prompt syntax. Seed Audio reads them as literal words and the result degrades.
|
|
127
|
-
- **Beds outlast the cut.** Always ask for more seconds than the clip needs so the edit has fade handles — and remember those extra seconds are billed on both surfaces.
|
|
128
|
-
- **Audio inside the video vs audio as an asset.** If the sound must be locked to what happens on screen, generate it with the video (Kling omni / Seedance / Omni Flash / Veo). If it needs to be moved, trimmed, re-used, or layered, generate it here and drop it on an audio track.
|
|
129
|
-
- Per-model prompting: `slates-prompting-seed-audio`, `slates-prompting-elevenlabs`.
|
|
130
|
-
|
|
131
|
-
## Cost is a tiebreaker, not the router
|
|
132
|
-
|
|
133
|
-
Route by capability first, then pick the cheapest tier that serves the job (per `slates-cost-discipline`). Never pick a model because its per-second price looked lowest — a cheap clip that has to be regenerated on the right model costs more than routing correctly once.
|
|
1
|
+
---
|
|
2
|
+
name: slates-model-selection
|
|
3
|
+
description: Which model to pick for a given job — the routing doctrine. Read BEFORE choosing any video or image model, before quoting a plan, and before defaulting anywhere. Seedance 2.5 is the DEFAULT video model (Eric, 2026-09-13: the best in the world — physics, effects, scale, 30s takes, 30 references, timestamps); Seedance 2.0 is the 4K seat and the cheaper one at every shared resolution; Kling 3.0 is the cost-effective seat for performances, start-frame animation and lip-sync; MiniMax H3 is the AUTHORED-AUDIO seat (three directable sound layers in one pass, declared reference relationships, 480p-4K) with MiniMax H3 Max beside it as a faster 768p-capped premium with omni-references; Veo 3.1 is a narrow niche (native synced audio in one gen, 16:9 or 9:16, 4/6/8s) and never the default.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Model selection — the routing doctrine
|
|
7
|
+
|
|
8
|
+
Pick the model FIRST, deliberately, before writing a prompt or quoting a plan. Model routing is a core part of the intelligence users are paying for: the agent knows what each model is good at and which ones underperform for a job — defaulting to the wrong model burns the user's credits on a weaker result.
|
|
9
|
+
|
|
10
|
+
## 🔑 The meta-rule — above the table
|
|
11
|
+
|
|
12
|
+
The tables below are a snapshot. This roster churns constantly (NB2 Lite, Omni Flash, Seedream 5 Lite, GPT Image 2.5 all landed recently) — **a table rots; a rule doesn't.** When the tables and this rule disagree, or when a model appears that the tables don't cover, run the rule:
|
|
13
|
+
|
|
14
|
+
> **Name ONE must-preserve requirement for the shot.** Not a vibe — the single thing that, if it breaks, makes the shot unusable: this face stays this face · the fluid behaves like fluid · the text stays legible · the take stays one unbroken move.
|
|
15
|
+
>
|
|
16
|
+
> **Inspect the output at its intended crop.** A frame that holds up as a thumbnail can fall apart at the size it will actually be watched. For a location, look at atmosphere, material texture, and anchor objects; for a character, identity, skin, pose, and gradients.
|
|
17
|
+
>
|
|
18
|
+
> **Choose the model that PROVES that requirement** and leaves only failures you can afford to rerun or mask.
|
|
19
|
+
>
|
|
20
|
+
> **When the roster changes, repeat the evidence test.** Do not carry today's ranking forward on reputation.
|
|
21
|
+
|
|
22
|
+
## Video routing
|
|
23
|
+
|
|
24
|
+
| Job | Model | Why |
|
|
25
|
+
|---|---|---|
|
|
26
|
+
| **General-purpose — the default for most shots** | **Seedance 2.5** | The strongest seat in the catalogue: physics, effects, scale and hero shots, 4–30s in one take, 30 image + 10 video + 10 audio references, audio-only refs, and the only Seedance seat that acts on timestamps. 480p / 720p / 1080p, no 4K. LENGTH is the price dial — quote any take over ~10s. |
|
|
27
|
+
| **Cost matters and the shot is a performance or a start-frame animation** | **Kling 3.0 std** | Cost-effective workhorse. Strong image-to-video: preserves identity, layout, and text from the start frame. 16:9 / 9:16 / 1:1, 3–15s. |
|
|
28
|
+
| Higher visual polish, no physics demands | Kling 3.0 pro | Mid-price fidelity bump on the same strengths. |
|
|
29
|
+
| Multi-character dialogue / audio co-generation | Kling 3.0 omni | Dialogue syntax, voice direction, language codes, `@element` refs. |
|
|
30
|
+
| **4K delivery**, or the same resolution cheaper than 2.5 | **Seedance 2.0** | The only Seedance with native 4K (4K video is Pro-only) and cheaper than 2.5 at every shared resolution (720p $0.15/s vs $0.231/s). Same physics and effects strengths; 15s takes, 15 references, no timestamps. |
|
|
31
|
+
| **One take longer than 15 seconds**, more than 15 references, an AUDIO-ONLY reference, or **beats that have to land at a named second** | **Seedance 2.5** | Only 2.5 does these (rules in `slates-prompting-seedance-2-5` § Timestamps); it is the default anyway. 🚨 Two live hazards: (a) with references attached, the words *add / remove / replace / change / extend / continue* make it reclassify the request as a video EDIT and fail AFTER the job queues — describe the finished frame, or use `seedance-2.5-edit`; (b) LENGTH is the price dial, not resolution — a 30s 720p face gen is 489 credits and a 30s 1080p faceless gen is 614, against a 1,000-credit welcome grant. Quote before any take over ~10s. |
|
|
32
|
+
| **The SOUND has to be directed, not just present** — a specific line delivered a specific way, scene sound that has to sit under it, and score that must stay out of the characters' world | **MiniMax H3** | The only seat where audio is authored in three separate layers in ONE pass (synchronised events in the body, ambience in a soundscape section, audience-only score in its own) rather than toggled on. 5–15s, 480p / 768p / 2K / 4K, 24fps, 32kHz stereo, 11 languages. Rules in `slates-prompting-minimax-h3`. |
|
|
33
|
+
| **A reference has to keep a DECLARED amount of itself** — especially moving one subject's characteristic onto a *different* subject | **MiniMax H3** | The only seat that understands a stated retention relationship (kept whole / kept in part / transferred onto another subject / loose echo). 9 images + 3 video + 3 audio, 12 files total. 🚨 The first 5 reference images are free and every one after that costs 4 credits — pass `referenceImages` to `slates_estimate_generation_cost` before a reference-heavy job. |
|
|
34
|
+
| **Turnaround is the requirement** on a text-to-video or start-frame shot at 480p/768p | **MiniMax H3 Max** | fal's self-hosted post-train of H3. **Measured 2026-08-27: a 5s 768p clip finished in 4.8s against 57s on base H3 — about 12x faster**, same prompt, queue to file. When turnaround is the requirement this is not a marginal win. 🚨 It is the PREMIUM seat, not a cheap H3 — $0.080/s at 768p against base H3's $0.060/s, 33% more, and it tops out at 768p. It still animates a start frame and an end frame — image-to-video is one of the two things it is for — and since 2026-09-09 it takes the full omni-reference set too (9 images + 3 video + 3 audio), so the seats now differ on ladder and price rather than on what they accept. Never the default; never reach for it to save money. |
|
|
35
|
+
| Native synchronized audio (dialogue + SFX generated WITH the video in one gen), 16:9, ≤8s | Veo 3.1 | Narrow, and now narrower: if the sound needs DIRECTING rather than merely existing, MiniMax H3 is the better seat. |
|
|
36
|
+
|
|
37
|
+
### Named Seedance escalation triggers
|
|
38
|
+
|
|
39
|
+
"Physics matter" is an abstract category and it under-fires. These are the beats Seedance is **observably** good at — if the shot contains one, escalate without deliberating:
|
|
40
|
+
|
|
41
|
+
- **Real-time → slow-motion contrast.** The signature beat; nearly every strong clip rides it.
|
|
42
|
+
- **The camera moving while debris, meteors, sparks or particles crash around the subject.** Distinctly a feature of this model, not just a thing it survives.
|
|
43
|
+
- **Massive scale that has to read as genuinely huge** — not "a big thing", a thing whose size is the point of the shot.
|
|
44
|
+
- **One continuous unbroken take.**
|
|
45
|
+
|
|
46
|
+
Concrete beats route better than an abstract category. Cost stays a tiebreaker, never the router (see below).
|
|
47
|
+
|
|
48
|
+
## Video EDIT routing (changing an existing clip)
|
|
49
|
+
|
|
50
|
+
| Job | Tool | Why |
|
|
51
|
+
|---|---|---|
|
|
52
|
+
| **Footage-synced VFX on real footage** — add/remove an effect, prop, or lighting change while the take stays the take (incl. talking heads) | **Omni Flash Edit** (`slates_edit_video`, `omni-flash-edit`) | **The edit-fidelity winner** (head-to-head receipt 2026-07-09, WITH a short prompt): lip movement held perfectly, audio near-identical, effect landed and released on cue — where Kling missed an action beat and drifted lips. Prompt-only, 3–10s clips, 720p out, ~6.4 cr/s (cheapest). Quirk: occasional tail jitter / doubled final speech beat — trim the tail on the timeline. Fidelity is EARNED by prompt discipline: one short line + "Keep everything else the same"; long prompts destroy it (see below). |
|
|
53
|
+
| **Identity swap needing reference images** — put @marcus into the clip, lock a style from refs | **Kling O3 Edit** (`slates_edit_video`) | The only edit engine that takes element/style reference images (frontal + angles lock identity). ~19¢/s. |
|
|
54
|
+
| **Spoken words must be bit-exact** (VO, legal copy, music) | **Kling O3 Edit** with `keepAudio` (default true) — or segment-splice | Kling keeps the ORIGINAL audio track verbatim — but re-synthesizes the video, so lips can drift slightly against it (7/09 receipt). Omni Flash regenerates audio (voice editing unsupported): on the 7/09 receipt it came back near-identical with perfect lips, but "near-identical" is not a guarantee. Zero-risk path for critical audio: segment-splice — edit only the non-talking seconds and keep the original track under the cut. |
|
|
55
|
+
| Style-transfer-heavy re-imagining, full relocate of the scene, or edit quality worth a premium at 1080p+ | Seedance edit/relocate (`videoReferenceAssetId` on `slates_generate_video`) | Seedance's strength is transfer intensity; it re-generates rather than surgically edits. Head-to-head receipt 2026-07-09 (photoreal-insert job, same clip): at 720p it LOST to Omni Flash edit on result while costing ~3× (vref bills input+output seconds; face-lane rates when people are in frame). Route here for its strengths or at 1080p/4K where its ceiling is higher — never as the cheap default. (2.5's relocate lane reaches 1080p too as of 2026-08-24, at $0.2457/s of combined input+output.) Takes long descriptive prompts fine (no Omni-style hard-fail on timing phrasing). |
|
|
56
|
+
| **A clip LONGER THAN 15 SECONDS** | **Seedance 2.5 Edit** (`slates_edit_video`, `seedance-2.5-edit`) | The only edit engine that takes a 4–30s clip — length is the whole reason to route here. 480p/720p/1080p out, native audio, prompt + clip only (no reference images). Output length AND aspect ratio follow the source, so the billed key is the ceiled source length; an edit bills roughly DOUBLE a plain 2.5 generation of the same length because every provider charges an edit on input + output seconds. Set `seedanceFace: true` when a face is visible — the faceless provider blocks faces outright. No consented-real-face route for editing. Inside 15s, choose on fidelity instead. |
|
|
57
|
+
| AI-edit the user's OWN footage | Omni Flash Edit (3–10s), Kling O3 Edit (3–15s, 720–3840px) or Seedance 2.5 Edit (4–30s) | Both take any MP4/MOV — not just Slates gens. Phone footage MUST be rotation-normalized first (players honor the rotation flag; models don't — raw portrait phone clips come back SIDEWAYS). |
|
|
58
|
+
|
|
59
|
+
- **Edit before re-roll.** A re-roll gambles away the parts the user already likes; an edit changes only what the prompt names. Quote the edit first when a clip is mostly right.
|
|
60
|
+
- **Ship via segment-splice.** Every edit model re-synthesizes the whole clip, so fidelity risk scales with clip length. For real deliverables: trim out ONLY the seconds where the change happens, edit that segment, splice it back over the original on the timeline with the ORIGINAL audio underneath. Most of the final video stays the untouched original — that's how the polished split-screen demos going around actually work, plus gesture-only beats with voiceover laid over in post.
|
|
61
|
+
- **One change per pass, short prompts.** On Omni Flash this is documented law ("overly descriptive prompts can lead to unintended changes" — long identity-lock preambles make drift WORSE, receipt 7/09); on Kling multi-beat instructions get dropped. Chain passes instead.
|
|
62
|
+
- Edited clips are themselves editable clips — chain passes; lineage links each output to its parent.
|
|
63
|
+
|
|
64
|
+
## Motion Transfer & Lip Sync routing (Kling-only tools)
|
|
65
|
+
|
|
66
|
+
Both tools are **Kling-only**. Every entry in them is a real Kling endpoint that bolts motion or lip movement onto a finished source as a dedicated post-process.
|
|
67
|
+
|
|
68
|
+
| Job | Tool | Why |
|
|
69
|
+
|---|---|---|
|
|
70
|
+
| Motion retarget onto a still character | Kling MC std/pro (`slates_generate_motion_transfer`) | Structured skeleton/depth retarget, ~32–42 credits / 5s, takes up to 30s driving clips. |
|
|
71
|
+
| Re-voice a clip, or animate a still portrait | Kling lip-sync / avatar (`slates_generate_lip_sync`) | ~4–29 credits / 5s blocks. |
|
|
72
|
+
|
|
73
|
+
**Want the Seedance version of either?** It is not a switch on these tools — it is a normal `slates_generate_video` on `seedance-2` with the clip attached as a **video reference** and the motion or dialogue written into the prompt ("the character from image 1 performs the exact motion from video 1"). That routes to the same endpoint the tool would have called, with the prompt visible and editable instead of ghost-written. Single-pass conditioning genuinely beats post-hoc retargeting on fast choreography, contact, cloth and hair — and it carries native audio — so escalate there whenever fidelity matters.
|
|
74
|
+
|
|
75
|
+
- Seedance video-reference gens bill COMBINED input+output seconds (`seedance-2*-vref-*` keys) — pass the clip duration and quote before confirming. Driving clips must be 2–15s on Seedance 2.0 and up to 30s on 2.5; past that it is Kling MC's lane.
|
|
76
|
+
- Faces on that route go through the normal cascade: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → `seedanceRealFace` + `realFaceConsent` for a real person (premium realface pricing).
|
|
77
|
+
|
|
78
|
+
**Rules:**
|
|
79
|
+
|
|
80
|
+
- **Default video = Seedance 2.5.** Route to Seedance 2.0 for 4K or when the same resolution must be cheaper, and to Kling 3.0 std when the budget matters and the shot is a performance or a start-frame animation — and say why in the plan ("4K delivery, routing to 2.0"; "budget dialogue shot, routing to Kling").
|
|
81
|
+
- **Veo is never the default.** 16:9 or 9:16 only, 4/6/8s only (and 8s only at 1080p/4K, or with reference images), and it is not the quality pick — treat it as a single-purpose tool for native-synced-audio shots. If audio can be added after (Kling lip-sync, edit stage), prefer Kling or Seedance + audio in post.
|
|
82
|
+
- **9:16 vertical → Kling or Seedance by preference**, not by necessity: Veo does take 9:16 on the route Slates uses. Route away from it because it is the niche seat, not because it can't.
|
|
83
|
+
- **Ratios and durations are enforced before submit.** `slates_generate_video` validates the aspect ratio, resolution and duration against the model you picked and refuses out-of-set values with the legal list — it will not silently ignore or downgrade them. The authoritative per-model sets are in the op's own param descriptions, which are generated from the capability SSOT; prefer those over any list written in prose here.
|
|
84
|
+
- **Image-to-video from an NB2 start frame** (the standard pipeline) → Seedance 2.5 by default, Kling when the budget matters and the motion is a performance. Not Veo.
|
|
85
|
+
- **User names a model explicitly → use it.** But if it's a mismatch for the job (crazy physics on Kling std, a 30s take on anything but Seedance 2.5, 4K on Seedance 2.5 which has none), say so in one line and offer the right route before generating.
|
|
86
|
+
|
|
87
|
+
## Image routing
|
|
88
|
+
|
|
89
|
+
**Video models (Kling, Seedance, Veo) cannot generate standalone images — ever.** A "premium hero reference image" is still an image job: it routes to an image model below, never to Seedance.
|
|
90
|
+
|
|
91
|
+
- **Default: Nano Banana 2** — strongest reference HANDLING (14 refs; GPT Image now takes more, at 16, but Banana is still the one that holds many subjects coherently), best legible text, the standard start-frame generator.
|
|
92
|
+
- **NB2 Lite** — the fast/draft seat: ~half NB2's price, ~2.7× faster, 1K only. Route iteration volume and drafts here; finals go back to NB2 full (2K/4K).
|
|
93
|
+
- **Nano Banana Pro** — the hero-frame/typography ceiling (~2× NB2). NB2 ≈ 95% of Pro; escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element frames must be perfect. Up to 14 refs — feed it a full subject library.
|
|
94
|
+
- **GPT Image 2.5** — two seats, `gpt-image-2-5-flare` and `gpt-image-2-5-sunburst`, **same price**. Readable text / panels / UI king: character sheets, shot grids, diagrams, text-bearing panels. **Also the photoreal front-runner (Eric, 2026-08-24)** — it beat both Nano Banana rails head-to-head on skin realism, which is why the AI-influencer ad lane generates every plate on this line. **The seat split is SPEED vs QUALITY, not generate vs edit** (OpenAI's own rule): Flare is the small, fast model with quality *comparable to* GPT Image 2 — drafts, exploration, volume; Sunburst is OpenAI's *most capable* image model, higher quality than GPT Image 2, deliberately slower — finals, hero frames, photoreal, and multi-reference edits, where its lead is widest. **Explore on Flare, finish on Sunburst.** Five quality tiers, cheapest first — `low` (layout checks only), `medium` (drafts), **`high` (the default)**, `xhigh`, `max` (the top). Uneven: `max` is 4× `high`, `xhigh` only ~1.8× it. **16 reference images**, the schema ceiling. **Transparent backgrounds** via `backgroundMode` — free, and the only image family that offers them.
|
|
95
|
+
|
|
96
|
+
🚨 **The tier names moved when 2.5 replaced GPT Image 2, and the strings did not.** GPT Image 2's `medium` is 2.5's `high`; its `high` is 2.5's `max` — same money, one rung of renaming. The 2026-08-24 photoreal result was measured at GPT Image 2 `high`, so **the tier that reproduces it is `max`**. Nobody has re-run it on 2.5; the ranking is inherited, not re-measured.
|
|
97
|
+
- **FLUX.2 Max** — photoreal texture, hex-color binding, typography, less censored.
|
|
98
|
+
- **Seedream 5 Lite** — uncensored + any-resolution flat price; volume exploration when the Gemini filter is in the way.
|
|
99
|
+
|
|
100
|
+
**Split rule of thumb:** readable text / panels / UI → GPT Image 2.5 (Flare to explore, Sunburst to finish); **photoreal people, finals and hero frames → Sunburst at `max`** — the 2026-08-24 result was measured at GPT Image 2's `high`, which is `max` here, and Flare only *matches* GPT Image 2 while Sunburst exceeds it; multi-reference edits where several references must all survive into one frame → Sunburst; edit-heavy work → the Banana line; drafts → GPT Image 2.5 Flare at `medium`, which now undercuts NB2 Lite on both price and resolution; uncensored or odd resolutions → Seedream/FLUX.
|
|
101
|
+
|
|
102
|
+
⚠️ **This line said the opposite until 2026-08-24** — it sent photoreal *away* from GPT Image on reputation, which is the exact failure § The meta-rule above warns about. Re-run the evidence test when the roster moves. It moved again on 2026-09-09, and the ranking was carried across rather than re-measured — exactly what the meta-rule says not to trust. Treat it as a starting hypothesis for 2.5, not a receipt. **The seat choice above is likewise reasoned from OpenAI's positioning, not measured:** run Flare-`max` against Sunburst-`max` on one plate and write the answer into `slates-prompting-gpt-image-2-5`.
|
|
103
|
+
|
|
104
|
+
## Audio routing
|
|
105
|
+
|
|
106
|
+
**Image and video models cannot generate standalone audio, and neither audio model can generate images or video.** A shot that needs synced audio generated WITH the picture is still a video job (Kling omni / Veo / Omni Flash / Seedance all carry native audio); the models below produce audio *as its own asset*, to lay on the timeline.
|
|
107
|
+
|
|
108
|
+
| Job | Model | Why |
|
|
109
|
+
|---|---|---|
|
|
110
|
+
| **Default — a whole audio scene in one pass**: room tone, ambience beds, crowds, nature, layered dialogue + effects, spoken lines inside a scene | **Seed Audio 1.0** (`seed-audio`) | One plain sentence in, a complete scene out. The continuity-bed workhorse; dialogue is performed inside the room, not cast. |
|
|
111
|
+
| **One named voice saying one line** — a character's own voice, a narrator, a clean VO to lip-sync against | **Inworld TTS-2** (`inworld-tts-2`) | The prompt IS the words, spoken verbatim and billed per character. Voice = the character's clip (cloned for the take), a description, or a preset. No room tone — mix it on the timeline. |
|
|
112
|
+
| **One effect that lands on a known frame**, or a seamless loop | **Sound Effects v2** (`eleven-sfx`) | The only surface with an exact duration control and a real loop mode. |
|
|
113
|
+
|
|
114
|
+
**There is no music model.** A song is imported (Slates reads audio files and puts them on the timeline), not generated. A line that has to be spoken in a SPECIFIC voice is generated on Inworld TTS-2 and lip-synced against; a line that belongs to a scene is performed by Seed Audio inside it.
|
|
115
|
+
|
|
116
|
+
### Named audio escalation triggers
|
|
117
|
+
|
|
118
|
+
- **"It needs to sound like a place"** → Seed Audio. Three separate SFX generations layered on the timeline is the wrong shape and costs more.
|
|
119
|
+
- **"Read this line"** → Seed Audio, with the line in quotes inside the scene sentence. Re-roll until the take is right, then lip-sync against it.
|
|
120
|
+
- **"That needs a thump right there"** → Sound Effects, with the duration set to roughly the length of the event.
|
|
121
|
+
- **"Give it a track"** → there is no music generation. Say so and offer to lay an imported track on an audio track.
|
|
122
|
+
|
|
123
|
+
**Rules:**
|
|
124
|
+
|
|
125
|
+
- **🚨 Seed Audio has NO duration parameter.** Length comes from the prompt text, so Slates writes the requested duration into the prompt and **bills what you asked for**. Choose the duration deliberately and never write a second, different length into the sentence. Full doctrine: `slates-prompting-seed-audio`.
|
|
126
|
+
- **Kling's audio syntax does not transfer.** `SFX:` / `Ambient noise:` / `Background music:` prefixes are Kling 3.0 *video* prompt syntax. Seed Audio reads them as literal words and the result degrades.
|
|
127
|
+
- **Beds outlast the cut.** Always ask for more seconds than the clip needs so the edit has fade handles — and remember those extra seconds are billed on both surfaces.
|
|
128
|
+
- **Audio inside the video vs audio as an asset.** If the sound must be locked to what happens on screen, generate it with the video (Kling omni / Seedance / Omni Flash / Veo). If it needs to be moved, trimmed, re-used, or layered, generate it here and drop it on an audio track.
|
|
129
|
+
- Per-model prompting: `slates-prompting-seed-audio`, `slates-prompting-elevenlabs`.
|
|
130
|
+
|
|
131
|
+
## Cost is a tiebreaker, not the router
|
|
132
|
+
|
|
133
|
+
Route by capability first, then pick the cheapest tier that serves the job (per `slates-cost-discipline`). Never pick a model because its per-second price looked lowest — a cheap clip that has to be regenerated on the right model costs more than routing correctly once.
|