@slatesvideo/shared 0.5.9 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,48 @@
1
- // Per-model prompting facts — reference/ingredient limits and the prompt
2
- // formula, as KNOWLEDGE (for skills + lead-magnet + desktop tooltips). The
3
- // RUNTIME source of truth for limits is slate/src/shared/pricing.ts
4
- // (MODEL_REGISTRY.maxRefImages / maxIngredientImages); these mirror it for
5
- // documentation. Code-verified 2026-06-25.
1
+ // Per-model prompting facts — routing doctrine and the prompt formula, as
2
+ // KNOWLEDGE (for skills + lead-magnet + op descriptions).
3
+ //
4
+ // 🚨 THE REFERENCE CAPS ARE NO LONGER TYPED HERE (2026-08-16). They are DERIVED
5
+ // from `MODEL_CAPABILITIES` in ./model-capabilities.ts, which is now the single
6
+ // definition — the same module `slate/src/shared/pricing.ts` spreads into every
7
+ // MODEL_REGISTRY entry. This file used to hand-mirror those numbers "for
8
+ // documentation"; a hand-mirror in the same package as the source is exactly the
9
+ // defect the capability SSOT exists to delete, so `caps()` below does the lookup
10
+ // and a wrong id throws at module load instead of shipping a stale number.
6
11
  //
7
12
  // Prose that ALSO appears in a skill or the tips card comes from
8
13
  // skills/_partials/*.md via PARTIALS — never restated here. A `notes` string is
9
14
  // a third rendering of a fact, and a third rendering is a third thing that can
10
15
  // survive a doctrine reversal the other two got.
11
16
  import { PARTIALS } from './partials.generated.js';
17
+ import { MODEL_CAPABILITIES } from './model-capabilities.js';
18
+ /**
19
+ * Reference caps for a fact, read out of the capability SSOT.
20
+ *
21
+ * The argument is a `MODEL_CAPABILITIES` key, which is a REGISTRY model id — and
22
+ * three facts here are FAMILY-level (`kling-v3`, `kling-v3-edit`, `veo-3.1`)
23
+ * with no registry row of their own, so they name a representative variant.
24
+ * That is safe because the caps are identical across the variants in each
25
+ * family (std/pro/omni/omni-pro all take 4; both edit rows take 4; both Veo
26
+ * seats take 3) — and if a future variant diverges, this becomes a wrong number
27
+ * rather than a crash, so split the fact rather than picking a side.
28
+ *
29
+ * Throws on an unknown id: a typo must fail the build, not ship as `null` caps
30
+ * that quietly tell an agent a model reads no references.
31
+ */
32
+ function caps(capabilityId) {
33
+ const c = MODEL_CAPABILITIES[capabilityId];
34
+ if (!c)
35
+ throw new Error(`MODEL_FACTS: no MODEL_CAPABILITIES entry for "${capabilityId}"`);
36
+ return {
37
+ maxRefImages: c.maxRefImages ?? null,
38
+ maxIngredients: c.maxIngredientImages ?? null,
39
+ maxReferenceVideos: c.maxReferenceVideos ?? null,
40
+ maxReferenceAudio: c.maxReferenceAudio ?? null,
41
+ maxReferenceVideoSeconds: c.maxReferenceVideoSeconds ?? null,
42
+ maxReferenceAudioSeconds: c.maxReferenceAudioSeconds ?? null,
43
+ maxReferenceFilesTotal: c.maxReferenceFilesTotal ?? null,
44
+ };
45
+ }
12
46
  /**
13
47
  * One sentence of multimodal-reference capacity for a model, derived. Returns
14
48
  * an empty string for a model that takes none, so a caller can append it
@@ -79,141 +113,143 @@ export const MODEL_FACTS = [
79
113
  // (gemini-3-pro-image-preview); do not conflate them.
80
114
  label: 'Nano Banana 2 (Gemini 3.1 Flash Image)',
81
115
  kind: 'image',
82
- maxRefImages: 14, // 10 object-fidelity + 4 character-consistency; categories don't trade.
83
- maxIngredients: null,
116
+ // 14 = 10 object-fidelity + 4 character-consistency; the categories don't trade.
117
+ ...caps('nano-banana-2'),
84
118
  notes: 'Default image model. 14 refs hard cap (10 object + 4 character). Brief it like a creative director, not tag soup. No negativePrompt field — use positive reframing. Best image start-frame for legible text. Knowledge cutoff Jan 2025.',
85
119
  },
86
120
  {
87
121
  id: 'nano-banana-2-lite',
88
122
  label: 'Nano Banana 2 Lite',
89
123
  kind: 'image',
90
- maxRefImages: 4,
91
- maxIngredients: null,
124
+ ...caps('nano-banana-2-lite'),
92
125
  notes: 'FAST/DRAFT image tier — ~half the price of NB2 full, ~2.7× faster, 1K output ONLY. Same Gemini content filter as NB2. Route here for iteration volume and drafts where 1K is fine; keep NB2 full for final 2K/4K. Character consistency + legible text hold up.',
93
126
  },
94
127
  {
95
128
  id: 'nano-banana-pro',
96
129
  label: 'Nano Banana Pro',
97
130
  kind: 'image',
98
- maxRefImages: 14,
99
- maxIngredients: null,
131
+ ...caps('nano-banana-pro'),
100
132
  notes: 'HERO-FRAME / typography PREMIUM image tier (Gemini 3 Pro backbone; ~2× NB2 price). NB2 ≈ 95% of Pro — route here only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element reasoning must be perfect. Up to 14 reference images (character locking, multi-subject fusion). Native 16:9 + 4K.',
101
133
  },
102
134
  {
103
135
  id: 'gpt-image-2',
104
136
  label: 'GPT Image 2',
105
137
  kind: 'image',
106
- maxRefImages: 10,
107
- maxIngredients: null,
108
- notes: 'TEXT/DIAGRAM/PANEL king — near-perfect character-level text, ordered panels, exact placement (~3s gens). Route here for character sheets, shot grids, and text-bearing panels. Quality tiers: medium (default, the value seat — half NB2 price at 1080p) / high (~4×, max text precision). Third filter regime (OpenAI moderate). 4K is API-only — even paid ChatGPT can\'t render it. Photoreal/character-locked/edit-heavy → Banana line instead.',
138
+ ...caps('gpt-image-2'),
139
+ notes: 'TEXT/DIAGRAM/PANEL king — near-perfect character-level text, ordered panels, exact placement (~3s gens). Route here for character sheets, shot grids, and text-bearing panels. Quality tiers: medium (default, the value seat — half NB2 price at 1080p) / high (~4×, max text precision). Third filter regime (OpenAI moderate). 4K is API-only — even paid ChatGPT can\'t render it. ALSO THE PHOTOREAL FRONT-RUNNER (Eric, 2026-08-24) — at quality high it beat both Nano Banana rails head-to-head on skin realism, so route photoreal people HERE, not away. Banana still owns edit-heavy work and the 14-reference ceiling. Killed if a head-to-head at the intended crop goes the other way — re-run the evidence test, never carry this forward on reputation.',
109
140
  },
110
141
  {
111
142
  id: 'flux-2-max',
112
143
  label: 'FLUX.2 Max',
113
144
  kind: 'image',
114
- maxRefImages: 4,
115
- maxIngredients: null,
145
+ ...caps('flux-2-max'),
116
146
  notes: 'Photoreal, less censored, up to ~4MP. Auto-routes to its edit endpoint when references are present. Lower ref cap than NB2.',
117
147
  },
118
148
  {
119
149
  id: 'seedream-5-lite',
120
150
  label: 'Seedream 5 Lite',
121
151
  kind: 'image',
122
- maxRefImages: 10,
123
- maxIngredients: null,
152
+ ...caps('seedream-5-lite'),
124
153
  notes: 'Cheapest image model (~flat price). Less censored. Routes to its edit endpoint with references.',
125
154
  },
126
155
  {
127
156
  id: 'seedance-2',
128
157
  label: 'Seedance 2.0',
129
158
  kind: 'video',
130
- maxRefImages: null,
131
- maxIngredients: 9, // ingredient images per video gen
132
- maxReferenceVideos: 3,
133
- maxReferenceAudio: 3,
134
- maxReferenceVideoSeconds: 15,
135
- maxReferenceAudioSeconds: 15,
136
- maxReferenceFilesTotal: 12,
159
+ ...caps('seedance-2'),
137
160
  audioRefNeedsCompanion: true,
138
- notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds. 2.0 STAYS THE DEFAULT over 2.5 because it is the only Seedance with 1080p and 4K.',
161
+ notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds — at a DISCOUNTED per-second rate on every provider, roughly 0.6x the plain rate. 2.0 STAYS THE DEFAULT over 2.5 for two reasons, and neither is 1080p any more (2.5 gained 1080p on 2026-08-24): it is the only Seedance with native 4K, and it is cheaper at every shared tier (720p $0.15/s vs $0.231/s).',
139
162
  },
140
163
  {
141
164
  id: 'seedance-2.5',
142
165
  label: 'Seedance 2.5',
143
166
  kind: 'video',
144
- maxRefImages: null,
145
- maxIngredients: 30, // 30 image refs; the model also takes 10 video + 10 audio (50 total)
146
- maxReferenceVideos: 10,
147
- maxReferenceAudio: 10,
148
- maxReferenceVideoSeconds: 30,
149
- maxReferenceAudioSeconds: 30,
150
- maxReferenceFilesTotal: 50,
167
+ ...caps('seedance-2.5'),
151
168
  // No companion requirement — audio-only references are one of the things
152
169
  // the second seat actually buys.
153
- notes: `A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is 480p or 720p on every route Slates offers: no 1080p, no 4K. Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), TIMED BEATS, or tighter prompt adherence. Pick 2.0 when resolution matters at all. VIDEO-ONLY. TIMESTAMPS: ${PARTIALS['seedance-25-timestamps-short']} Multi-view subject reference images are also supported on 2.5 (up to 5 subjects) where 2.0 wanted one view per subject. 🚨 COST DISCIPLINE: 720p STOPS READING AS "THE CHEAP ONE" HERE. A 30s 720p clip on the real-face route is 710 credits and on the AI-face route 484 — more than a 15s 1080p Seedance 2.0 face generation (411), against a 1,000-credit welcome grant. Always quote with slates_estimate_generation_cost before a long take, and explore at SHORT LENGTH (4-8s) rather than at low resolution — a 480p pass does not de-risk a 720p render, because generation is stochastic and the 720p run is a different take, not the same shot rendered better. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.`,
170
+ notes: `A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is the EXPENSIVE seat: 480p, 720p or 1080p (1080p added 2026-08-24), no 4K, and it costs MORE than 2.0 at every tier they share — 54% more at 720p ($0.231/s vs $0.15/s faceless). Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), TIMED BEATS, or tighter prompt adherence. Pick 2.0 for 4K, and for the same resolution at a lower price. VIDEO-ONLY. TIMESTAMPS: ${PARTIALS['seedance-25-timestamps-short']} Multi-view subject reference images are also supported on 2.5 (up to 5 subjects) where 2.0 wanted one view per subject. 🚨 COST DISCIPLINE: LENGTH IS THE PRICE DIAL HERE, NOT RESOLUTION. A 30s 720p clip is 347 credits faceless / 489 on the AI-face route / 710 on the real-face route, and a 30s 1080p faceless take is 614 — 61% of a 1,000-credit welcome grant on ONE clip. Even 720p is not "the cheap one": 30s at 720p on the AI-face route beats a 15s 1080p Seedance 2.0 face generation (411). Always quote with slates_estimate_generation_cost before a long take, and explore at SHORT LENGTH (4-8s) rather than at low resolution — a 480p pass does not de-risk a 720p render, because generation is stochastic and the 720p run is a different take, not the same shot rendered better. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.`,
154
171
  },
155
172
  {
156
173
  id: 'seedance-2.5-edit',
157
174
  label: 'Seedance 2.5 Edit',
158
175
  kind: 'video',
159
- maxRefImages: null,
160
- maxIngredients: 0, // prompt + source clip only on slates_edit_video
161
- notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s), though ByteDance recommends staying inside 20s for quality. That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p output, native audio. Prompt + source clip only on this op — no reference images (the MODEL takes 1-5 reference images on an edit; Slates has not wired that path). Phrase the change as "from A to B", and TIMESTAMP a partial edit ("…from 4-6 seconds…") — 2.5 reads whole-second timestamps on edits, and without a range the instruction applies to the whole clip. AUDIO is editable on this same row: change a line, change an accent, translate dialogue with re-fitted lips, strip or replace BGM and sound effects. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
176
+ // 0 ingredients: prompt + source clip only on slates_edit_video.
177
+ ...caps('seedance-2.5-edit'),
178
+ notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s), though ByteDance recommends staying inside 20s for quality. That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p/1080p output, native audio. Prompt + source clip only on this op — no reference images (the MODEL takes 1-5 reference images on an edit; Slates has not wired that path). Phrase the change as "from A to B", and TIMESTAMP a partial edit ("…from 4-6 seconds…") — 2.5 reads whole-second timestamps on edits, and without a range the instruction applies to the whole clip. AUDIO is editable on this same row: change a line, change an accent, translate dialogue with re-fitted lips, strip or replace BGM and sound effects. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
162
179
  },
163
180
  {
164
181
  id: 'kling-v3',
165
182
  label: 'Kling 3.0',
166
183
  kind: 'video',
167
- maxRefImages: null,
168
- maxIngredients: 4,
184
+ // Family-level fact — caps are identical across std/pro/omni/omni-pro.
185
+ ...caps('kling-v3.0-std'),
169
186
  notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools (MC std/pro, lip-sync, avatar) — those two tools are Kling-only.',
170
187
  },
171
188
  {
172
189
  id: 'kling-v3-edit',
173
190
  label: 'Kling O3 Video Edit',
174
191
  kind: 'video',
175
- maxRefImages: null,
176
- maxIngredients: 4, // combined subject elements + style refs per edit
192
+ // Family-level fact; 4 = combined subject elements + style refs per edit.
193
+ ...caps('kling-v3.0-omni-edit'),
177
194
  notes: 'VIDEO-TO-VIDEO EDIT — the REF-DRIVEN edit tool: takes an EXISTING 3–15s clip and changes what the prompt names, with element/style reference images (@ElementN = frontal + angles) locking subject identity; max 4 combined refs. keep_audio preserves the ORIGINAL audio verbatim (spoken words cannot drift) — but video lips can drift slightly against it, and multi-beat instructions get under-executed (7/09 receipt: missed a second action beat Omni Flash edit landed) — ONE beat per pass. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the 7/09 fidelity head-to-head. Billed per second of output (≈ clip length, rounded up). Seedance edit/relocate is the alternative for style-transfer-heavy jobs.',
178
195
  },
179
196
  {
180
197
  id: 'veo-3.1',
181
198
  label: 'Veo 3.1',
182
199
  kind: 'video',
183
- maxRefImages: null,
184
- maxIngredients: 3,
200
+ // Family-level fact — fast and standard declare the same caps.
201
+ ...caps('veo-3.1-fast'),
185
202
  notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one gen. 16:9 only, 4/6/8s only. Otherwise Kling (default) or Seedance (physics/premium) win.',
186
203
  },
187
204
  {
188
205
  id: 'omni-flash',
189
206
  label: 'Gemini Omni Flash',
190
207
  kind: 'video',
191
- maxRefImages: null,
192
- maxIngredients: 7, // ref2v image_urls; 7 mirrors Google's own reference limit
208
+ // 7 ref2v image_urls — mirrors Google's own reference limit.
209
+ ...caps('omni-flash'),
193
210
  notes: 'CHEAP 720p tier with native synced audio included — t2v, single-start-frame i2v, or reference-to-video with up to 7 reference images. 3-10s, 16:9/9:16 only. No last frame, no video/audio references. VIDEO-ONLY. New seat: quality vs Kling/Seedance unproven pending comparison gens — do not route hero shots here; use it for cheap drafts, audio-in-one-gen at low cost, ref2v character consistency trials, and its edit variant.',
194
211
  },
195
212
  {
196
213
  id: 'omni-flash-edit',
197
214
  label: 'Omni Flash Edit',
198
215
  kind: 'video',
199
- maxRefImages: null,
200
- maxIngredients: 0, // prompt + source clip ONLY — no element/style refs on this endpoint
216
+ // 0: prompt + source clip ONLY — no element/style refs on this endpoint.
217
+ ...caps('omni-flash-edit'),
201
218
  notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (7/09 head-to-head vs Kling edit on real talking footage: lips held perfectly, audio near-identical, both action beats landed). Takes an EXISTING 3-10s clip and changes what the prompt names, footage-synced (prop/effect/environment/lighting swaps). Fidelity is EARNED by prompt discipline: ONE short instruction + "Keep everything else the same." — long descriptive prompts DESTROY it (Google-documented + 7/09 receipt). Never name objects as metaphors ("candle-like" → literal candle). Quirk: occasional tail jitter/doubled last speech beat — trim the tail. NO reference images (identity swaps needing refs → Kling edit); bit-exact audio needs → Kling keep_audio or segment-splice. 720p output, cheapest edit seat (~2/3 of Kling edit Std).',
202
219
  },
220
+ {
221
+ id: 'minimax-h3',
222
+ label: 'MiniMax H3',
223
+ kind: 'video',
224
+ ...caps('minimax-h3'),
225
+ // fal's reference-to-video refuses an audio-only reference: "Audio cannot
226
+ // be the only reference input; provide at least one reference image or
227
+ // video with it." Same behavioural rule as Seedance 2.0.
228
+ audioRefNeedsCompanion: true,
229
+ notes: 'THE AUTHORED-AUDIO SEAT. Reach for H3 when the sound is part of the shot rather than a switch on it: it writes synchronised dialogue, scene sound and an audience-only score in ONE pass, as three separate layers of the prompt, at 24fps with 32kHz stereo, across 11 stably-supported languages (Arabic, Chinese, English, French, German, Italian, Japanese, Korean, Portuguese, Russian, Spanish). Kling and Seedance treat audio as on/off; Veo generates it but gives you no way to direct the layers. The second thing only H3 gives you is a DECLARED REFERENCE RELATIONSHIP — you state how much of each reference survives (kept whole, partly kept, transferred onto a different subject, or a loose echo), including moving one subject\'s characteristic onto another. VIDEO-ONLY. 5-15s, 480p / 768p / 2K / 4K (768p default and native; 2K and 4K are upscales of a 768p base). $0.060/s at 768p — the cheapest 768-class second in the catalogue. Omni-reference ceiling: 9 images + 3 video clips + 3 audio clips, 12 files total, video and audio each 15s combined; an audio reference needs an image or video alongside it. 🚨 REFERENCE IMAGES PAST THE FIFTH COST 4 CREDITS EACH, on top of the per-second price — the first five are free, the model takes nine, and four paid images on a 10s 768p clip add 16 credits to a 30-credit generation. Attach the references the shot needs, not the maximum. 4K video is Pro-only (the server returns PRO_REQUIRED for a base account); 2K is open to every tier. \u2b06\ufe0f 2K AND 4K ARE UPSCALES OF A 768p RENDER, NOT LARGER GENERATIONS \u2014 fal states this outright, and in our own 2026-08-27 test the 2K pass came back with MORE artifacting than the 768p original it was built from, while costing 33 credits for a 5s take against 15 and taking almost twice as long. Treat them as a delivery-size convenience, never as a quality tier: generate at 768p, judge it there, and upscale in post if the pixels are genuinely needed.',
230
+ },
231
+ {
232
+ id: 'minimax-h3-max',
233
+ label: 'MiniMax H3 Max',
234
+ kind: 'video',
235
+ // No reference caps: fal publishes no reference-to-video endpoint for this
236
+ // row, so `caps()` returns nulls and the composer refuses references.
237
+ ...caps('minimax-h3-max'),
238
+ notes: 'THE SPEED SEAT, and the EXPENSIVE one at the tier they share — never the cheap H3 and never the default. fal\'s own post-train of the open H3 weights, self-hosted. 🚨 MEASURED 2026-08-27, same prompt and params on both rows: a 5s 768p text-to-video took **4.8 seconds** on Max against **57 seconds** on base H3 — **about 12x faster**, queue to finished file. That is the seat\'s whole case and it is now our own number, not fal\'s (fal claims under 3s; the literal claim did not hold at 4.8s wall-clock, the order of magnitude did). It also carries a thin quality edge on the with-audio Arena boards (1,204 vs 1,184 image-to-video, 1,235 vs 1,226 text-to-video — real, but 20 and 9 ELO, and vendor-reported). It gives up everything above 768p (no 2K, no 4K — the upscaler is not in the open weights) and takes NO references of any kind (no reference-to-video endpoint exists). It costs $0.080/s at 768p against base H3\'s $0.060/s: 33% more for a shorter ladder. So route here when a fast turnaround on a 480p/768p text-to-video or start-frame shot is worth the premium, and to base H3 for resolution, references, or the same tier cheaper. Same native audio, same 5-15s window, same six aspect ratios.',
239
+ },
203
240
  {
204
241
  id: 'seed-audio',
205
242
  label: 'Seed Audio 1.0',
206
243
  kind: 'audio',
207
- maxRefImages: 1, // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
208
- maxIngredients: null,
244
+ // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
245
+ ...caps('seed-audio'),
209
246
  notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
210
247
  },
211
248
  {
212
249
  id: 'eleven-sfx',
213
250
  label: 'ElevenLabs Sound Effects v2',
214
251
  kind: 'audio',
215
- maxRefImages: null,
216
- maxIngredients: null,
252
+ ...caps('eleven-sfx'),
217
253
  notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
218
254
  },
219
255
  ];
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
16
16
  /** Footer callout paragraphs. */
17
17
  footer?: string[];
18
18
  }
19
- export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
19
+ export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
20
20
  export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
21
21
  /** Null when no tips exist for the key — callers render an honest fallback. */
22
22
  export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
@@ -109,7 +109,7 @@ const SEEDANCE_25 = {
109
109
  ...SEEDANCE,
110
110
  label: 'Seedance 2.5',
111
111
  intro: [
112
- 'Seedance 2.5 is a SECOND SEAT next to 2.0, not an upgrade of it. It buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), audio-only references and integer-second timestamps — and it gives up 1080p and 4K. It is 480p or 720p on every route Slates offers.',
112
+ 'Seedance 2.5 is a SECOND SEAT next to 2.0, not an upgrade of it. It buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), audio-only references and integer-second timestamps — and it gives up 4K. It runs at 480p, 720p or 1080p, and it costs more than 2.0 at every tier they share, so pick 2.0 when you want the same resolution cheaper or you want 4K.',
113
113
  "ByteDance's official advanced formula has 8 slots: precise subject + action details + scene/environment + lighting & color tone + camera movement + visual style + image quality + constraints. Sweet spot 60-150 words for a single shot, longer for multi-shot.",
114
114
  ],
115
115
  columns: [
@@ -155,7 +155,7 @@ const SEEDANCE_25_EDIT = {
155
155
  label: 'Seedance 2.5 Edit',
156
156
  intro: [
157
157
  'Seedance 2.5 Edit changes an existing clip: attach the clip, describe only what should be different, and the original motion, framing and timing are kept. It is the only editor in Slates that takes a clip longer than 15 seconds — 4 to 30s, against Kling O3 Edit\'s 3-15s and Omni Flash Edit\'s 3-10s.',
158
- 'Output length and aspect ratio follow the SOURCE clip, so there is no duration or ratio control — the clip you attach is the quote. Output is 480p or 720p with native audio.',
158
+ 'Output length and aspect ratio follow the SOURCE clip, so there is no duration or ratio control — the clip you attach is the quote. Output is 480p, 720p or 1080p with native audio.',
159
159
  ],
160
160
  columns: [
161
161
  [
@@ -579,6 +579,67 @@ const ELEVEN_SFX = {
579
579
  ],
580
580
  ],
581
581
  };
582
+ const MINIMAX_H3 = {
583
+ label: 'MiniMax H3',
584
+ intro: [
585
+ 'MiniMax H3 generates picture and sound in one pass — 24fps, 32kHz stereo, 5-15 seconds, 11 stably-supported languages. It is the only video model in Slates where audio is AUTHORED rather than switched on: synchronised dialogue and action sounds go in the body of the prompt, ambience goes in a soundscape section, and audience-only music goes in a score section. Put a sound in the wrong section and it is dropped, doubled, or attributed to the wrong source.',
586
+ 'Two seats. Base H3 runs 480p / 768p / 2K / 4K and reads up to 9 reference images plus 3 video and 3 audio clips. H3 Max is fal\'s faster post-train: 768p ceiling, no references at all, and dearer than base H3 at 768p — a deliberate speed pick, never the cheap one. 768p is the default on both because it is the tier the model natively generates; 2K and 4K are upscales of a 768p base.',
587
+ ],
588
+ columns: [
589
+ [
590
+ {
591
+ heading: 'Three audio layers, three places',
592
+ example: 'body: "First batch of the morning."\nSoundscape: shutters scrape, trays clink\nScore: solo piano, slow, no swell',
593
+ note: 'Dialogue, singing and diegetic music (a radio in the scene) go in the BODY on the beat they land. Ambience goes in the soundscape. The score is audience-only — name instruments and tempo, not moods.',
594
+ critical: true,
595
+ },
596
+ {
597
+ heading: 'Shots and cut times',
598
+ example: '[Shot 2] At 00:03.500, the camera cuts to...',
599
+ note: 'The first shot carries no timestamp; later shots open with the bracket and a rising cut time inside the clip length. Transition verbs: cuts to / transitions to / changes to / switches to.',
600
+ },
601
+ {
602
+ heading: 'Write the camera into the sentence',
603
+ example: 'The camera pushes in with small amplitude at slow speed toward the letter in her hands.',
604
+ note: 'Named moves (push in, pull out, arc, tracking, POV, roll) with amplitude and speed modifiers. Never stack them as labels.',
605
+ },
606
+ {
607
+ heading: 'Voiceover needs both halves',
608
+ example: 'says in an off-screen voiceover: "..." — his lips remain completely closed.',
609
+ note: 'The off-screen phrase alone still animates a mouth. State the closed lips explicitly.',
610
+ },
611
+ ],
612
+ [
613
+ {
614
+ heading: 'Cite references by number',
615
+ example: 'Marcus (image 1) walks into the workshop (image 2)...',
616
+ note: `H3 takes references as typed slots and expects plain numbered prose — image 1, video 1, audio 1. Do not hand-write angle-bracket tags. ${PARTIALS['reference-tips-short']}`,
617
+ },
618
+ {
619
+ heading: 'Say how much of a reference survives',
620
+ example: 'Give the man in image 3 the weathered leather texture of the jacket in image 4.',
621
+ note: 'H3 is the only seat that understands transferring a characteristic onto a DIFFERENT subject. State each reference\'s job and how much of it should carry through — kept whole, kept in part, transferred, or a loose echo.',
622
+ },
623
+ {
624
+ heading: 'Reference images past the fifth cost extra',
625
+ note: 'The first 5 are free; each one after that adds 4 credits at every resolution and length, and the model takes 9. Four extra images on a 10s 768p clip add 16 credits to a 30-credit generation. Attach what the shot needs, not the ceiling.',
626
+ critical: true,
627
+ },
628
+ {
629
+ heading: '2K and 4K are upscales, not bigger renders',
630
+ note: 'Only 480p and 768p are generated natively; 2K and 4K enlarge a finished 768p take. In our own testing the 2K pass showed MORE artifacting than the 768p original while costing 33 credits for a 5-second take against 15. Generate and judge at 768p; step up only when a delivery spec demands the pixels.',
631
+ critical: true,
632
+ },
633
+ {
634
+ heading: 'Frames or references, never both',
635
+ note: 'A start and/or end frame runs on a different endpoint from references — the reference endpoint has no frame slots. Slates refuses the combination rather than dropping one side. An audio reference also cannot travel alone: pair it with an image or video.',
636
+ },
637
+ ],
638
+ ],
639
+ footer: [
640
+ 'Slates disables the provider\'s prompt expander, so what you write is what the model reads — nothing will pad a thin prompt. Aim for a 350-500 word body on a reference-carrying shot, and let dialogue-heavy scenes run longer if that is what fits the spoken timeline.',
641
+ ],
642
+ };
582
643
  export const PROMPTING_TIPS = {
583
644
  seedance: SEEDANCE,
584
645
  'seedance-2-5': SEEDANCE_25,
@@ -588,6 +649,7 @@ export const PROMPTING_TIPS = {
588
649
  veo: VEO,
589
650
  'omni-flash': OMNI_FLASH,
590
651
  'omni-flash-edit': OMNI_FLASH_EDIT,
652
+ 'minimax-h3': MINIMAX_H3,
591
653
  'nano-banana': NANO_BANANA,
592
654
  'nano-banana-lite': NANO_BANANA_LITE,
593
655
  'seed-audio': SEED_AUDIO,