@slatesvideo/shared 0.6.2 → 0.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,11 +9,22 @@
9
9
  // defect the capability SSOT exists to delete, so `caps()` below does the lookup
10
10
  // and a wrong id throws at module load instead of shipping a stale number.
11
11
  //
12
- // Prose that ALSO appears in a skill or the tips card comes from
13
- // skills/_partials/*.md via PARTIALS — never restated here. A `notes` string is
14
- // a third rendering of a fact, and a third rendering is a third thing that can
15
- // survive a doctrine reversal the other two got.
16
- import { PARTIALS } from './partials.generated.js';
12
+ // 🚨 `notes` IS ROUTING ONLY — why you would pick THIS seat over its neighbour.
13
+ // Nothing else. Not capability numbers (MODEL_CAPABILITIES owns those and already
14
+ // GENERATES prose for them into every op's param descriptions), not prices (the
15
+ // rate functions own those, and the agent's own REAL NUMBERS ONLY rule forbids it
16
+ // repeating a figure it cannot point to in a tool result), not prompt craft (the
17
+ // matching slates-prompting-* skill owns that, loaded on demand).
18
+ //
19
+ // It was all four for a while: 16,799 characters of `notes` carried 80 resolution
20
+ // tokens, 42 duration claims, 23 hard-typed prices and two skills' worth of craft
21
+ // into a prompt prefix that is always in context. A `notes` string is a third
22
+ // rendering of a fact, and a third rendering is a third thing that can survive a
23
+ // doctrine reversal the other two got. `scripts/agent-surface-lockstep-check.mjs`
24
+ // check 5 now fails the build if a price or a capability number reappears here.
25
+ //
26
+ // Relative cost claims STAY ("dearer than 2.0 at every shared tier") — that is
27
+ // routing. The figures go, because those are data.
17
28
  import { MODEL_CAPABILITIES } from './model-capabilities.js';
18
29
  /**
19
30
  * Reference caps for a fact, read out of the capability SSOT.
@@ -108,6 +119,7 @@ export function multimodalRefModels() {
108
119
  export const MODEL_FACTS = [
109
120
  {
110
121
  id: 'nano-banana-2',
122
+ route: 'generate',
111
123
  // Gemini 3.1 FLASH Image — verified against the runtime slug map in
112
124
  // slate/src/main/api/google.ts. Nano Banana PRO is a different model
113
125
  // (gemini-3-pro-image-preview); do not conflate them.
@@ -115,110 +127,124 @@ export const MODEL_FACTS = [
115
127
  kind: 'image',
116
128
  // 14 = 10 object-fidelity + 4 character-consistency; the categories don't trade.
117
129
  ...caps('nano-banana-2'),
118
- notes: 'Default image model. 14 refs hard cap (10 object + 4 character). Brief it like a creative director, not tag soup. No negativePrompt field — use positive reframing. Best image start-frame for legible text. Knowledge cutoff Jan 2025.',
130
+ notes: 'DEFAULT image model and the all-rounder — route here unless another seat\'s speciality is the point. Best start-frame for legible in-scene text. Knowledge cutoff Jan 2025: anything later needs reference images.',
119
131
  },
120
132
  {
121
133
  id: 'nano-banana-2-lite',
134
+ route: 'generate',
122
135
  label: 'Nano Banana 2 Lite',
123
136
  kind: 'image',
124
137
  ...caps('nano-banana-2-lite'),
125
- notes: 'FAST/DRAFT image tier — ~half the price of NB2 full, ~2.7× faster, 1K output ONLY. Same Gemini content filter as NB2. Route here for iteration volume and drafts where 1K is fine; keep NB2 full for final 2K/4K. Character consistency + legible text hold up.',
138
+ notes: 'FAST/DRAFT image tier — markedly cheaper and faster than NB2 full, at draft quality. Route here for iteration volume, then re-run the winner on NB2 full. Same Gemini content filter as NB2.',
126
139
  },
127
140
  {
128
141
  id: 'nano-banana-pro',
142
+ route: 'generate',
129
143
  label: 'Nano Banana Pro',
130
144
  kind: 'image',
131
145
  ...caps('nano-banana-pro'),
132
- notes: 'HERO-FRAME / typography PREMIUM image tier (Gemini 3 Pro backbone; ~2× NB2 price). NB2 ≈ 95% of Pro — route here only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element reasoning must be perfect. Up to 14 reference images (character locking, multi-subject fusion). Native 16:9 + 4K.',
146
+ notes: 'HERO-FRAME / typography PREMIUM image tier. NB2 is about 95% of Pro — escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene or deep multi-element reasoning must be perfect, and say why.',
133
147
  },
134
148
  {
135
149
  id: 'gpt-image-2',
150
+ route: 'generate',
136
151
  label: 'GPT Image 2',
137
152
  kind: 'image',
138
153
  ...caps('gpt-image-2'),
139
- notes: 'TEXT/DIAGRAM/PANEL king — near-perfect character-level text, ordered panels, exact placement (~3s gens). Route here for character sheets, shot grids, and text-bearing panels. Quality tiers: medium (default, the value seat — half NB2 price at 1080p) / high (~4×, max text precision). Third filter regime (OpenAI moderate). 4K is API-only — even paid ChatGPT can\'t render it. ALSO THE PHOTOREAL FRONT-RUNNER (Eric, 2026-08-24) — at quality high it beat both Nano Banana rails head-to-head on skin realism, so route photoreal people HERE, not away. Banana still owns edit-heavy work and the 14-reference ceiling. Killed if a head-to-head at the intended crop goes the other way — re-run the evidence test, never carry this forward on reputation.',
154
+ notes: 'TEXT / DIAGRAM / PANEL king — near-perfect character-level text, ordered panels, exact placement. Route here for character sheets, shot grids and text-bearing panels. ALSO THE PHOTOREAL FRONT-RUNNER (Eric, 2026-08-24): at quality high it beat both Nano Banana rails head-to-head on skin realism, so route photoreal people HERE rather than away. Banana still owns edit-heavy work and the largest reference ceiling. Its own content filter, distinct from Gemini\'s. Killed if a head-to-head at the intended crop goes the other way — re-run the evidence test, never carry this forward on reputation.',
140
155
  },
141
156
  {
142
157
  id: 'flux-2-max',
158
+ route: 'generate',
143
159
  label: 'FLUX.2 Max',
144
160
  kind: 'image',
145
161
  ...caps('flux-2-max'),
146
- notes: 'Photoreal, less censored, up to ~4MP. Auto-routes to its edit endpoint when references are present. Lower ref cap than NB2.',
162
+ notes: 'Photoreal image seat, less censored than the Gemini rails. Auto-routes to its edit endpoint when references are present.',
147
163
  },
148
164
  {
149
165
  id: 'seedream-5-lite',
166
+ route: 'generate',
150
167
  label: 'Seedream 5 Lite',
151
168
  kind: 'image',
152
169
  ...caps('seedream-5-lite'),
153
- notes: 'Cheapest image model (~flat price). Less censored. Routes to its edit endpoint with references.',
170
+ notes: 'CHEAPEST image seat, flat-priced. Less censored. Routes to its edit endpoint when references are present.',
154
171
  },
155
172
  {
156
173
  id: 'seedance-2',
174
+ route: 'generate',
157
175
  label: 'Seedance 2.0',
158
176
  kind: 'video',
159
177
  ...caps('seedance-2'),
160
178
  audioRefNeedsCompanion: true,
161
- notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds — at a DISCOUNTED per-second rate on every provider, roughly 0.6x the plain rate. 2.0 STAYS THE DEFAULT over 2.5 for two reasons, and neither is 1080p any more (2.5 gained 1080p on 2026-08-24): it is the only Seedance with native 4K, and it is cheaper at every shared tier (720p $0.15/s vs $0.231/s).',
179
+ notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction or scale matter, and for hero shots. VIDEO-ONLY. Strong image-to-video and own-footage restyle. 4K is Pro-gated (base accounts get PRO_REQUIRED). Stays the default over 2.5: it is the only Seedance with native 4K and it is cheaper at every tier the two share.',
162
180
  },
163
181
  {
164
182
  id: 'seedance-2.5',
183
+ route: 'generate',
165
184
  label: 'Seedance 2.5',
166
185
  kind: 'video',
167
186
  ...caps('seedance-2.5'),
168
187
  // No companion requirement — audio-only references are one of the things
169
188
  // the second seat actually buys.
170
- notes: `A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is the EXPENSIVE seat: 480p, 720p or 1080p (1080p added 2026-08-24), no 4K, and it costs MORE than 2.0 at every tier they share — 54% more at 720p ($0.231/s vs $0.15/s faceless). Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), TIMED BEATS, or tighter prompt adherence. Pick 2.0 for 4K, and for the same resolution at a lower price. VIDEO-ONLY. TIMESTAMPS: ${PARTIALS['seedance-25-timestamps-short']} Multi-view subject reference images are also supported on 2.5 (up to 5 subjects) where 2.0 wanted one view per subject. 🚨 COST DISCIPLINE: LENGTH IS THE PRICE DIAL HERE, NOT RESOLUTION. A 30s 720p clip is 347 credits faceless / 489 on the AI-face route / 710 on the real-face route, and a 30s 1080p faceless take is 614 — 61% of a 1,000-credit welcome grant on ONE clip. Even 720p is not "the cheap one": 30s at 720p on the AI-face route beats a 15s 1080p Seedance 2.0 face generation (411). Always quote with slates_estimate_generation_cost before a long take, and explore at SHORT LENGTH (4-8s) rather than at low resolution — a 480p pass does not de-risk a 720p render, because generation is stochastic and the 720p run is a different take, not the same shot rendered better. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.`,
189
+ notes: 'A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE — and the dearer one at every tier they share. Pick 2.5 when the shot needs LENGTH, MANY references, an AUDIO-ONLY reference, TIMED BEATS, or tighter prompt adherence; pick 2.0 for 4K and for the same resolution cheaper. VIDEO-ONLY. Timestamp grammar, and the edit/extend words that make the provider reclassify a fresh generation and fail it, are in slates-prompting-seedance-2-5.',
171
190
  },
172
191
  {
173
192
  id: 'seedance-2.5-edit',
193
+ route: 'edit',
174
194
  label: 'Seedance 2.5 Edit',
175
195
  kind: 'video',
176
196
  // 0 ingredients: prompt + source clip only on slates_edit_video.
177
197
  ...caps('seedance-2.5-edit'),
178
- notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s), though ByteDance recommends staying inside 20s for quality. That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p/1080p output, native audio. Prompt + source clip only on this op — no reference images (the MODEL takes 1-5 reference images on an edit; Slates has not wired that path). Phrase the change as "from A to B", and TIMESTAMP a partial edit ("…from 4-6 seconds…") — 2.5 reads whole-second timestamps on edits, and without a range the instruction applies to the whole clip. AUDIO is editable on this same row: change a line, change an accent, translate dialogue with re-fitted lips, strip or replace BGM and sound effects. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
198
+ notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video, and the only edit engine that takes a clip longer than the other two reach — that length is the whole reason to route here. Inside their range, compare on fidelity instead: Omni Flash edit won the prompt-only head-to-head, and Kling edit is the one that takes reference images. Edits audio on the same row (re-voice, re-accent, translate with re-fitted lips, replace BGM). Costs roughly double a plain 2.5 generation of the same length, because an edit bills input plus output seconds.',
179
199
  },
180
200
  {
181
201
  id: 'kling-v3',
202
+ route: 'generate',
182
203
  label: 'Kling 3.0',
183
204
  kind: 'video',
184
205
  // Family-level fact — caps are identical across std/pro/omni/omni-pro.
185
206
  ...caps('kling-v3.0-std'),
186
- notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools (MC std/pro, lip-sync, avatar) — those two tools are Kling-only.',
207
+ notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync, and the widest aspect-ratio set. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.',
187
208
  },
188
209
  {
189
210
  id: 'kling-v3-edit',
211
+ route: 'edit',
190
212
  label: 'Kling O3 Video Edit',
191
213
  kind: 'video',
192
214
  // Family-level fact; 4 = combined subject elements + style refs per edit.
193
215
  ...caps('kling-v3.0-omni-edit'),
194
- notes: 'VIDEO-TO-VIDEO EDIT — the REF-DRIVEN edit tool: takes an EXISTING 3–15s clip and changes what the prompt names, with element/style reference images (@ElementN = frontal + angles) locking subject identity; max 4 combined refs. keep_audio preserves the ORIGINAL audio verbatim (spoken words cannot drift) — but video lips can drift slightly against it, and multi-beat instructions get under-executed (7/09 receipt: missed a second action beat Omni Flash edit landed) — ONE beat per pass. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the 7/09 fidelity head-to-head. Billed per second of output (≈ clip length, rounded up). Seedance edit/relocate is the alternative for style-transfer-heavy jobs.',
216
+ notes: 'VIDEO-TO-VIDEO EDIT, the REF-DRIVEN one: it is the only edit seat that takes element/style reference images to lock subject identity, and its keep_audio preserves the original audio verbatim. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the fidelity head-to-head. One instruction beat per pass — multi-beat prompts get under-executed.',
195
217
  },
196
218
  {
197
219
  id: 'veo-3.1',
220
+ route: 'generate',
198
221
  label: 'Veo 3.1',
199
222
  kind: 'video',
200
223
  // Family-level fact — fast and standard declare the same caps.
201
224
  ...caps('veo-3.1-fast'),
202
- notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one gen. 16:9 only, 4/6/8s only. Otherwise Kling (default) or Seedance (physics/premium) win.',
225
+ notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one pass, and the narrowest aspect-ratio and duration sets in the catalogue are acceptable. Otherwise Kling (default) or Seedance (physics/premium) win.',
203
226
  },
204
227
  {
205
228
  id: 'omni-flash',
229
+ route: 'generate',
206
230
  label: 'Gemini Omni Flash',
207
231
  kind: 'video',
208
232
  // 7 ref2v image_urls — mirrors Google's own reference limit.
209
233
  ...caps('omni-flash'),
210
- notes: 'CHEAP 720p tier with native synced audio included — t2v, single-start-frame i2v, or reference-to-video with up to 7 reference images. 3-10s, 16:9/9:16 only. No last frame, no video/audio references. VIDEO-ONLY. New seat: quality vs Kling/Seedance unproven pending comparison gens — do not route hero shots here; use it for cheap drafts, audio-in-one-gen at low cost, ref2v character consistency trials, and its edit variant.',
234
+ notes: 'CHEAP tier with native synced audio included. Route here for cheap drafts, audio-in-one-pass at low cost, and reference-to-video character-consistency trials. VIDEO-ONLY. Quality against Kling/Seedance is unproven — do not route hero shots here.',
211
235
  },
212
236
  {
213
237
  id: 'omni-flash-edit',
238
+ route: 'edit',
214
239
  label: 'Omni Flash Edit',
215
240
  kind: 'video',
216
241
  // 0: prompt + source clip ONLY — no element/style refs on this endpoint.
217
242
  ...caps('omni-flash-edit'),
218
- notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (7/09 head-to-head vs Kling edit on real talking footage: lips held perfectly, audio near-identical, both action beats landed). Takes an EXISTING 3-10s clip and changes what the prompt names, footage-synced (prop/effect/environment/lighting swaps). Fidelity is EARNED by prompt discipline: ONE short instruction + "Keep everything else the same." — long descriptive prompts DESTROY it (Google-documented + 7/09 receipt). Never name objects as metaphors ("candle-like" → literal candle). Quirk: occasional tail jitter/doubled last speech beat — trim the tail. NO reference images (identity swaps needing refs → Kling edit); bit-exact audio needs → Kling keep_audio or segment-splice. 720p output, cheapest edit seat (~2/3 of Kling edit Std).',
243
+ notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (head-to-head vs Kling edit on real talking footage: lips held, audio near-identical, both action beats landed) and the cheapest edit seat. Footage-synced prop, effect, environment and lighting swaps. Takes NO reference images — identity swaps needing refs go to Kling edit. Fidelity is EARNED by prompt discipline; the exact form is in slates-prompting-omni-flash.',
219
244
  },
220
245
  {
221
246
  id: 'minimax-h3',
247
+ route: 'generate',
222
248
  label: 'MiniMax H3',
223
249
  kind: 'video',
224
250
  ...caps('minimax-h3'),
@@ -226,34 +252,72 @@ export const MODEL_FACTS = [
226
252
  // be the only reference input; provide at least one reference image or
227
253
  // video with it." Same behavioural rule as Seedance 2.0.
228
254
  audioRefNeedsCompanion: true,
229
- notes: 'THE AUTHORED-AUDIO SEAT. Reach for H3 when the sound is part of the shot rather than a switch on it: it writes synchronised dialogue, scene sound and an audience-only score in ONE pass, as three separate layers of the prompt, at 24fps with 32kHz stereo, across 11 stably-supported languages (Arabic, Chinese, English, French, German, Italian, Japanese, Korean, Portuguese, Russian, Spanish). Kling and Seedance treat audio as on/off; Veo generates it but gives you no way to direct the layers. The second thing only H3 gives you is a DECLARED REFERENCE RELATIONSHIP — you state how much of each reference survives (kept whole, partly kept, transferred onto a different subject, or a loose echo), including moving one subject\'s characteristic onto another. VIDEO-ONLY. 5-15s, 480p / 768p / 2K / 4K (768p default and native; 2K and 4K are upscales of a 768p base). $0.060/s at 768p — the cheapest 768-class second in the catalogue. Omni-reference ceiling: 9 images + 3 video clips + 3 audio clips, 12 files total, video and audio each 15s combined; an audio reference needs an image or video alongside it. 🚨 REFERENCE IMAGES PAST THE FIFTH COST 4 CREDITS EACH, on top of the per-second price — the first five are free, the model takes nine, and four paid images on a 10s 768p clip add 16 credits to a 30-credit generation. Attach the references the shot needs, not the maximum. 4K video is Pro-only (the server returns PRO_REQUIRED for a base account); 2K is open to every tier. \u2b06\ufe0f 2K AND 4K ARE UPSCALES OF A 768p RENDER, NOT LARGER GENERATIONS \u2014 fal states this outright, and in our own 2026-08-27 test the 2K pass came back with MORE artifacting than the 768p original it was built from, while costing 33 credits for a 5s take against 15 and taking almost twice as long. Treat them as a delivery-size convenience, never as a quality tier: generate at 768p, judge it there, and upscale in post if the pixels are genuinely needed.',
255
+ notes: 'THE AUTHORED-AUDIO SEAT — reach for H3 when the sound is part of the shot rather than a switch on it: synchronised dialogue, scene sound and an audience-only score directed as three separate layers in ONE pass, across eleven languages. Kling and Seedance treat audio as on/off; Veo generates it but gives you no way to direct the layers. Only H3 also carries a DECLARED REFERENCE RELATIONSHIP (kept whole, partly kept, transferred, or a loose echo). VIDEO-ONLY. Its top two resolution tiers are UPSCALES of the native render, not larger generations — judge at native and upscale in post. Reference images past the fifth are a PAID key dimension: pass referenceImages when quoting.',
230
256
  },
231
257
  {
232
258
  id: 'minimax-h3-max',
259
+ route: 'generate',
233
260
  label: 'MiniMax H3 Max',
234
261
  kind: 'video',
235
262
  // No reference caps: fal publishes no reference-to-video endpoint for this
236
263
  // row, so `caps()` returns nulls and the composer refuses references.
237
264
  ...caps('minimax-h3-max'),
238
- notes: 'THE SPEED SEAT, and the EXPENSIVE one at the tier they share — never the cheap H3 and never the default. fal\'s own post-train of the open H3 weights, self-hosted. 🚨 MEASURED 2026-08-27, same prompt and params on both rows: a 5s 768p text-to-video took **4.8 seconds** on Max against **57 seconds** on base H3 — **about 12x faster**, queue to finished file. That is the seat\'s whole case and it is now our own number, not fal\'s (fal claims under 3s; the literal claim did not hold at 4.8s wall-clock, the order of magnitude did). It also carries a thin quality edge on the with-audio Arena boards (1,204 vs 1,184 image-to-video, 1,235 vs 1,226 text-to-video — real, but 20 and 9 ELO, and vendor-reported). It gives up everything above 768p (no 2K, no 4K — the upscaler is not in the open weights). 🚨 FRAMES ARE UNAFFECTED — it takes a start frame and an end frame exactly like base H3, on `minimax/h3-max/image-to-video`, which is the route the image-to-video Arena score above is measured on. What it lacks is the REFERENCE endpoint (`minimax/h3-max/reference-to-video` 404s), so the omni-reference set — up to 9 identity/style/environment images plus reference video and audio — is base-H3 only. Never describe this row as taking no image input: an image-to-video shot is one of the two things it is FOR. It costs $0.080/s at 768p against base H3\'s $0.060/s: 33% more for a shorter ladder. So route here when a fast turnaround on a 480p/768p text-to-video or start-frame shot is worth the premium, and to base H3 for resolution, references, or the same tier cheaper. Same native audio, same 5-15s window, same six aspect ratios.',
265
+ notes: 'THE SPEED SEAT, and the DEARER one at the tier they share — never the cheap H3 and never the default. fal\'s post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It gives up the upper resolution tiers and the REFERENCE endpoint, so the omni-reference set is base-H3 only — but it still animates start and end frames, which is one of the two things it is FOR. Never describe this row as taking no image input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.',
266
+ },
267
+ {
268
+ id: 'ltx-2-5',
269
+ route: 'generate',
270
+ label: 'LTX-2.5',
271
+ kind: 'video',
272
+ // No reference caps: fal publishes text-to-video and image-to-video for LTX
273
+ // and no reference endpoint at all, so `caps()` returns nulls and the
274
+ // composer refuses references. Start/end FRAMES are unaffected.
275
+ ...caps('ltx-2-5'),
276
+ notes: 'THE VOLUME SEAT — the cheapest native 1080p second in the catalogue, and the row for MANY takes rather than one hero shot. Native synced audio is included free at every tier, unlike Kling where sound is a paid key dimension. It also reaches the highest resolution tier below 4K and makes the LONGEST clips in the catalogue. VIDEO-ONLY. INPUTS ARE FRAMES, NOT REFERENCES: start frame plus an optional end frame, and no reference endpoint at all — for character consistency across shots use H3 or Kling. Route here for batch coverage, long takes, and anything where the credit budget is the binding constraint.',
277
+ },
278
+ {
279
+ id: 'ltx-2-5-pro',
280
+ route: 'generate',
281
+ label: 'LTX-2.5 Pro',
282
+ kind: 'video',
283
+ ...caps('ltx-2-5-pro'),
284
+ notes: 'THE FIDELITY SEAT of the LTX pair — the full diffusion build against the base row\'s distilled one. 🚨 IT IS NOT A SUPERSET OF THE BASE ROW, which is the opposite of every other Pro seat here: it reaches a SHORTER resolution ladder and makes SHORTER clips, and it costs more at both tiers they share. Reaching for it because the name says Pro costs more AND takes away reach. Everything else matches the base row. Route here only when a specific shot needs the fidelity and fits inside its narrower envelope.',
239
285
  },
240
286
  {
241
287
  id: 'seed-audio',
288
+ route: 'generate',
242
289
  label: 'Seed Audio 1.0',
243
290
  kind: 'audio',
244
291
  // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
245
292
  ...caps('seed-audio'),
246
- notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
293
+ notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd and nature soundscapes, and quick scratch VO. AUDIO-ONLY. Takes one image XOR up to three audio clips as references, never both. Prompt form and the length rule are in slates-prompting-seed-audio.',
247
294
  },
248
295
  {
249
296
  id: 'eleven-sfx',
297
+ route: 'generate',
250
298
  label: 'ElevenLabs Sound Effects v2',
251
299
  kind: 'audio',
252
300
  ...caps('eleven-sfx'),
253
- notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
301
+ notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
254
302
  },
255
303
  ];
256
304
  const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
305
+ /**
306
+ * Routing prose for one lane, generated from the SSOT.
307
+ *
308
+ * THE ONE RENDERER. The Studio Agent's system prompt, the MCP server's
309
+ * instructions and the generate/edit ops' `model` descriptions all call this —
310
+ * so "never restate model routing in an op description" (slates-mcp/CLAUDE.md)
311
+ * is now enforced by there being nothing to restate. Before this, the video op
312
+ * carried 1,282 characters of hand-written routing that repeated MODEL_FACTS
313
+ * phrase for phrase ("SECOND SEAT", "AUTHORED-AUDIO", "never the default"),
314
+ * in the same file that forbids exactly that.
315
+ */
316
+ export function describeRouting(kind, route = 'generate') {
317
+ return MODEL_FACTS.filter((f) => f.kind === kind && f.route === route)
318
+ .map((f) => `${f.label}: ${f.notes}`)
319
+ .join('\n');
320
+ }
257
321
  export function getModelFact(id) {
258
322
  return FACT_BY_ID.get(id);
259
323
  }
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
16
16
  /** Footer callout paragraphs. */
17
17
  footer?: string[];
18
18
  }
19
- export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
19
+ export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'ltx-2-5' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
20
20
  export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
21
21
  /** Null when no tips exist for the key — callers render an honest fallback. */
22
22
  export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
@@ -640,6 +640,70 @@ const MINIMAX_H3 = {
640
640
  'Slates disables the provider\'s prompt expander, so what you write is what the model reads — nothing will pad a thin prompt. Aim for a 350-500 word body on a reference-carrying shot, and let dialogue-heavy scenes run longer if that is what fits the spoken timeline.',
641
641
  ],
642
642
  };
643
+ const LTX_2_5 = {
644
+ label: 'LTX-2.5',
645
+ intro: [
646
+ 'LTX-2.5 scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE, not the last. Lightricks ranks the six parts of a prompt in this order: sound, camera, character detail, shot type and scene, then scene dressing — and scene dressing is the first thing to cut when a prompt sprawls. Everything goes in ONE flowing paragraph, not a list of labelled sections.',
647
+ 'Two seats. Base LTX-2.5 is the distilled build: 720p / 1080p / 1440p / 4K and clips from 6 to 20 seconds, and it is the cheapest native 1080p second in Slates. LTX-2.5 Pro is the full diffusion build ("Diffusion Fidelity Rendering" spends extra compute on busy frames) but reaches a SHORTER ladder — 1080p and 10 seconds maximum — while costing about a third more. Pro is for a dense final render; base is for iteration, long takes and 4K.',
648
+ ],
649
+ columns: [
650
+ [
651
+ {
652
+ heading: 'Anchor every sound to something in frame',
653
+ example: 'the rope creaks against the cleat, gulls somewhere off the port bow',
654
+ note: 'Write the audio line last, then check each cue has a visible or at least locatable source. Anything unanchored gets invented for you. "Not visible but locatable" passes — a whistle is fine if you name the marshal post it comes from.',
655
+ critical: true,
656
+ },
657
+ {
658
+ heading: 'Never write mood words for sound',
659
+ example: 'Bad: "tense atmosphere, sense of dread"\nGood: "a loose shutter knocks twice against the frame"',
660
+ note: 'Atmosphere adjectives produce nothing. If a scene feels thin, add one more MOVING OBJECT with a sound attached to it rather than another adjective.',
661
+ },
662
+ {
663
+ heading: 'Dialogue takes quotes, language and accent',
664
+ example: '"We should not have come back," in English with a slight German accent.',
665
+ note: 'Give the character a beat of stillness before they speak so the lip sync has something to lock against. Describe the beat: looks, waits, speaks, looks away.',
666
+ },
667
+ {
668
+ heading: 'Emotion is physical, not abstract',
669
+ example: 'Bad: "she looks anxious"\nGood: "her jaw sets, she turns the ring on her finger twice"',
670
+ note: 'The model renders actions, not adjectives. Tension in the jaw, weight shifts, fidgeting hands — these read; "anxious" does not.',
671
+ },
672
+ ],
673
+ [
674
+ {
675
+ heading: 'Multishot: two to four shots, and re-establish at every cut',
676
+ example: 'wide establishing shot — hard cut — macro close-up — match cut — medium shot',
677
+ note: 'One generation can carry several connected shots holding character, light and voice across the cuts. Name the edit ("hard cut", "dissolve") in the prose, then RESET scale, angle, lens and light. Two to four is the working range.',
678
+ critical: true,
679
+ },
680
+ {
681
+ heading: 'Re-identify characters at every cut',
682
+ example: 'Good: "the woman in the bronze gown"\nBad: "she"',
683
+ note: 'Pronouns lose the character across a cut. Repeat the original descriptor every time. Also state what the SOUND does at the cut — silence is not assumed.',
684
+ },
685
+ {
686
+ heading: 'Write the camera into the sentence',
687
+ example: 'a slow push-in settles as she reaches the door, then holds',
688
+ note: 'Slates does not expose the camera_motion enum, and prose is the better tool anyway: a written move can be tied to a specific moment, an enum value cannot. Name lens, framing and the moment the move resolves.',
689
+ },
690
+ {
691
+ heading: 'Durations are even numbers only, from six',
692
+ note: '6, 8, 10, 12, 14, 16, 18 or 20 seconds — there is no 5s or 7s LTX clip. And the long end is 1080p-and-below only: at 1440p and 4K the ceiling drops to 10s. Slates always sends an explicit length rather than letting the model pick one, so what you choose is what you are billed for.',
693
+ critical: true,
694
+ },
695
+ {
696
+ heading: 'Do not ask for text on screen',
697
+ note: 'Neither the spelling nor its stability frame to frame can be relied on. Signage, labels and captions belong in post.',
698
+ },
699
+ ],
700
+ ],
701
+ footer: [
702
+ 'Frames, not references. LTX takes a start frame and an optional end frame (which generates a transition between the two) — it has no reference endpoint at all, so identity, style and environment reference images are not available on this model. For character consistency across separate shots, use MiniMax H3 or Kling.',
703
+ 'Aspect ratios are 16:9 and 9:16 only, and native audio is included free at every resolution — there is no sound surcharge on either seat.',
704
+ 'In image-to-video, do not cut away from the opening frame too early: you have paid for that frame, so let it play before the first move.',
705
+ ],
706
+ };
643
707
  export const PROMPTING_TIPS = {
644
708
  seedance: SEEDANCE,
645
709
  'seedance-2-5': SEEDANCE_25,
@@ -650,6 +714,7 @@ export const PROMPTING_TIPS = {
650
714
  'omni-flash': OMNI_FLASH,
651
715
  'omni-flash-edit': OMNI_FLASH_EDIT,
652
716
  'minimax-h3': MINIMAX_H3,
717
+ 'ltx-2-5': LTX_2_5,
653
718
  'nano-banana': NANO_BANANA,
654
719
  'nano-banana-lite': NANO_BANANA_LITE,
655
720
  'seed-audio': SEED_AUDIO,