@slatesvideo/shared 0.6.10 → 0.6.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -69,14 +69,14 @@ export function multimodalRefSummary(id) {
69
69
  return '';
70
70
  const parts = [];
71
71
  if (v > 0)
72
- parts.push(`${v} reference video${v === 1 ? '' : 's'} (${f.maxReferenceVideoSeconds}s combined)`);
72
+ parts.push(`${v} video${v === 1 ? '' : 's'} (${f.maxReferenceVideoSeconds}s total)`);
73
73
  if (a > 0)
74
- parts.push(`${a} reference audio clip${a === 1 ? '' : 's'} (${f.maxReferenceAudioSeconds}s combined)`);
75
- const total = f.maxReferenceFilesTotal ? `, ${f.maxReferenceFilesTotal} files max across all modalities` : '';
74
+ parts.push(`${a} audio clip${a === 1 ? '' : 's'} (${f.maxReferenceAudioSeconds}s total)`);
75
+ const total = f.maxReferenceFilesTotal ? `, ${f.maxReferenceFilesTotal} files max` : '';
76
76
  const companion = f.audioRefNeedsCompanion
77
- ? ' Audio needs at least one image or video reference alongside it.'
78
- : ' Audio-only references are allowed.';
79
- return `${f.label}: up to ${parts.join(' and ')}${total}.${companion}`;
77
+ ? '; audio needs an image or video alongside.'
78
+ : '; audio-only is allowed.';
79
+ return `${f.label}: up to ${parts.join(' + ')}${total}${companion}`;
80
80
  }
81
81
  /**
82
82
  * The prompt words that make Seedance 2.5 reclassify a reference-carrying
@@ -120,6 +120,7 @@ export const MODEL_FACTS = [
120
120
  {
121
121
  id: 'nano-banana-2',
122
122
  route: 'generate',
123
+ tier: 'default',
123
124
  // Gemini 3.1 FLASH Image — verified against the runtime slug map in
124
125
  // slate/src/main/api/google.ts. Nano Banana PRO is a different model
125
126
  // (gemini-3-pro-image-preview); do not conflate them.
@@ -132,6 +133,7 @@ export const MODEL_FACTS = [
132
133
  {
133
134
  id: 'nano-banana-2-lite',
134
135
  route: 'generate',
136
+ tier: 'specialist',
135
137
  label: 'Nano Banana 2 Lite',
136
138
  kind: 'image',
137
139
  ...caps('nano-banana-2-lite'),
@@ -140,6 +142,7 @@ export const MODEL_FACTS = [
140
142
  {
141
143
  id: 'nano-banana-pro',
142
144
  route: 'generate',
145
+ tier: 'specialist',
143
146
  label: 'Nano Banana Pro',
144
147
  kind: 'image',
145
148
  ...caps('nano-banana-pro'),
@@ -148,6 +151,7 @@ export const MODEL_FACTS = [
148
151
  {
149
152
  id: 'gpt-image-2-5-flare',
150
153
  route: 'generate',
154
+ tier: 'specialist',
151
155
  label: 'GPT Image 2.5 Flare',
152
156
  kind: 'image',
153
157
  ...caps('gpt-image-2-5-flare'),
@@ -156,6 +160,7 @@ export const MODEL_FACTS = [
156
160
  {
157
161
  id: 'gpt-image-2-5-sunburst',
158
162
  route: 'generate',
163
+ tier: 'specialist',
159
164
  label: 'GPT Image 2.5 Sunburst',
160
165
  kind: 'image',
161
166
  ...caps('gpt-image-2-5-sunburst'),
@@ -164,6 +169,7 @@ export const MODEL_FACTS = [
164
169
  {
165
170
  id: 'flux-2-max',
166
171
  route: 'generate',
172
+ tier: 'specialist',
167
173
  label: 'FLUX.2 Max',
168
174
  kind: 'image',
169
175
  ...caps('flux-2-max'),
@@ -172,6 +178,7 @@ export const MODEL_FACTS = [
172
178
  {
173
179
  id: 'seedream-5-lite',
174
180
  route: 'generate',
181
+ tier: 'specialist',
175
182
  label: 'Seedream 5 Lite',
176
183
  kind: 'image',
177
184
  ...caps('seedream-5-lite'),
@@ -180,25 +187,28 @@ export const MODEL_FACTS = [
180
187
  {
181
188
  id: 'seedance-2',
182
189
  route: 'generate',
190
+ tier: 'specialist',
183
191
  label: 'Seedance 2.0',
184
192
  kind: 'video',
185
193
  ...caps('seedance-2'),
186
194
  audioRefNeedsCompanion: true,
187
- notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction or scale matter, and for hero shots. VIDEO-ONLY. Strong image-to-video and own-footage restyle. 4K is Pro-gated (base accounts get PRO_REQUIRED). Stays the default over 2.5: it is the only Seedance with native 4K and it is cheaper at every tier the two share.',
195
+ notes: 'THE 4K AND VALUE SEAT beside the 2.5 default — the only Seedance with native 4K (Pro-gated; base accounts get PRO_REQUIRED) and cheaper than 2.5 at every resolution they share, with the same physics, effects and scale strengths; shorter takes, fewer references, no timestamps. VIDEO-ONLY. A bare "seedance" still resolves here for older CLIs that expect 4K.',
188
196
  },
189
197
  {
190
198
  id: 'seedance-2.5',
191
199
  route: 'generate',
200
+ tier: 'default',
192
201
  label: 'Seedance 2.5',
193
202
  kind: 'video',
194
203
  ...caps('seedance-2.5'),
195
204
  // No companion requirement — audio-only references are one of the things
196
205
  // the second seat actually buys.
197
- notes: 'A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE — and the dearer one at every tier they share. Pick 2.5 when the shot needs LENGTH, MANY references, an AUDIO-ONLY reference, TIMED BEATS, or tighter prompt adherence; pick 2.0 for 4K and for the same resolution cheaper. VIDEO-ONLY. Timestamp grammar, and the edit/extend words that make the provider reclassify a fresh generation and fail it, are in slates-prompting-seedance-2-5.',
206
+ notes: 'DEFAULT VIDEO MODEL — the strongest seat for physics, effects, scale and hero shots, and the only Seedance that takes long single takes, many references, audio-only references and integer-second timestamps. No 4K, and dearer than 2.0 at every shared resolution: go to 2.0 for 4K or the same resolution cheaper. LENGTH is the price dial — quote long takes first. VIDEO-ONLY. Timestamp grammar and the edit/extend words that make the provider reclassify and fail a generation are in slates-prompting-seedance-2-5.',
198
207
  },
199
208
  {
200
209
  id: 'seedance-2.5-edit',
201
210
  route: 'edit',
211
+ tier: 'specialist',
202
212
  label: 'Seedance 2.5 Edit',
203
213
  kind: 'video',
204
214
  // 0 ingredients: prompt + source clip only on slates_edit_video.
@@ -208,15 +218,17 @@ export const MODEL_FACTS = [
208
218
  {
209
219
  id: 'kling-v3',
210
220
  route: 'generate',
221
+ tier: 'specialist',
211
222
  label: 'Kling 3.0',
212
223
  kind: 'video',
213
224
  // Family-level fact — caps are identical across std/pro/omni/omni-pro.
214
225
  ...caps('kling-v3.0-std'),
215
- notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync, and the widest aspect-ratio set. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.',
226
+ notes: 'THE COST-EFFECTIVE SEAT — strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync and the widest aspect-ratio set; pick it when the budget matters and the shot is a performance or a start-frame animation. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.',
216
227
  },
217
228
  {
218
229
  id: 'kling-v3-edit',
219
230
  route: 'edit',
231
+ tier: 'specialist',
220
232
  label: 'Kling O3 Video Edit',
221
233
  kind: 'video',
222
234
  // Family-level fact; 4 = combined subject elements + style refs per edit.
@@ -226,15 +238,17 @@ export const MODEL_FACTS = [
226
238
  {
227
239
  id: 'veo-3.1',
228
240
  route: 'generate',
241
+ tier: 'niche',
229
242
  label: 'Veo 3.1',
230
243
  kind: 'video',
231
244
  // Family-level fact — fast and standard declare the same caps.
232
245
  ...caps('veo-3.1-fast'),
233
- notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one pass, and the narrowest aspect-ratio and duration sets in the catalogue are acceptable. Otherwise Kling (default) or Seedance (physics/premium) win.',
246
+ notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one pass, and the narrowest aspect-ratio and duration sets in the catalogue are acceptable. Otherwise Seedance 2.5 (the default) or Kling (cost-effective performance) win.',
234
247
  },
235
248
  {
236
249
  id: 'omni-flash',
237
250
  route: 'generate',
251
+ tier: 'specialist',
238
252
  label: 'Gemini Omni Flash',
239
253
  kind: 'video',
240
254
  // 7 ref2v image_urls — mirrors Google's own reference limit.
@@ -244,6 +258,7 @@ export const MODEL_FACTS = [
244
258
  {
245
259
  id: 'omni-flash-edit',
246
260
  route: 'edit',
261
+ tier: 'default',
247
262
  label: 'Omni Flash Edit',
248
263
  kind: 'video',
249
264
  // 0: prompt + source clip ONLY — no element/style refs on this endpoint.
@@ -253,6 +268,7 @@ export const MODEL_FACTS = [
253
268
  {
254
269
  id: 'minimax-h3',
255
270
  route: 'generate',
271
+ tier: 'specialist',
256
272
  label: 'MiniMax H3',
257
273
  kind: 'video',
258
274
  ...caps('minimax-h3'),
@@ -265,16 +281,24 @@ export const MODEL_FACTS = [
265
281
  {
266
282
  id: 'minimax-h3-max',
267
283
  route: 'generate',
284
+ tier: 'specialist',
268
285
  label: 'MiniMax H3 Max',
269
286
  kind: 'video',
270
- // No reference caps: fal publishes no reference-to-video endpoint for this
271
- // row, so `caps()` returns nulls and the composer refuses references.
287
+ // References landed 2026-09-09 when the "h3-max/reference-to-video returns
288
+ // 404" claim was retired against the fetched schema (caps come from
289
+ // model-capabilities.ts; the rip is second-brain/business/projects/slates/
290
+ // provider-docs/fal-minimax-h3-openapi-schemas.md, section 6).
272
291
  ...caps('minimax-h3-max'),
273
- notes: 'THE SPEED SEAT, and the DEARER one at the tier they share — never the cheap H3 and never the default. fal\'s post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It gives up the upper resolution tiers and the REFERENCE endpoint, so the omni-reference set is base-H3 only — but it still animates start and end frames, which is one of the two things it is FOR. Never describe this row as taking no image input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.',
292
+ // Quoted off THAT endpoint's reference_audio_urls description, not copied
293
+ // from the base row: "Audio cannot be the only reference input; provide at
294
+ // least one reference image or video with it."
295
+ audioRefNeedsCompanion: true,
296
+ notes: 'THE SPEED SEAT, and the DEARER one at the tier they share — never the cheap H3 and never the default. fal\'s post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It gives up the upper resolution tiers. It takes the same omni-reference set as base H3 and animates start and end frames — but not both in one call: its reference endpoint has no start/end-frame fields, where base H3\'s does. Never describe this row as taking no image or reference input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.',
274
297
  },
275
298
  {
276
299
  id: 'ltx-2-5',
277
300
  route: 'generate',
301
+ tier: 'specialist',
278
302
  label: 'LTX-2.5',
279
303
  kind: 'video',
280
304
  // No reference caps: fal publishes text-to-video and image-to-video for LTX
@@ -286,6 +310,7 @@ export const MODEL_FACTS = [
286
310
  {
287
311
  id: 'ltx-2-5-pro',
288
312
  route: 'generate',
313
+ tier: 'specialist',
289
314
  label: 'LTX-2.5 Pro',
290
315
  kind: 'video',
291
316
  ...caps('ltx-2-5-pro'),
@@ -294,6 +319,7 @@ export const MODEL_FACTS = [
294
319
  {
295
320
  id: 'seed-audio',
296
321
  route: 'generate',
322
+ tier: 'default',
297
323
  label: 'Seed Audio 1.0',
298
324
  kind: 'audio',
299
325
  // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
@@ -303,6 +329,7 @@ export const MODEL_FACTS = [
303
329
  {
304
330
  id: 'eleven-sfx',
305
331
  route: 'generate',
332
+ tier: 'specialist',
306
333
  label: 'ElevenLabs Sound Effects v2',
307
334
  kind: 'audio',
308
335
  ...caps('eleven-sfx'),
@@ -311,12 +338,30 @@ export const MODEL_FACTS = [
311
338
  {
312
339
  id: 'inworld-tts-2',
313
340
  route: 'generate',
341
+ tier: 'specialist',
314
342
  label: 'Inworld Realtime TTS-2',
315
343
  kind: 'audio',
316
344
  ...caps('inworld-tts-2'),
317
345
  notes: 'THE VOICE SEAT — one named voice saying one line, billed per CHARACTER not per second. Route here when WHO is speaking matters. NOT scene audio — that is seed-audio; a single effect is eleven-sfx.',
318
346
  },
319
347
  ];
348
+ /**
349
+ * One default per lane, asserted where the data is declared. Two rows both
350
+ * reading "DEFAULT" is exactly the drift `tier` exists to make impossible, and
351
+ * a lane with no default leaves an agent (and slates-web) nothing to lead with.
352
+ */
353
+ for (const kind of ['image', 'video', 'audio']) {
354
+ for (const route of ['generate', 'edit']) {
355
+ const lane = MODEL_FACTS.filter((f) => f.kind === kind && f.route === route);
356
+ if (lane.length === 0)
357
+ continue;
358
+ const defaults = lane.filter((f) => f.tier === 'default').map((f) => f.id);
359
+ if (defaults.length !== 1) {
360
+ throw new Error(`MODEL_FACTS: ${kind}/${route} must have exactly one tier: 'default' row, found ${defaults.length}` +
361
+ (defaults.length ? ` (${defaults.join(', ')})` : ''));
362
+ }
363
+ }
364
+ }
320
365
  const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
321
366
  /**
322
367
  * Routing prose for one lane, generated from the SSOT.
@@ -115,7 +115,7 @@ const SEEDANCE_25 = {
115
115
  ...SEEDANCE,
116
116
  label: 'Seedance 2.5',
117
117
  intro: [
118
- 'Seedance 2.5 is a SECOND SEAT next to 2.0, not an upgrade of it. It buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), audio-only references and integer-second timestamps — and it gives up 4K. It runs at 480p, 720p or 1080p, and it costs more than 2.0 at every tier they share, so pick 2.0 when you want the same resolution cheaper or you want 4K.',
118
+ 'Seedance 2.5 is the default video model. Against 2.0 it buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), audio-only references and integer-second timestamps — and it gives up 4K. It runs at 480p, 720p or 1080p, and it costs more than 2.0 at every tier they share, so pick 2.0 when you want the same resolution cheaper or you want 4K.',
119
119
  "ByteDance's official advanced formula has 8 slots: precise subject + action details + scene/environment + lighting & color tone + camera movement + visual style + image quality + constraints. Sweet spot 60-150 words for a single shot, longer for multi-shot.",
120
120
  ],
121
121
  columns: [
@@ -348,7 +348,7 @@ const OMNI_FLASH = {
348
348
  },
349
349
  {
350
350
  heading: 'Know its seat',
351
- note: 'Cheap drafts, iteration volume, and audio-in-one-gen at low cost. For hero shots, Kling 3.0 (general default) or Seedance 2.0 (premium/physics) still win.',
351
+ note: 'Cheap drafts, iteration volume, and audio-in-one-gen at low cost. For hero shots, Seedance 2.5 (the default) or Seedance 2.0 (4K, cheaper) still win.',
352
352
  },
353
353
  ],
354
354
  ],