@slatesvideo/shared 0.6.9 → 0.6.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -100,6 +100,105 @@ const MINIMAX_H3_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
100
100
  * to lose track of what was actually generated.
101
101
  */
102
102
  const LTX_2_5_ASPECT_RATIOS = ['16:9', '9:16'];
103
+ export const GPT_QUALITY_TIERS = ['low', 'medium', 'high', 'xhigh', 'max'];
104
+ export const GPT_BACKGROUNDS = ['auto', 'transparent', 'opaque'];
105
+ // Product output sizes; schema bounds and metering receipt live in the GPT harvest.
106
+ export const GPT_IMAGE_25_SIZES = {
107
+ '1k': {
108
+ '1:1': { width: 1024, height: 1024 },
109
+ '16:9': { width: 1360, height: 768 },
110
+ '9:16': { width: 768, height: 1360 },
111
+ '4:3': { width: 1168, height: 880 },
112
+ '3:4': { width: 880, height: 1168 },
113
+ },
114
+ '2k': {
115
+ '1:1': { width: 1440, height: 1440 },
116
+ '16:9': { width: 1920, height: 1080 },
117
+ '9:16': { width: 1080, height: 1920 },
118
+ '4:3': { width: 1664, height: 1248 },
119
+ '3:4': { width: 1248, height: 1664 },
120
+ },
121
+ '3k': {
122
+ '1:1': { width: 1920, height: 1920 },
123
+ '16:9': { width: 2560, height: 1440 },
124
+ '9:16': { width: 1440, height: 2560 },
125
+ '4:3': { width: 2224, height: 1664 },
126
+ '3:4': { width: 1664, height: 2224 },
127
+ },
128
+ '4k': {
129
+ // 1:1 and 16:9 sit EXACTLY on the 8,294,400 ceiling — 3840×2160 is one of
130
+ // fal's own priced sizes, so the bound is inclusive. 4:3 / 3:4 are the two
131
+ // that had to move; see the constraint note above.
132
+ '1:1': { width: 2880, height: 2880 },
133
+ '16:9': { width: 3840, height: 2160 },
134
+ '9:16': { width: 2160, height: 3840 },
135
+ '4:3': { width: 3264, height: 2448 },
136
+ '3:4': { width: 2448, height: 3264 },
137
+ },
138
+ };
139
+ /**
140
+ * fal's named ~1MP presets per aspect ratio, with custom dims where fal has no
141
+ * preset. The `1k` rung of every non-GPT image model resolves through this.
142
+ */
143
+ export const FAL_1MP_SIZES = {
144
+ '1:1': 'square_hd',
145
+ '4:3': 'landscape_4_3',
146
+ '3:4': 'portrait_4_3',
147
+ '16:9': 'landscape_16_9',
148
+ '9:16': 'portrait_16_9',
149
+ '2:3': { width: 832, height: 1248 },
150
+ '3:2': { width: 1248, height: 832 },
151
+ '4:5': { width: 896, height: 1120 },
152
+ '5:4': { width: 1120, height: 896 },
153
+ '21:9': { width: 1344, height: 576 },
154
+ };
155
+ /** Pixel dims for a megapixel target at an aspect ratio, rounded to multiples of 8. */
156
+ export function computeFalDimensions(aspectRatio, targetMP) {
157
+ const parts = aspectRatio.split(':').map(Number);
158
+ const w = parts[0] || 16;
159
+ const h = parts[1] || 9;
160
+ const ratio = w / h;
161
+ const targetPixels = targetMP * 1_000_000;
162
+ return {
163
+ width: Math.round(Math.sqrt(targetPixels * ratio) / 8) * 8,
164
+ height: Math.round(Math.sqrt(targetPixels / ratio) / 8) * 8,
165
+ };
166
+ }
167
+ /**
168
+ * The `image_size` a non-GPT fal image request carries, for one model × aspect ×
169
+ * resolution rung.
170
+ *
171
+ * 🚨 THIS IS A BILLING INPUT, WHICH IS WHY IT LIVES HERE (moved out of
172
+ * slate/src/main/api/fal.ts, 2026-09-10). The resolution rung is a segment of
173
+ * every image cost key, and nothing in the REQUEST names it — fal is told pixel
174
+ * dimensions, not "2k". The proxy therefore recovers the rung by running this
175
+ * function over the model's declared `imageResolutions` × `aspectRatios` and
176
+ * matching the body's `image_size`, exactly as it recovers a GPT Image rung from
177
+ * `GPT_IMAGE_25_SIZES`. A second copy of this arithmetic would mean the desktop
178
+ * and the server could disagree about what a request is worth, silently.
179
+ *
180
+ * GPT Image does NOT come through here — that family carries explicit pixel
181
+ * classes in `GPT_IMAGE_25_SIZES` and an explicit `quality` rung.
182
+ */
183
+ export function falImageSize(model, aspectRatio, resolution) {
184
+ const ar = aspectRatio || '16:9';
185
+ const isSeedream5 = model === 'seedream-5-lite';
186
+ const res = resolution || (isSeedream5 ? '2k' : '1k');
187
+ // 1K (~1MP): fal's named presets, or small custom dims where there is none.
188
+ if (res === '1k') {
189
+ return FAL_1MP_SIZES[ar] || 'landscape_16_9';
190
+ }
191
+ // Seedream 5 Lite: custom dims must be ≥3.69MP (2560×1440) and ≤9.44MP
192
+ // (3072×3072), so its three rungs target 4 / 7 / 9 MP — all inside that band.
193
+ if (isSeedream5) {
194
+ if (res === '4k')
195
+ return computeFalDimensions(ar, 9);
196
+ return res === '3k' ? computeFalDimensions(ar, 7) : computeFalDimensions(ar, 4);
197
+ }
198
+ if (res === '2k')
199
+ return computeFalDimensions(ar, 2);
200
+ return computeFalDimensions(ar, 4);
201
+ }
103
202
  /**
104
203
  * The provider every AGENT generation actually lands on for Kling and Veo.
105
204
  *
@@ -122,14 +221,22 @@ export const AGENT_ROUTE_PROVIDER = 'fal';
122
221
  export const MODEL_CAPABILITIES = {
123
222
  // ── Image models ───────────────────────────────────────────────────────────
124
223
  'nano-banana-2': {
224
+ imageResolutions: ['1k', '2k', '4k'],
225
+ // fal's nano-banana-2 schema caps `num_images` at 4 (read 2026-09-09). This
226
+ // is the ONLY model that batches: the MCP's headless path (no projectId) asks
227
+ // fal for one batch, and every other route — desktop and agent alike — fires
228
+ // N separate single-image generations. The proxy bills the batch size.
229
+ maxBatchImages: 4,
125
230
  aspectRatios: FULL_ASPECT_RATIOS,
126
231
  maxRefImages: 14,
127
232
  },
128
233
  'nano-banana-2-lite': {
234
+ imageResolutions: ['1k'],
129
235
  aspectRatios: FULL_ASPECT_RATIOS,
130
236
  maxRefImages: 4, // fal edit endpoint caps input images at 4
131
237
  },
132
238
  'nano-banana-pro': {
239
+ imageResolutions: ['1k', '2k', '4k'],
133
240
  aspectRatios: FULL_ASPECT_RATIOS,
134
241
  maxRefImages: 14,
135
242
  },
@@ -169,18 +276,22 @@ export const MODEL_CAPABILITIES = {
169
276
  // limits either: the MCP's 4,000-character prompt against fal's 32,000, and
170
277
  // image quantity, which is a fan-out and has no provider ceiling at all.
171
278
  'gpt-image-2-5-flare': {
279
+ imageResolutions: ['2k', '3k', '4k'],
172
280
  aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
173
281
  maxRefImages: 16,
174
282
  },
175
283
  'gpt-image-2-5-sunburst': {
284
+ imageResolutions: ['2k', '3k', '4k'],
176
285
  aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
177
286
  maxRefImages: 16,
178
287
  },
179
288
  'flux-2-max': {
289
+ imageResolutions: ['1k', '2k', '4k'],
180
290
  aspectRatios: FULL_ASPECT_RATIOS,
181
291
  maxRefImages: 4,
182
292
  },
183
293
  'seedream-5-lite': {
294
+ imageResolutions: ['2k', '3k', '4k'],
184
295
  aspectRatios: FULL_ASPECT_RATIOS,
185
296
  maxRefImages: 10,
186
297
  },
@@ -377,6 +488,9 @@ export const MODEL_CAPABILITIES = {
377
488
  // the base row's branch — a different ladder AND a different price at the one
378
489
  // tier they share. Every lookup downstream is an exact-id map, not a prefix.
379
490
  'minimax-h3': {
491
+ // fal reference-to-video schema, 2026-09-09: each audio clip is 2-15s.
492
+ referenceAudioDuration: { min: 2, max: 15 },
493
+ referenceVideoDuration: { min: 2, max: 15 },
380
494
  aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
381
495
  // The full ladder. 480p/768p are NATIVE generation modes; 2K and 4K upscale
382
496
  // a 768p base result through H3-Regenerate-2K, which is API-only and not in
@@ -408,6 +522,9 @@ export const MODEL_CAPABILITIES = {
408
522
  maxReferenceAudioSeconds: 15,
409
523
  },
410
524
  'minimax-h3-max': {
525
+ // fal reference-to-video schema, 2026-09-09: each audio clip is 2-15s.
526
+ referenceAudioDuration: { min: 2, max: 15 },
527
+ referenceVideoDuration: { min: 2, max: 15 },
411
528
  aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
412
529
  // 🚨 REFERENCES LANDED 2026-09-09, AFTER A FALSE CLAIM WAS RETIRED. This row
413
530
  // shipped from v1.5.5 declaring zero reference capacity because a comment
@@ -431,15 +548,7 @@ export const MODEL_CAPABILITIES = {
431
548
  // arms — quoted off this endpoint, not inherited.
432
549
  maxReferenceVideoSeconds: 15,
433
550
  maxReferenceAudioSeconds: 15,
434
- // 1080P IS REAL ON THIS ROW and was missing until 2026-09-09. The schema's
435
- // resolution enum is ["480P","768P","1080P"] on all three h3-max endpoints.
436
- // 2K/4K genuinely are absent: the H3-Regenerate-2K upscaler is API-only and
437
- // is not in the open weights fal self-hosts, which is the actual mechanism
438
- // behind the shorter ladder — 1080p was never part of that story.
439
- //
440
- // DEFAULT stays 768p: it is the tier the model natively generates, and
441
- // 1080p is a 2x price step ($0.160/s against $0.080/s).
442
- videoResolution: { options: ['480p', '768p', '1080p'], default: '768p' },
551
+ videoResolution: { options: ['480p', '768p'], default: '768p' },
443
552
  duration: { min: 5, max: 15, mode: 'continuous' },
444
553
  },
445
554
  // ── LTX-2.5 (both seats on fal — added 2026-08-29) ─────────────────────────
@@ -817,4 +926,28 @@ export function describeReferenceImageCaps(models) {
817
926
  return n === 0 ? '0 (prompt + source clip only)' : String(n);
818
927
  });
819
928
  }
929
+ /** H3 Max reference accounting, fal's worked tables read 2026-09-09.
930
+ * https://fal.ai/models/minimax/h3-max/reference-to-video
931
+ * 1080p video-reference pricing is unpublished; never infer it from output rates.
932
+ */
933
+ export const MINIMAX_MAX_REFERENCE = {
934
+ freeTokens: 4096,
935
+ imagePixelsPerToken: 1024,
936
+ normalizedImageEdge: 1024,
937
+ audioTokensPerSecond: 80,
938
+ videoTokensPerSecond: { '480p': 2886, '768p': 7459.2 },
939
+ };
940
+ export function minimaxMaxReferenceTokens(input) {
941
+ const rate = MINIMAX_MAX_REFERENCE.videoTokensPerSecond[input.resolution];
942
+ if (input.videoSeconds > 0 && rate === undefined) {
943
+ throw new Error(`H3 Max video-reference pricing is unavailable at ${input.resolution}; choose a priced resolution.`);
944
+ }
945
+ for (const n of [input.imagePixels, input.videoSeconds, input.audioSeconds]) {
946
+ if (!Number.isFinite(n) || n < 0)
947
+ throw new Error('Reference metadata must be finite and nonnegative');
948
+ }
949
+ return Math.max(0, Math.ceil(input.imagePixels / MINIMAX_MAX_REFERENCE.imagePixelsPerToken +
950
+ input.videoSeconds * (rate ?? 0) + input.audioSeconds * MINIMAX_MAX_REFERENCE.audioTokensPerSecond -
951
+ MINIMAX_MAX_REFERENCE.freeTokens));
952
+ }
820
953
  //# sourceMappingURL=model-capabilities.js.map
@@ -10,6 +10,33 @@ export interface ModelFact {
10
10
  * only while that substring is unique, which is how isOmniFlashModel broke.
11
11
  */
12
12
  route: 'generate' | 'edit';
13
+ /**
14
+ * Where this seat sits in the routing story, as DATA rather than as a word
15
+ * inside `notes`. `default` is the seat an agent (or a web page) reaches for
16
+ * when nothing about the shot argues otherwise: exactly ONE per kind per
17
+ * route, asserted at module load below. `specialist` is picked for a named
18
+ * reason the notes give (premium, speed, volume, audio, cheapest, edit fidelity).
19
+ * `niche` is never the default and never headlines; the notes say why.
20
+ *
21
+ * WHY A FIELD: "DEFAULT" lived only as a word inside notes. The 2026-08-10
22
+ * Seedance 2.5 commit wrote "the DEFAULT video model" into Seedance 2.0's note
23
+ * meaning the default SEEDANCE seat (a bare "seedance" resolves to 2.0), and for
24
+ * a month two rows read as the default while the routing doctrine then in force
25
+ * (Eric, 2026-07-03: Kling 3.0 the general-purpose default, Seedance the premium
26
+ * escalation) never changed. The marketing site meanwhile headlined Veo, a
27
+ * `niche` row, because no check could read a tier out of prose. The tier is
28
+ * DATA now; the notes describe, they do not rank.
29
+ *
30
+ * CURRENT DOCTRINE (Eric, 2026-09-13, superseding 2026-07-03): SEEDANCE 2.5 IS
31
+ * THE DEFAULT VIDEO MODEL, in the app picker and in agent routing — "it's the
32
+ * best in the world". 2.0 is the specialist for native 4K and for the same
33
+ * resolution cheaper; Kling is the specialist for cost-effective start-frame,
34
+ * performance and lip-sync work and the only engine behind Motion Transfer and
35
+ * Lip Sync. Recorded in the vault's prompting-ssot.md the same day. Changing the
36
+ * default again lands there first, then here, then in slates-model-selection.md. slates-web reads this to order its model lineup
37
+ * and to fail its build when a niche seat is named more often than the default.
38
+ */
39
+ tier: 'default' | 'specialist' | 'niche';
13
40
  /** Max reference images (image models) — null if not applicable. */
14
41
  maxRefImages: number | null;
15
42
  /** Max ingredient images (video models) — null if not applicable. */
@@ -69,14 +69,14 @@ export function multimodalRefSummary(id) {
69
69
  return '';
70
70
  const parts = [];
71
71
  if (v > 0)
72
- parts.push(`${v} reference video${v === 1 ? '' : 's'} (${f.maxReferenceVideoSeconds}s combined)`);
72
+ parts.push(`${v} video${v === 1 ? '' : 's'} (${f.maxReferenceVideoSeconds}s total)`);
73
73
  if (a > 0)
74
- parts.push(`${a} reference audio clip${a === 1 ? '' : 's'} (${f.maxReferenceAudioSeconds}s combined)`);
75
- const total = f.maxReferenceFilesTotal ? `, ${f.maxReferenceFilesTotal} files max across all modalities` : '';
74
+ parts.push(`${a} audio clip${a === 1 ? '' : 's'} (${f.maxReferenceAudioSeconds}s total)`);
75
+ const total = f.maxReferenceFilesTotal ? `, ${f.maxReferenceFilesTotal} files max` : '';
76
76
  const companion = f.audioRefNeedsCompanion
77
- ? ' Audio needs at least one image or video reference alongside it.'
78
- : ' Audio-only references are allowed.';
79
- return `${f.label}: up to ${parts.join(' and ')}${total}.${companion}`;
77
+ ? '; audio needs an image or video alongside.'
78
+ : '; audio-only is allowed.';
79
+ return `${f.label}: up to ${parts.join(' + ')}${total}${companion}`;
80
80
  }
81
81
  /**
82
82
  * The prompt words that make Seedance 2.5 reclassify a reference-carrying
@@ -120,6 +120,7 @@ export const MODEL_FACTS = [
120
120
  {
121
121
  id: 'nano-banana-2',
122
122
  route: 'generate',
123
+ tier: 'default',
123
124
  // Gemini 3.1 FLASH Image — verified against the runtime slug map in
124
125
  // slate/src/main/api/google.ts. Nano Banana PRO is a different model
125
126
  // (gemini-3-pro-image-preview); do not conflate them.
@@ -132,6 +133,7 @@ export const MODEL_FACTS = [
132
133
  {
133
134
  id: 'nano-banana-2-lite',
134
135
  route: 'generate',
136
+ tier: 'specialist',
135
137
  label: 'Nano Banana 2 Lite',
136
138
  kind: 'image',
137
139
  ...caps('nano-banana-2-lite'),
@@ -140,6 +142,7 @@ export const MODEL_FACTS = [
140
142
  {
141
143
  id: 'nano-banana-pro',
142
144
  route: 'generate',
145
+ tier: 'specialist',
143
146
  label: 'Nano Banana Pro',
144
147
  kind: 'image',
145
148
  ...caps('nano-banana-pro'),
@@ -148,6 +151,7 @@ export const MODEL_FACTS = [
148
151
  {
149
152
  id: 'gpt-image-2-5-flare',
150
153
  route: 'generate',
154
+ tier: 'specialist',
151
155
  label: 'GPT Image 2.5 Flare',
152
156
  kind: 'image',
153
157
  ...caps('gpt-image-2-5-flare'),
@@ -156,6 +160,7 @@ export const MODEL_FACTS = [
156
160
  {
157
161
  id: 'gpt-image-2-5-sunburst',
158
162
  route: 'generate',
163
+ tier: 'specialist',
159
164
  label: 'GPT Image 2.5 Sunburst',
160
165
  kind: 'image',
161
166
  ...caps('gpt-image-2-5-sunburst'),
@@ -164,6 +169,7 @@ export const MODEL_FACTS = [
164
169
  {
165
170
  id: 'flux-2-max',
166
171
  route: 'generate',
172
+ tier: 'specialist',
167
173
  label: 'FLUX.2 Max',
168
174
  kind: 'image',
169
175
  ...caps('flux-2-max'),
@@ -172,6 +178,7 @@ export const MODEL_FACTS = [
172
178
  {
173
179
  id: 'seedream-5-lite',
174
180
  route: 'generate',
181
+ tier: 'specialist',
175
182
  label: 'Seedream 5 Lite',
176
183
  kind: 'image',
177
184
  ...caps('seedream-5-lite'),
@@ -180,25 +187,28 @@ export const MODEL_FACTS = [
180
187
  {
181
188
  id: 'seedance-2',
182
189
  route: 'generate',
190
+ tier: 'specialist',
183
191
  label: 'Seedance 2.0',
184
192
  kind: 'video',
185
193
  ...caps('seedance-2'),
186
194
  audioRefNeedsCompanion: true,
187
- notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction or scale matter, and for hero shots. VIDEO-ONLY. Strong image-to-video and own-footage restyle. 4K is Pro-gated (base accounts get PRO_REQUIRED). Stays the default over 2.5: it is the only Seedance with native 4K and it is cheaper at every tier the two share.',
195
+ notes: 'THE 4K AND VALUE SEAT beside the 2.5 default — the only Seedance with native 4K (Pro-gated; base accounts get PRO_REQUIRED) and cheaper than 2.5 at every resolution they share, with the same physics, effects and scale strengths; shorter takes, fewer references, no timestamps. VIDEO-ONLY. A bare "seedance" still resolves here for older CLIs that expect 4K.',
188
196
  },
189
197
  {
190
198
  id: 'seedance-2.5',
191
199
  route: 'generate',
200
+ tier: 'default',
192
201
  label: 'Seedance 2.5',
193
202
  kind: 'video',
194
203
  ...caps('seedance-2.5'),
195
204
  // No companion requirement — audio-only references are one of the things
196
205
  // the second seat actually buys.
197
- notes: 'A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE — and the dearer one at every tier they share. Pick 2.5 when the shot needs LENGTH, MANY references, an AUDIO-ONLY reference, TIMED BEATS, or tighter prompt adherence; pick 2.0 for 4K and for the same resolution cheaper. VIDEO-ONLY. Timestamp grammar, and the edit/extend words that make the provider reclassify a fresh generation and fail it, are in slates-prompting-seedance-2-5.',
206
+ notes: 'DEFAULT VIDEO MODEL — the strongest seat for physics, effects, scale and hero shots, and the only Seedance that takes long single takes, many references, audio-only references and integer-second timestamps. No 4K, and dearer than 2.0 at every shared resolution: go to 2.0 for 4K or the same resolution cheaper. LENGTH is the price dial — quote long takes first. VIDEO-ONLY. Timestamp grammar and the edit/extend words that make the provider reclassify and fail a generation are in slates-prompting-seedance-2-5.',
198
207
  },
199
208
  {
200
209
  id: 'seedance-2.5-edit',
201
210
  route: 'edit',
211
+ tier: 'specialist',
202
212
  label: 'Seedance 2.5 Edit',
203
213
  kind: 'video',
204
214
  // 0 ingredients: prompt + source clip only on slates_edit_video.
@@ -208,15 +218,17 @@ export const MODEL_FACTS = [
208
218
  {
209
219
  id: 'kling-v3',
210
220
  route: 'generate',
221
+ tier: 'specialist',
211
222
  label: 'Kling 3.0',
212
223
  kind: 'video',
213
224
  // Family-level fact — caps are identical across std/pro/omni/omni-pro.
214
225
  ...caps('kling-v3.0-std'),
215
- notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync, and the widest aspect-ratio set. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.',
226
+ notes: 'THE COST-EFFECTIVE SEAT — strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync and the widest aspect-ratio set; pick it when the budget matters and the shot is a performance or a start-frame animation. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.',
216
227
  },
217
228
  {
218
229
  id: 'kling-v3-edit',
219
230
  route: 'edit',
231
+ tier: 'specialist',
220
232
  label: 'Kling O3 Video Edit',
221
233
  kind: 'video',
222
234
  // Family-level fact; 4 = combined subject elements + style refs per edit.
@@ -226,15 +238,17 @@ export const MODEL_FACTS = [
226
238
  {
227
239
  id: 'veo-3.1',
228
240
  route: 'generate',
241
+ tier: 'niche',
229
242
  label: 'Veo 3.1',
230
243
  kind: 'video',
231
244
  // Family-level fact — fast and standard declare the same caps.
232
245
  ...caps('veo-3.1-fast'),
233
- notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one pass, and the narrowest aspect-ratio and duration sets in the catalogue are acceptable. Otherwise Kling (default) or Seedance (physics/premium) win.',
246
+ notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one pass, and the narrowest aspect-ratio and duration sets in the catalogue are acceptable. Otherwise Seedance 2.5 (the default) or Kling (cost-effective performance) win.',
234
247
  },
235
248
  {
236
249
  id: 'omni-flash',
237
250
  route: 'generate',
251
+ tier: 'specialist',
238
252
  label: 'Gemini Omni Flash',
239
253
  kind: 'video',
240
254
  // 7 ref2v image_urls — mirrors Google's own reference limit.
@@ -244,6 +258,7 @@ export const MODEL_FACTS = [
244
258
  {
245
259
  id: 'omni-flash-edit',
246
260
  route: 'edit',
261
+ tier: 'default',
247
262
  label: 'Omni Flash Edit',
248
263
  kind: 'video',
249
264
  // 0: prompt + source clip ONLY — no element/style refs on this endpoint.
@@ -253,6 +268,7 @@ export const MODEL_FACTS = [
253
268
  {
254
269
  id: 'minimax-h3',
255
270
  route: 'generate',
271
+ tier: 'specialist',
256
272
  label: 'MiniMax H3',
257
273
  kind: 'video',
258
274
  ...caps('minimax-h3'),
@@ -265,16 +281,24 @@ export const MODEL_FACTS = [
265
281
  {
266
282
  id: 'minimax-h3-max',
267
283
  route: 'generate',
284
+ tier: 'specialist',
268
285
  label: 'MiniMax H3 Max',
269
286
  kind: 'video',
270
- // No reference caps: fal publishes no reference-to-video endpoint for this
271
- // row, so `caps()` returns nulls and the composer refuses references.
287
+ // References landed 2026-09-09 when the "h3-max/reference-to-video returns
288
+ // 404" claim was retired against the fetched schema (caps come from
289
+ // model-capabilities.ts; the rip is second-brain/business/projects/slates/
290
+ // provider-docs/fal-minimax-h3-openapi-schemas.md, section 6).
272
291
  ...caps('minimax-h3-max'),
273
- notes: 'THE SPEED SEAT, and the DEARER one at the tier they share — never the cheap H3 and never the default. fal\'s post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It gives up the upper resolution tiers and the REFERENCE endpoint, so the omni-reference set is base-H3 only — but it still animates start and end frames, which is one of the two things it is FOR. Never describe this row as taking no image input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.',
292
+ // Quoted off THAT endpoint's reference_audio_urls description, not copied
293
+ // from the base row: "Audio cannot be the only reference input; provide at
294
+ // least one reference image or video with it."
295
+ audioRefNeedsCompanion: true,
296
+ notes: 'THE SPEED SEAT, and the DEARER one at the tier they share — never the cheap H3 and never the default. fal\'s post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It gives up the upper resolution tiers. It takes the same omni-reference set as base H3 and animates start and end frames — but not both in one call: its reference endpoint has no start/end-frame fields, where base H3\'s does. Never describe this row as taking no image or reference input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.',
274
297
  },
275
298
  {
276
299
  id: 'ltx-2-5',
277
300
  route: 'generate',
301
+ tier: 'specialist',
278
302
  label: 'LTX-2.5',
279
303
  kind: 'video',
280
304
  // No reference caps: fal publishes text-to-video and image-to-video for LTX
@@ -286,6 +310,7 @@ export const MODEL_FACTS = [
286
310
  {
287
311
  id: 'ltx-2-5-pro',
288
312
  route: 'generate',
313
+ tier: 'specialist',
289
314
  label: 'LTX-2.5 Pro',
290
315
  kind: 'video',
291
316
  ...caps('ltx-2-5-pro'),
@@ -294,6 +319,7 @@ export const MODEL_FACTS = [
294
319
  {
295
320
  id: 'seed-audio',
296
321
  route: 'generate',
322
+ tier: 'default',
297
323
  label: 'Seed Audio 1.0',
298
324
  kind: 'audio',
299
325
  // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
@@ -303,6 +329,7 @@ export const MODEL_FACTS = [
303
329
  {
304
330
  id: 'eleven-sfx',
305
331
  route: 'generate',
332
+ tier: 'specialist',
306
333
  label: 'ElevenLabs Sound Effects v2',
307
334
  kind: 'audio',
308
335
  ...caps('eleven-sfx'),
@@ -311,12 +338,30 @@ export const MODEL_FACTS = [
311
338
  {
312
339
  id: 'inworld-tts-2',
313
340
  route: 'generate',
341
+ tier: 'specialist',
314
342
  label: 'Inworld Realtime TTS-2',
315
343
  kind: 'audio',
316
344
  ...caps('inworld-tts-2'),
317
345
  notes: 'THE VOICE SEAT — one named voice saying one line, billed per CHARACTER not per second. Route here when WHO is speaking matters. NOT scene audio — that is seed-audio; a single effect is eleven-sfx.',
318
346
  },
319
347
  ];
348
+ /**
349
+ * One default per lane, asserted where the data is declared. Two rows both
350
+ * reading "DEFAULT" is exactly the drift `tier` exists to make impossible, and
351
+ * a lane with no default leaves an agent (and slates-web) nothing to lead with.
352
+ */
353
+ for (const kind of ['image', 'video', 'audio']) {
354
+ for (const route of ['generate', 'edit']) {
355
+ const lane = MODEL_FACTS.filter((f) => f.kind === kind && f.route === route);
356
+ if (lane.length === 0)
357
+ continue;
358
+ const defaults = lane.filter((f) => f.tier === 'default').map((f) => f.id);
359
+ if (defaults.length !== 1) {
360
+ throw new Error(`MODEL_FACTS: ${kind}/${route} must have exactly one tier: 'default' row, found ${defaults.length}` +
361
+ (defaults.length ? ` (${defaults.join(', ')})` : ''));
362
+ }
363
+ }
364
+ }
320
365
  const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
321
366
  /**
322
367
  * Routing prose for one lane, generated from the SSOT.
@@ -115,7 +115,7 @@ const SEEDANCE_25 = {
115
115
  ...SEEDANCE,
116
116
  label: 'Seedance 2.5',
117
117
  intro: [
118
- 'Seedance 2.5 is a SECOND SEAT next to 2.0, not an upgrade of it. It buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), audio-only references and integer-second timestamps — and it gives up 4K. It runs at 480p, 720p or 1080p, and it costs more than 2.0 at every tier they share, so pick 2.0 when you want the same resolution cheaper or you want 4K.',
118
+ 'Seedance 2.5 is the default video model. Against 2.0 it buys one 30-second take instead of 15, up to 30 image references (plus 10 video and 10 audio), audio-only references and integer-second timestamps — and it gives up 4K. It runs at 480p, 720p or 1080p, and it costs more than 2.0 at every tier they share, so pick 2.0 when you want the same resolution cheaper or you want 4K.',
119
119
  "ByteDance's official advanced formula has 8 slots: precise subject + action details + scene/environment + lighting & color tone + camera movement + visual style + image quality + constraints. Sweet spot 60-150 words for a single shot, longer for multi-shot.",
120
120
  ],
121
121
  columns: [
@@ -348,7 +348,7 @@ const OMNI_FLASH = {
348
348
  },
349
349
  {
350
350
  heading: 'Know its seat',
351
- note: 'Cheap drafts, iteration volume, and audio-in-one-gen at low cost. For hero shots, Kling 3.0 (general default) or Seedance 2.0 (premium/physics) still win.',
351
+ note: 'Cheap drafts, iteration volume, and audio-in-one-gen at low cost. For hero shots, Seedance 2.5 (the default) or Seedance 2.0 (4K, cheaper) still win.',
352
352
  },
353
353
  ],
354
354
  ],
@@ -14,7 +14,21 @@ export interface ReferenceGroup {
14
14
  /** Display + citation name: 'Marcus' | 'the cafe' | 'noir'. Used verbatim. */
15
15
  name: string;
16
16
  kind: ReferenceKind;
17
- /** A group can carry several images for workflows that genuinely need them. */
17
+ /**
18
+ * A group can carry several images for workflows that genuinely need them.
19
+ *
20
+ * 🚨 A `character` GROUP MAY ALSO CARRY ONE `audio` MEDIUM — that character's
21
+ * assigned VOICE (2026-09-09). It is the same idea as the identity image, on
22
+ * the other axis of identity: the mention attaches what the character IS, and
23
+ * a voice is part of that. The audio takes its number from the audio counter,
24
+ * so adding one renumbers no image, and it is cited INLINE beside her name
25
+ * ("Sarah (image 1, voice timbre from audio 1)") rather than as the neutral
26
+ * `audio-ref` sentence — step 3e holds the receipt for why that is legal.
27
+ *
28
+ * A character group with a voice and NO identity image is a real state, not a
29
+ * defect: `voice-without-photo` is legal on any model that reads audio alone
30
+ * (Seedance 2.5). It cites no image and the timbre line carries the binding.
31
+ */
18
32
  media: ReferenceMedia[];
19
33
  /**
20
34
  * What is SAID in this group's reference audio, typed by the user.