@slatesvideo/shared 0.5.10 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,7 +29,7 @@ export declare const getCreditBalance: Operation<Record<string, never>>;
29
29
  export declare const listAvailableModels: Operation<{
30
30
  filter?: string;
31
31
  }>;
32
- export declare const VIDEO_MODELS: readonly ["kling-v3.0-std", "kling-v3.0-pro", "kling-v3.0-omni", "veo-3.1-fast", "veo-3.1-standard", "seedance-2", "seedance-2.5", "omni-flash"];
32
+ export declare const VIDEO_MODELS: readonly ["kling-v3.0-std", "kling-v3.0-pro", "kling-v3.0-omni", "veo-3.1-fast", "veo-3.1-standard", "seedance-2", "seedance-2.5", "omni-flash", "minimax-h3", "minimax-h3-max"];
33
33
  type VideoModel = (typeof VIDEO_MODELS)[number];
34
34
  /** The exact `model` ids `slates_edit_video` accepts. Edit rows are deliberately
35
35
  * NOT in VIDEO_MODELS — they take a source clip, not frames. */
@@ -39,12 +39,15 @@ export declare const estimateGenerationCost: Operation<{
39
39
  model: string;
40
40
  quantity?: number;
41
41
  duration?: number;
42
- videoResolution?: '480p' | '720p' | '1080p' | '4k';
42
+ /** The FULL union — the runtime Zod enum is generated from MODEL_CAPABILITIES
43
+ * and `assertVideoCapabilities` is the per-model narrowing authority. */
44
+ videoResolution?: VideoResolution;
43
45
  resolution?: '1k' | '2k' | '3k' | '4k';
44
46
  quality?: 'medium' | 'high';
45
47
  sound?: boolean;
46
48
  seedanceFace?: boolean;
47
49
  seedanceRealFace?: boolean;
50
+ referenceImages?: number;
48
51
  }>;
49
52
  export declare const listProjects: Operation<Record<string, never>>;
50
53
  export declare const createProject: Operation<{
@@ -193,7 +196,9 @@ export declare const editImage: Operation<{
193
196
  export declare function videoCostKey(input: {
194
197
  model: VideoModel;
195
198
  duration: number;
196
- videoResolution?: '480p' | '720p' | '1080p' | '4k';
199
+ /** Widened to the full union because the Zod enum is GENERATED from
200
+ * MODEL_CAPABILITIES; the runtime schema is the narrowing authority. */
201
+ videoResolution?: VideoResolution;
197
202
  sound?: boolean;
198
203
  seedanceFace?: boolean;
199
204
  seedanceRealFace?: boolean;
@@ -201,6 +206,11 @@ export declare function videoCostKey(input: {
201
206
  * transfer / lip-sync on video / relocate). >0 bills the vref key on TOTAL
202
207
  * (input+output) seconds — the server re-derives this by probing the refs. */
203
208
  videoRefSeconds?: number;
209
+ /** MiniMax H3 base only: how many reference IMAGES the request carries. Past
210
+ * the fifth, fal charges per image, so it is a KEY DIMENSION — omit it on a
211
+ * reference-heavy call and the quote under-reports what the proxy bills
212
+ * (which it re-derives from the request itself). */
213
+ referenceImages?: number;
204
214
  }): string;
205
215
  export declare function klingEditCostKey(model: 'kling-v3.0-omni-edit' | 'kling-v3.0-omni-pro-edit', duration: number): string;
206
216
  export declare function omniFlashEditCostKey(duration: number): string;
@@ -211,7 +221,9 @@ export declare const SEEDANCE_25_EDIT_MIN_SECONDS: number;
211
221
  export declare const SEEDANCE_25_EDIT_MAX_SECONDS: number;
212
222
  export declare function seedanceEditCostKey(input: {
213
223
  duration: number;
214
- videoResolution?: '480p' | '720p';
224
+ /** Widened to the full union because the Zod enum is GENERATED from
225
+ * MODEL_CAPABILITIES; the runtime schema is the narrowing authority. */
226
+ videoResolution?: VideoResolution;
215
227
  seedanceFace?: boolean;
216
228
  seedanceRealFace?: boolean;
217
229
  }): string;
@@ -172,11 +172,24 @@ export const VIDEO_MODELS = [
172
172
  'veo-3.1-standard',
173
173
  'seedance-2',
174
174
  // Seedance 2.5 is a SECOND SEAT, not a replacement: 30s takes, 30 image
175
- // references, audio-only references — and 480p/720p ONLY. 2.0 keeps the
176
- // ladder to native 4K and stays the default. Its EDIT row is not here; edit
177
- // models live on slates_edit_video, same as the Kling and Omni Flash ones.
175
+ // references, audio-only references, up to 1080p (2026-08-24) — but no 4K,
176
+ // and dearer than 2.0 at every shared tier, so 2.0 stays the default. Its
177
+ // EDIT row is not here; edit models live on slates_edit_video, same as the
178
+ // Kling and Omni Flash ones.
178
179
  'seedance-2.5',
179
180
  'omni-flash',
181
+ // MiniMax H3, two seats in one family (2026-08-27). Base H3 is the AUTHORED-
182
+ // AUDIO seat — three directable sound layers in one pass, declared reference
183
+ // relationships, 480p to 4K, and the cheapest 768-class second we sell.
184
+ // H3 Max is fal's self-hosted post-train: faster, capped at 768p, takes NO
185
+ // references, and costs MORE than base H3 at the tier they share — a
186
+ // premium-speed seat, never a cheap H3.
187
+ //
188
+ // NEVER PREFIX-MATCH: 'minimax-h3-max' starts with 'minimax-h3'. Every
189
+ // branch keyed on these ids matches EXACTLY; a prefix test silently bills
190
+ // the Max row at base rates and offers it 2K/4K it cannot render.
191
+ 'minimax-h3',
192
+ 'minimax-h3-max',
180
193
  ];
181
194
  // ── Capability-derived param vocabulary + guard ─────────────────
182
195
  //
@@ -193,6 +206,51 @@ const VIDEO_RESOLUTIONS = videoResolutionUnion(VIDEO_MODELS);
193
206
  const VIDEO_DURATION_BOUNDS = durationBounds(VIDEO_MODELS);
194
207
  /** Omni Flash's combined reference-image cap, read from the SSOT (7). */
195
208
  const omniFlashRefCap = getModelCapability('omni-flash')?.maxIngredientImages ?? 0;
209
+ /** Every resolution token any video model speaks, for the forgiving cost-key
210
+ * parser. Derived, so 768p and 2k arrived with the models rather than with a
211
+ * later bug report. */
212
+ const VIDEO_RESOLUTION_VOCAB = VIDEO_RESOLUTIONS;
213
+ // ── MiniMax H3: the reference-image surcharge, as a KEY DIMENSION ────────────
214
+ //
215
+ // fal charges $0.080 per reference image PAST THE FIRST FIVE on
216
+ // `minimax/h3/reference-to-video`, and nowhere else — not on image-to-video
217
+ // start/end frames (those are FL2VA inputs, not Ref2VA references), and not on
218
+ // h3-max, which has no reference endpoint at all. Left unmodelled it inverts
219
+ // the margin: a 10s 768p clip earns $0.30 and four extra images cost $0.32.
220
+ //
221
+ // `/proxy/generate` resolves ONE key to ONE integer, so a surcharge has to live
222
+ // IN the key — exactly the shape Kling's `-audio` suffix uses. The bare key
223
+ // stays identical to every other model's; the suffix appears only when the
224
+ // option costs money.
225
+ //
226
+ // The rounding is EXACT and worth knowing before anyone changes the rate:
227
+ // $0.080 x 1.5 x 100 = 12 cents, and 12 is divisible by CENTS_PER_CREDIT (3),
228
+ // so every extra image is 4 credits at every resolution and every duration with
229
+ // zero drift. A rate that is not a multiple of 2 cents breaks that property.
230
+ const MINIMAX_MODELS = new Set(['minimax-h3', 'minimax-h3-max']);
231
+ /** Reference images fal does not charge for. */
232
+ const MINIMAX_FREE_REF_IMAGES = 5;
233
+ /** K for the `-ref{K}` suffix: images past the free five, capped by the model's
234
+ * own declared ceiling. 0 for h3-max (no reference transport) and for anything
235
+ * that is not a MiniMax row. Mirrors refImageSurchargeCount() in
236
+ * slate/src/shared/pricing.ts. */
237
+ /** Reference IMAGES a generate_video call carries — the four free-reference
238
+ * arrays, combined, exactly as the desktop composer counts them. */
239
+ function minimaxRefImageCount(input) {
240
+ return ((input.ingredientAssetIds?.length ?? 0) +
241
+ (input.characterAssetIds?.length ?? 0) +
242
+ (input.environmentAssetIds?.length ?? 0) +
243
+ (input.styleAssetIds?.length ?? 0));
244
+ }
245
+ function minimaxRefSurchargeCount(model, referenceImages) {
246
+ if (!MINIMAX_MODELS.has(model))
247
+ return 0;
248
+ const cap = getModelCapability(model)?.maxIngredientImages ?? 0;
249
+ const n = Math.floor(referenceImages ?? 0);
250
+ if (!Number.isFinite(n) || n <= 0)
251
+ return 0;
252
+ return Math.max(0, Math.min(n, cap) - MINIMAX_FREE_REF_IMAGES);
253
+ }
196
254
  /** The exact `model` ids `slates_edit_video` accepts. Edit rows are deliberately
197
255
  * NOT in VIDEO_MODELS — they take a source clip, not frames. */
198
256
  export const EDIT_VIDEO_MODELS = [
@@ -236,6 +294,7 @@ export const estimateGenerationCost = {
236
294
  sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
237
295
  seedanceFace: z.boolean().optional().describe('Seedance AI-face route (pricier key).'),
238
296
  seedanceRealFace: z.boolean().optional().describe('Seedance consented real-face route (premium key).'),
297
+ referenceImages: z.number().int().min(0).optional().describe(`minimax-h3 only — how many reference IMAGES the generation will carry. The first ${MINIMAX_FREE_REF_IMAGES} are free and each one after that is a paid dimension of the cost key, so a quote that omits this UNDER-REPORTS a reference-heavy job. Ignored by every other model — including minimax-h3-max, which has no reference endpoint (its start/end frames are free and are not reference images).`),
239
298
  }),
240
299
  async run(input, ctx) {
241
300
  const registry = await ctx.cloud().get('/api/agent/models');
@@ -324,13 +383,15 @@ export const estimateGenerationCost = {
324
383
  videoResolution: input.videoResolution ??
325
384
  resolved.videoResolution ??
326
385
  // Seedance quotes are resolution-scaled, so a missing resolution has
327
- // to fall back to the model's OWN default — 2.5 has no 1080p at all,
328
- // and a blanket '1080p' here quoted a key that does not exist. Read
329
- // from the SSOT; this used to be a hand-typed model-id ternary.
386
+ // to fall back to the model's OWN default — the ladders differ per
387
+ // row (2.5 has no 4K) and a blanket literal quotes a key that does
388
+ // not exist. Read from the SSOT; this used to be a hand-typed
389
+ // model-id ternary.
330
390
  defaultVideoResolutionFor(resolved.model),
331
391
  sound: input.sound ?? resolved.sound,
332
392
  seedanceFace: input.seedanceFace ?? resolved.seedanceFace,
333
393
  seedanceRealFace: input.seedanceRealFace,
394
+ referenceImages: input.referenceImages ?? resolved.referenceImages,
334
395
  });
335
396
  }
336
397
  }
@@ -1500,6 +1561,13 @@ const KLING_TIER_MAP = {
1500
1561
  // Exported for scripts/pricing-consistency-check.mjs (slates-api repo), which
1501
1562
  // asserts this builder byte-matches the desktop's klingCreditKey/seedanceCreditKey.
1502
1563
  export function videoCostKey(input) {
1564
+ // EXACT-ID MAP, NEVER A PREFIX: 'minimax-h3-max' starts with 'minimax-h3'.
1565
+ // Mirrors minimaxCreditKey() in slate/src/shared/pricing.ts.
1566
+ if (MINIMAX_MODELS.has(input.model)) {
1567
+ const res = input.videoResolution ?? defaultVideoResolutionFor(input.model);
1568
+ const k = minimaxRefSurchargeCount(input.model, input.referenceImages);
1569
+ return `${input.model}-${res}-${input.duration}s${k > 0 ? `-ref${k}` : ''}`;
1570
+ }
1503
1571
  if (input.model.startsWith('seedance')) {
1504
1572
  // Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
1505
1573
  // × vref × res × duration). AI-face route bills the `-face-` key (~45% over
@@ -1507,10 +1575,11 @@ export function videoCostKey(input) {
1507
1575
  // (fal partner endpoint). A reference video flips to `-vref-{res}-{T}s`,
1508
1576
  // T = in + out.
1509
1577
  //
1510
- // ⚠️ EVERY BOUND HERE IS VERSION-SCOPED. 2.5 is 480p/720p only, runs to 30s,
1511
- // and takes references to 30s combined — so its vref total reaches 60, DOUBLE
1512
- // 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would quote a real key at a
1513
- // fraction of the real bill.
1578
+ // ⚠️ EVERY BOUND HERE IS VERSION-SCOPED. 2.5 runs 480p/720p/1080p (no 4K),
1579
+ // reaches 30s, and takes references to 30s combined — so its vref total
1580
+ // reaches 60, DOUBLE 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would
1581
+ // quote a real key at a fraction of the real bill. The ladder itself lives in
1582
+ // MODEL_CAPABILITIES; only the per-version DEFAULT is stated below.
1514
1583
  const v25 = input.model.startsWith('seedance-2.5');
1515
1584
  const res = input.videoResolution ?? (v25 ? '720p' : '1080p');
1516
1585
  const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
@@ -1670,10 +1739,22 @@ function resolveVideoModel(raw) {
1670
1739
  out.duration = parseInt(dur[1], 10);
1671
1740
  s = s.replace(/-(\d+)s\b/, '');
1672
1741
  }
1673
- const res = /-(480p|720p|1080p|4k)\b/.exec(s);
1742
+ // MiniMax H3's paid-reference suffix, stripped like every other key dimension
1743
+ // so a pasted `minimax-h3-768p-10s-ref2` resolves instead of erroring. The
1744
+ // number it carries is K (images PAST the free five), so it is converted back
1745
+ // to a TOTAL before anything can re-surcharge it.
1746
+ const ref = /-ref(\d+)\b/.exec(s);
1747
+ if (ref) {
1748
+ out.referenceImages = MINIMAX_FREE_REF_IMAGES + parseInt(ref[1], 10);
1749
+ s = s.replace(/-ref(\d+)\b/, '');
1750
+ }
1751
+ // The RESOLUTION vocabulary is GENERATED from MODEL_CAPABILITIES — the
1752
+ // hand-typed list that stood here went stale the day 768p and 2k shipped.
1753
+ const resRe = new RegExp(`-(${VIDEO_RESOLUTION_VOCAB.join('|')})\\b`);
1754
+ const res = resRe.exec(s);
1674
1755
  if (res) {
1675
1756
  out.videoResolution = res[1];
1676
- s = s.replace(/-(480p|720p|1080p|4k)\b/, '');
1757
+ s = s.replace(resRe, '');
1677
1758
  }
1678
1759
  if (/-audio\b/.test(s)) {
1679
1760
  out.sound = true;
@@ -1712,6 +1793,19 @@ function resolveVideoModel(raw) {
1712
1793
  'gemini-omni-flash': 'omni-flash',
1713
1794
  'gemini-omni-flash-preview': 'omni-flash',
1714
1795
  'omni-flash-preview': 'omni-flash',
1796
+ // The MAX spellings must come out as MAX. Bare `minimax`, `h3` and
1797
+ // `hailuo-3` all mean the BASE row — it holds the full ladder and the
1798
+ // references, and it is cheaper at the tier they share.
1799
+ 'minimax-h3-max': 'minimax-h3-max',
1800
+ 'minimax-h3max': 'minimax-h3-max',
1801
+ 'h3-max': 'minimax-h3-max',
1802
+ 'hailuo-3-max': 'minimax-h3-max',
1803
+ minimax: 'minimax-h3',
1804
+ 'minimax-h3': 'minimax-h3',
1805
+ h3: 'minimax-h3',
1806
+ 'hailuo-3': 'minimax-h3',
1807
+ 'hailuo-3.0': 'minimax-h3',
1808
+ hailuo: 'minimax-h3',
1715
1809
  };
1716
1810
  if (aliases[s]) {
1717
1811
  out.model = aliases[s];
@@ -1737,19 +1831,24 @@ function promptingSkillFor(model) {
1737
1831
  return 'slates-prompting-seedance';
1738
1832
  if (model.startsWith('omni-flash'))
1739
1833
  return 'slates-prompting-omni-flash';
1834
+ // ONE skill covers both H3 seats — the prompt grammar is identical and only
1835
+ // the ladder and the reference transport differ — so a prefix is right here.
1836
+ if (model.startsWith('minimax-h3'))
1837
+ return 'slates-prompting-minimax-h3';
1740
1838
  return 'slates-cost-discipline';
1741
1839
  }
1742
1840
  export const generateVideo = {
1743
1841
  id: 'slates_generate_video',
1744
- description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
1842
+ description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3 / slates-prompting-minimax-h3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
1745
1843
  input: z.object({
1746
1844
  prompt: z.string().min(1).max(4000),
1747
1845
  // ROUTING doctrine only. Every capability number was stripped on 2026-08-16
1748
1846
  // and now lives in the aspectRatio / duration / videoResolution descriptions
1749
1847
  // below, generated from MODEL_CAPABILITIES. "Veo = 16:9 only" and
1750
1848
  // "seedance-2.5 480p/720p" were both stated here AND there, and the two
1751
- // copies disagreed.
1752
- model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (longer takes, far more references, audio-only refs — but no 1080p or 4K, so stay on seedance-2 whenever resolution matters), Veo = native-synced-audio niche only, never the default, omni-flash = cheap tier with audio included (t2v, single-start-frame i2v, or reference images; no last frame, no video/audio refs). All are VIDEO-only. Each model's legal aspect ratios, durations and resolutions are in those params' own descriptions — read them there, not from memory. For per-call cost, call slates_estimate_generation_cost.`),
1849
+ // copies disagreed — and the second of those went stale on 2026-08-24 when
1850
+ // 2.5 gained 1080p, which is exactly the failure mode generating it fixes.
1851
+ model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (longer takes, far more references, audio-only refs — but no 4K, and dearer than seedance-2 at every shared resolution, so stay on seedance-2 unless length or reference count is the point), Veo = native-synced-audio niche only, never the default, omni-flash = cheap tier with audio included (t2v, single-start-frame i2v, or reference images; no last frame, no video/audio refs), minimax-h3 = the AUTHORED-AUDIO seat (dialogue, scene sound and score directed as three separate layers in one pass, plus declared reference relationships; reference images past the fifth are a PAID key dimension — pass referenceImages when quoting), minimax-h3-max = the same model post-trained by fal for SPEED, capped at 768p, and DEARER than minimax-h3 at the tier they share — a deliberate pick, never a default and never the cheap H3; it still takes firstFrameAssetId/lastFrameAssetId, but has no reference endpoint, so the reference set is minimax-h3 only. All are VIDEO-only. Each model's legal aspect ratios, durations and resolutions are in those params' own descriptions — read them there, not from memory. For per-call cost, call slates_estimate_generation_cost.`),
1753
1852
  projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
1754
1853
  // 🚨 THESE THREE DESCRIPTIONS ARE GENERATED FROM `MODEL_CAPABILITIES`.
1755
1854
  // Never hand-write a ratio, resolution or duration into them again — every
@@ -1874,6 +1973,78 @@ export const generateVideo = {
1874
1973
  });
1875
1974
  }
1876
1975
  }
1976
+ // MiniMax H3's SHAPE constraints. Counts and caps are read from the
1977
+ // capability SSOT; only the endpoint SHAPE is stated here, because it is
1978
+ // not a number the registry models: fal publishes text-to-video,
1979
+ // image-to-video and reference-to-video for `minimax/h3`, and only the
1980
+ // first two for `minimax/h3-max` (its reference-to-video 404s). The
1981
+ // reference endpoint has no frame parameters at all, so frames and
1982
+ // references are mutually exclusive — a shape mismatch, not a preference.
1983
+ if (MINIMAX_MODELS.has(input.model)) {
1984
+ const cap = getModelCapability(input.model);
1985
+ const refImages = minimaxRefImageCount(input);
1986
+ const refMedia = (input.videoReferenceAssetIds?.length ?? 0) +
1987
+ (input.audioReferenceAssetIds?.length ?? 0) +
1988
+ (input.videoReferenceAssetId ? 1 : 0) +
1989
+ (input.audioReferenceAssetId ? 1 : 0);
1990
+ const maxImages = cap?.maxIngredientImages ?? 0;
1991
+ const maxVideos = cap?.maxReferenceVideos ?? 0;
1992
+ const maxAudio = cap?.maxReferenceAudio ?? 0;
1993
+ const maxTotal = cap?.maxReferenceFilesTotal ?? 0;
1994
+ if (maxImages === 0 && (refImages > 0 || refMedia > 0)) {
1995
+ return ok({
1996
+ requires_clarification: true,
1997
+ missing: [],
1998
+ message: `${input.model} takes a prompt and up to two frames (start and/or end) — it has no reference endpoint at all, so reference images, video and audio cannot be sent. Drop them, or switch to minimax-h3, which reads ${getModelCapability('minimax-h3')?.maxIngredientImages ?? 9} images plus reference video and audio.`,
1999
+ });
2000
+ }
2001
+ if ((refImages > 0 || refMedia > 0) && (input.firstFrameAssetId || input.lastFrameAssetId)) {
2002
+ return ok({
2003
+ requires_clarification: true,
2004
+ missing: [],
2005
+ message: `${input.model} can use first/last frames OR references, not both — they are different endpoints and the reference one has no frame slots. Drop one side.`,
2006
+ });
2007
+ }
2008
+ if (refImages > maxImages) {
2009
+ return ok({
2010
+ requires_clarification: true,
2011
+ missing: [],
2012
+ message: `${input.model} takes at most ${maxImages} reference images combined (you passed ${refImages}). Trim the list — and note the first ${MINIMAX_FREE_REF_IMAGES} are free while each one after that adds a paid dimension to the cost key.`,
2013
+ });
2014
+ }
2015
+ if ((input.videoReferenceAssetIds?.length ?? 0) + (input.videoReferenceAssetId ? 1 : 0) > maxVideos) {
2016
+ return ok({
2017
+ requires_clarification: true,
2018
+ missing: [],
2019
+ message: `${input.model} takes at most ${maxVideos} reference videos.`,
2020
+ });
2021
+ }
2022
+ if ((input.audioReferenceAssetIds?.length ?? 0) + (input.audioReferenceAssetId ? 1 : 0) > maxAudio) {
2023
+ return ok({
2024
+ requires_clarification: true,
2025
+ missing: [],
2026
+ message: `${input.model} takes at most ${maxAudio} reference audio clips.`,
2027
+ });
2028
+ }
2029
+ if (maxTotal > 0 && refImages + refMedia > maxTotal) {
2030
+ return ok({
2031
+ requires_clarification: true,
2032
+ missing: [],
2033
+ message: `${input.model} takes at most ${maxTotal} reference files across all modalities (you passed ${refImages + refMedia}).`,
2034
+ });
2035
+ }
2036
+ // fal: "Audio cannot be the only reference input; provide at least one
2037
+ // reference image or video with it."
2038
+ const audioRefs = (input.audioReferenceAssetIds?.length ?? 0) + (input.audioReferenceAssetId ? 1 : 0);
2039
+ const videoRefs = (input.videoReferenceAssetIds?.length ?? 0) + (input.videoReferenceAssetId ? 1 : 0);
2040
+ if (audioRefs > 0 && refImages === 0 && videoRefs === 0) {
2041
+ return ok({
2042
+ requires_clarification: true,
2043
+ missing: [],
2044
+ message: `${input.model} rejects an audio-only reference set — add at least one reference image or video alongside it.`,
2045
+ });
2046
+ }
2047
+ }
1877
2048
  // ⛔ The hand-written Veo duration block that stood here is GONE. It read
1878
2049
  // `![4,6,8].includes(duration)` plus "4K only at 8s" — and the registry
1879
2050
  // forces 8s at 1080p TOO, so it happily quoted a 4s 1080p Veo the provider
@@ -1985,6 +2156,10 @@ export const generateVideo = {
1985
2156
  sound: input.sound,
1986
2157
  seedanceFace: input.seedanceFace,
1987
2158
  seedanceRealFace: input.seedanceRealFace,
2159
+ // MiniMax H3 base: reference images past the fifth are a PAID key
2160
+ // dimension. Counted from the same four arrays the request sends, so the
2161
+ // quote and the server's own re-derivation see the same number.
2162
+ referenceImages: minimaxRefImageCount(input),
1988
2163
  // Σ ceil(d - 0.05) over every reference clip, both shapes — the same
1989
2164
  // expression the desktop's estimateCost and the handler's key builder
1990
2165
  // use. Quoting only the singular would understate a multi-clip call.
@@ -2005,7 +2180,9 @@ export const generateVideo = {
2005
2180
  const entry = registry.models.find((m) => m.model === costKey);
2006
2181
  if (!entry) {
2007
2182
  throw new Error(`Model variant not in registry: ${costKey}. ` +
2008
- `Available video models: ${registry.models.filter((m) => m.model.startsWith('kling') || m.model.startsWith('veo') || m.model.startsWith('seedance') || m.model.startsWith('omni-flash')).map((m) => m.model).slice(0, 20).join(', ')}`);
2183
+ // Prefixes DERIVED from the model list, so a new family cannot be
2184
+ // missing from the hint the way minimax-h3 was on day one.
2185
+ `Available video models: ${registry.models.filter((m) => VIDEO_MODELS.some((v) => m.model.startsWith(v.split('.')[0]))).map((m) => m.model).slice(0, 20).join(', ')}`);
2009
2186
  }
2010
2187
  const totalCents = creditCost(entry);
2011
2188
  // Pre-flight confirm gate. Fires when:
@@ -2510,7 +2687,7 @@ export const generateMotionTransfer = {
2510
2687
  // ── Edit video (Kling O3 video-to-video) ────────────────────────
2511
2688
  export const editVideo = {
2512
2689
  id: 'slates_edit_video',
2513
- description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat), or seedance-2.5-edit (4–30s clips — the ONLY engine that takes a clip over 15s; 480p/720p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds — always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
2690
+ description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat), or seedance-2.5-edit (4–30s clips — the ONLY engine that takes a clip over 15s; up to 1080p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds — always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
2514
2691
  input: z.object({
2515
2692
  projectId: z.string().uuid().describe('Project the source clip lives in.'),
2516
2693
  sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
@@ -2579,6 +2756,9 @@ export const editVideo = {
2579
2756
  const costKey = isSeedanceEdit
2580
2757
  ? seedanceEditCostKey({
2581
2758
  duration: billedSeconds,
2759
+ // Cast to the full union, never to the row's current two-or-three
2760
+ // values: MODEL_CAPABILITIES is the narrowing authority and that list
2761
+ // moves (1080p landed on the 2.5 edit row 2026-08-24).
2582
2762
  videoResolution: (input.videoResolution ?? defaultVideoResolutionFor('seedance-2.5-edit')),
2583
2763
  seedanceFace: input.seedanceFace === true,
2584
2764
  })
@@ -3312,6 +3492,15 @@ function resolveGuideTopic(topic) {
3312
3492
  return 'slates-prompting-veo-3';
3313
3493
  if (t.startsWith('omni-flash') || t.startsWith('gemini-omni') || t === 'omni flash')
3314
3494
  return 'slates-prompting-omni-flash';
3495
+ // MiniMax H3 — both seats share one skill. Placed BEFORE the seed/seedance
3496
+ // block for the same reason seed-audio is: no prefix collision exists today,
3497
+ // but a `minimax-*` id must never fall through to a Seedance guide.
3498
+ if (t.startsWith('minimax') ||
3499
+ t.startsWith('hailuo') ||
3500
+ t === 'h3' ||
3501
+ t.startsWith('h3-')) {
3502
+ return 'slates-prompting-minimax-h3';
3503
+ }
3315
3504
  if (t.startsWith('kling-mc'))
3316
3505
  return 'slates-prompting-motion-transfer';
3317
3506
  if (t === 'edit-video' || t === 'video-edit' || t === 'edit video' || t === 'video edit')
@@ -1,6 +1,13 @@
1
1
  /** Every aspect ratio any Slates model accepts. There is no `9:21`. */
2
2
  export type AspectRatio = '1:1' | '2:3' | '3:2' | '3:4' | '4:3' | '4:5' | '5:4' | '9:16' | '16:9' | '21:9';
3
- export type VideoResolution = '480p' | '720p' | '1080p' | '4k';
3
+ /**
4
+ * 🚨 `768p` and `2k` entered this vocabulary with MiniMax H3 (2026-08-27) and
5
+ * are NOT aliases of anything already here. 768p is H3's native generation tier
6
+ * and prices between 480p and 2K ($0.060/s vs 720p Seedance's $0.15/s — a
7
+ * different tier of a different model, not a rename); 2K is H3's upscaled tier.
8
+ * Aliasing either onto 720p/1080p would build a cost key that does not exist.
9
+ */
10
+ export type VideoResolution = '480p' | '720p' | '768p' | '1080p' | '2k' | '4k';
4
11
  /**
5
12
  * The full ten, in display order. `9:21` was in the MCP op's enum and in NO
6
13
  * model — it was invented downstream. Do not add a ratio here that no model
@@ -73,6 +73,21 @@ const VEO_FAL_ASPECT_RATIOS = ['16:9', '9:16'];
73
73
  const OMNI_FLASH_ASPECT_RATIOS = ['16:9', '9:16'];
74
74
  /** Seedance (both seats, and the edit row): six — notably NO `4:5`. */
75
75
  const SEEDANCE_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
76
+ /**
77
+ * MiniMax H3, both seats: six. Read off fal's live OpenAPI 2026-08-27 for
78
+ * `minimax/h3/text-to-video` and `minimax/h3-max/text-to-video` — identical
79
+ * enums. It happens to be the same six Seedance takes; kept as its OWN constant
80
+ * because a provider that adds a ratio adds it to ITS family, and sharing the
81
+ * Seedance constant would silently move H3 the next time ByteDance moves.
82
+ *
83
+ * Two endpoint quirks the registry deliberately does not model:
84
+ * · `image-to-video` has NO `aspect_ratio` param at all — the output follows
85
+ * the start frame. The handler simply omits it there.
86
+ * · `reference-to-video` adds an `adaptive` value on top of these six. We
87
+ * never send it: the composer always has an explicit ratio, and `adaptive`
88
+ * is not an AspectRatio in this vocabulary.
89
+ */
90
+ const MINIMAX_H3_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
76
91
  /**
77
92
  * The provider every AGENT generation actually lands on for Kling and Veo.
78
93
  *
@@ -254,8 +269,16 @@ export const MODEL_CAPABILITIES = {
254
269
  },
255
270
  'seedance-2.5': {
256
271
  aspectRatios: SEEDANCE_ASPECT_RATIOS,
257
- // 🚨 480p/720p ONLY, on BytePlus, EvoLink AND fal. No 1080p, no 4K.
258
- videoResolution: { options: ['480p', '720p'], default: '720p' },
272
+ // 1080p landed 2026-08-24 on ALL THREE rails — BytePlus and EvoLink publish
273
+ // 1080p rate rows and fal's live OpenAPI enum reads
274
+ // ['480p','720p','1080p']. There is still NO 4K on 2.5 (2.0 is the only
275
+ // Seedance with one), which is what keeps `is4kVideoKey` version-blind.
276
+ //
277
+ // DEFAULT STAYS 720p, deliberately: a 30s take at 1080p is ~614 credits
278
+ // against a 1,000-credit welcome grant, and that is at the promotional
279
+ // 1080p rate — it rises when the promo lapses. Reaching a tier and
280
+ // defaulting to it are different decisions.
281
+ videoResolution: { options: ['480p', '720p', '1080p'], default: '720p' },
259
282
  duration: { min: 4, max: 30, mode: 'continuous' },
260
283
  maxIngredientImages: 30,
261
284
  maxReferenceVideos: 10,
@@ -267,7 +290,10 @@ export const MODEL_CAPABILITIES = {
267
290
  },
268
291
  'seedance-2.5-edit': {
269
292
  aspectRatios: SEEDANCE_ASPECT_RATIOS,
270
- videoResolution: { options: ['480p', '720p'], default: '720p' },
293
+ // Same ladder as the generation row (1080p added 2026-08-24). EvoLink's
294
+ // rate card carries 1080p on the edit/extend row and BytePlus's video-input
295
+ // column runs the full tier list; an edit bills that tier × 2.
296
+ videoResolution: { options: ['480p', '720p', '1080p'], default: '720p' },
271
297
  duration: { min: 4, max: 30, mode: 'continuous' },
272
298
  // 🚨 ZERO, AND IT MUST MATCH WHAT THE HANDLER SENDS. The model's edit task
273
299
  // type does accept reference images, but slate's
@@ -285,6 +311,66 @@ export const MODEL_CAPABILITIES = {
285
311
  // NO multimodal reference caps, deliberately: on an edit row the clip IS the
286
312
  // canvas and arrives through `sourceVideo`, not as a reference.
287
313
  },
314
+ // ── MiniMax H3 (both seats on fal — added 2026-08-27) ──────────────────────
315
+ //
316
+ // Every value below is READ OFF fal's live OpenAPI, fetched 2026-08-27:
317
+ // minimax/h3/{text-to-video,image-to-video,reference-to-video}
318
+ // minimax/h3-max/{text-to-video,image-to-video}
319
+ // `minimax/h3-max/reference-to-video` returns 404 — it does not exist, which
320
+ // is why the Max row declares no reference capacity at all.
321
+ //
322
+ // 🚨 NEVER PREFIX-MATCH THESE TWO IDS. `minimax-h3-max` starts with
323
+ // `minimax-h3`, so any `startsWith('minimax-h3')` swallows the Max row into
324
+ // the base row's branch — a different ladder AND a different price at the one
325
+ // tier they share. Every lookup downstream is an exact-id map, not a prefix.
326
+ 'minimax-h3': {
327
+ aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
328
+ // The full ladder. 480p/768p are NATIVE generation modes; 2K and 4K upscale
329
+ // a 768p base result through H3-Regenerate-2K, which is API-only and not in
330
+ // the open weights — that is why fal can undercut list at the bottom two
331
+ // tiers and matches it exactly at the top two.
332
+ //
333
+ // DEFAULT 768p, NOT fal's own default of 2K. 768p is the tier the model was
334
+ // trained to output and the one every benchmark quotes; 2K is a 2.2x price
335
+ // step and 4K a 2.7x step, and reaching a tier is a different decision from
336
+ // defaulting to it (same reasoning that keeps Seedance 2.5 on 720p).
337
+ videoResolution: { options: ['480p', '768p', '2k', '4k'], default: '768p' },
338
+ // 5, not 4. MiniMax's own model card says 4-15s; fal's schema — which is
339
+ // what our request actually hits — says `minimum: 5`. The endpoint wins.
340
+ duration: { min: 5, max: 15, mode: 'continuous' },
341
+ // Ref2VA omni-reference caps, verbatim from the reference-to-video schema:
342
+ // reference_image_urls maxItems 9, reference_video_urls maxItems 3,
343
+ // reference_audio_urls maxItems 3, and in every one of the three
344
+ // descriptions: "Reference images, videos, and audio clips must add up to
345
+ // at most 12 files."
346
+ maxIngredientImages: 9,
347
+ maxReferenceVideos: 3,
348
+ maxReferenceAudio: 3,
349
+ maxReferenceFilesTotal: 12,
350
+ // COMBINED, not per clip. fal states "2-15 seconds each, combined duration
351
+ // at most 15 seconds" for both media arms — so the per-clip floor of 2s is
352
+ // the shared reference-video minimum already enforced by the composer, and
353
+ // 15 is the sum these fields have always meant.
354
+ maxReferenceVideoSeconds: 15,
355
+ maxReferenceAudioSeconds: 15,
356
+ },
357
+ 'minimax-h3-max': {
358
+ aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
359
+ // 480p/768p ONLY — fal's post-train of the open weights, and the 2K
360
+ // upscaler was never open-sourced. Declaring the shorter ladder here IS the
361
+ // whole Max-seat mechanism: `assertVideoCapabilities` refuses 2K/4K on this
362
+ // id, the desktop picker renders only what this entry declares, and the
363
+ // agent's Zod enum stays the union while the per-model guard narrows.
364
+ // Anything shaped like "disable the higher tiers when Max is selected" is
365
+ // re-implementing a guard that already exists.
366
+ videoResolution: { options: ['480p', '768p'], default: '768p' },
367
+ duration: { min: 5, max: 15, mode: 'continuous' },
368
+ // NO reference caps, deliberately: fal publishes text-to-video and
369
+ // image-to-video for h3-max and NOTHING else (reference-to-video 404s), so
370
+ // there is no transport for a reference of any modality. A cap declared
371
+ // above what the handler sends is a SILENT DROP — the exact failure
372
+ // `seedance-2.5-edit` shipped with. Absent means the composer refuses.
373
+ },
288
374
  // ── Audio ──────────────────────────────────────────────────────────────────
289
375
  //
290
376
  // `aspectRatios: []` is deliberate, not an oversight: audio has no frame, and
@@ -374,7 +460,9 @@ export function aspectRatioUnion(models, provider) {
374
460
  }
375
461
  /** Union of every resolution the given models accept. */
376
462
  export function videoResolutionUnion(models) {
377
- const order = ['480p', '720p', '1080p', '4k'];
463
+ // Ascending by output height, so an enum reads as a ladder. 768p sits between
464
+ // 720p and 1080p; 2k (≈2560×1440) between 1080p and 4k.
465
+ const order = ['480p', '720p', '768p', '1080p', '2k', '4k'];
378
466
  const seen = new Set();
379
467
  for (const m of models)
380
468
  for (const r of videoResolutionsFor(m))