@slatesvideo/shared 0.6.9 → 0.6.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,4 @@
1
+ import { MINIMAX_MAX_REFERENCE, minimaxMaxReferenceTokens, MODEL_CAPABILITIES, GPT_QUALITY_TIERS, GPT_BACKGROUNDS } from '../prompts/model-capabilities.js';
1
2
  // Operations layer — the ONE place every Slates agent tool is defined.
2
3
  // Both the MCP server and the CLI register these as their tool / command
3
4
  // surface. Every operation:
@@ -352,11 +353,11 @@ export const VIDEO_MODELS = [
352
353
  'veo-3.1-fast',
353
354
  'veo-3.1-standard',
354
355
  'seedance-2',
355
- // Seedance 2.5 is a SECOND SEAT, not a replacement: 30s takes, 30 image
356
- // references, audio-only references, up to 1080p (2026-08-24) — but no 4K,
357
- // and dearer than 2.0 at every shared tier, so 2.0 stays the default. Its
358
- // EDIT row is not here; edit models live on slates_edit_video, same as the
359
- // Kling and Omni Flash ones.
356
+ // Seedance 2.5 is the DEFAULT video model (Eric, 2026-09-13): 30s takes, 30
357
+ // image references, audio-only references, up to 1080p (2026-08-24). 2.0 stays
358
+ // beside it as the 4K seat, cheaper at every shared tier. 2.5's EDIT row is
359
+ // not here; edit models live on slates_edit_video, same as the Kling and Omni
360
+ // Flash ones.
360
361
  'seedance-2.5',
361
362
  'omni-flash',
362
363
  // MiniMax H3, two seats in one family (2026-08-27). Base H3 is the AUTHORED-
@@ -495,7 +496,7 @@ function editClipBounds(model) {
495
496
  const EDIT_VIDEO_RESOLUTIONS = videoResolutionUnion(['seedance-2.5-edit']);
496
497
  export const estimateGenerationCost = {
497
498
  id: 'slates_estimate_generation_cost',
498
- description: 'Pre-flight cost estimate. Call before any generate_* op so the user sees "this will cost N credits" up front. Takes the SAME base model ids as the generate ops (video: "seedance-2" + duration + videoResolution; image: "nano-banana-2" + resolution) — exact registry cost keys also work. Pairs with the confirm gate.',
499
+ description: 'Quote credits before any generate_* op. Accepts the same base model ids and parameters as generation, or an exact registry cost key. Pairs with the confirm gate.',
499
500
  input: z.object({
500
501
  model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
501
502
  quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
@@ -512,11 +513,13 @@ export const estimateGenerationCost = {
512
513
  characters: z.number().int().min(1).max(TTS_MAX_CHARACTERS).optional().describe(`${TTS_MODEL} only — the LENGTH OF THE TEXT to speak (${TTS_BUCKET_CHARS}-char buckets).`),
513
514
  videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe('Video only. Omitted, each model quotes at its own default. Per-model ladders: see slates_generate_video\'s videoResolution.'),
514
515
  resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k: GPT Image/seedream-5-lite).'),
515
- quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
516
+ quality: z.enum(GPT_QUALITY_TIERS).optional().describe('GPT Image tier; default high.'),
516
517
  aspectRatio: z.string().optional().describe('Image only. 1:1/4:3/3:4 cost more than 16:9.'),
517
518
  sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
518
519
  seedanceFace: z.boolean().optional().describe('Seedance AI-face route (pricier key).'),
519
520
  seedanceRealFace: z.boolean().optional().describe('Seedance consented real-face route (premium key).'),
521
+ videoRefSeconds: z.number().nonnegative().optional().describe('Combined reference-video seconds, measured from the clips.'),
522
+ audioRefSeconds: z.number().nonnegative().optional().describe('Combined reference-audio seconds, including attached character voices.'),
520
523
  referenceImages: z.number().int().min(0).optional().describe(`MiniMax H3 rows only — how many reference IMAGES the generation will carry. Each image past the row's free allowance is a paid dimension of the cost key, so a quote that omits this UNDER-REPORTS a reference-heavy job. The allowances differ: minimax-h3 gives ${MINIMAX_FREE_REF_IMAGES_BY_MODEL['minimax-h3']} free, minimax-h3-max gives ${MINIMAX_FREE_REF_IMAGES_BY_MODEL['minimax-h3-max']}. Ignored by every other model. Start/end frames are free on both rows and are not reference images.`),
521
524
  }),
522
525
  async run(input, ctx) {
@@ -651,9 +654,11 @@ export const estimateGenerationCost = {
651
654
  seedanceFace: input.seedanceFace ?? resolved.seedanceFace,
652
655
  seedanceRealFace: input.seedanceRealFace,
653
656
  referenceImages: input.referenceImages ?? resolved.referenceImages,
657
+ videoRefSeconds: input.videoRefSeconds, audioRefSeconds: input.audioRefSeconds,
654
658
  });
655
659
  }
656
660
  }
661
+ await loadDynamicPrice(ctx, byKey, key);
657
662
  const perCredits = key != null ? byKey.get(key) : undefined;
658
663
  if (key == null || perCredits == null) {
659
664
  // Every id in the error comes from the SSOT arrays. The image half was
@@ -1337,18 +1342,8 @@ async function previewAssets(ctx, refs) {
1337
1342
  return out;
1338
1343
  }
1339
1344
  /** The exact `model` ids `slates_generate_image` accepts. */
1340
- /**
1341
- * The image models whose ladder includes the 3k (1440p) class — a MIRROR of
1342
- * `imageResolutions` in slate's MODEL_REGISTRY, which this package cannot read.
1343
- * Exported so `pricing-consistency-check.mjs` can prove the mirror still
1344
- * matches; without that proof a model that gains 3k in the registry just goes
1345
- * quietly unreachable through the op.
1346
- */
1347
- export const THREE_K_IMAGE_MODELS = [
1348
- 'gpt-image-2-5-flare',
1349
- 'gpt-image-2-5-sunburst',
1350
- 'seedream-5-lite',
1351
- ];
1345
+ /** Derived compatibility export for callers enumerating the 3k ladder. */
1346
+ export const THREE_K_IMAGE_MODELS = Object.keys(MODEL_CAPABILITIES).filter((id) => MODEL_CAPABILITIES[id].imageResolutions?.includes('3k'));
1352
1347
  export const IMAGE_MODELS = [
1353
1348
  'nano-banana-2',
1354
1349
  'nano-banana-2-lite',
@@ -1443,8 +1438,8 @@ export const generateImage = {
1443
1438
  model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
1444
1439
  projectId: z.string().uuid().optional().describe('Save into this Slates project. Renderer refreshes live. Required for every model except nano-banana-2.'),
1445
1440
  resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('1k drafts, 2k hero, 4k final. nano-banana-2-lite: 1k only. GPT Image classes 1024²/1080p/1440p/2160p. Never default this.'),
1446
- quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image only. UNEVEN ladder: max=4× high, xhigh~1.8×. medium drafts; default high.'),
1447
- backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
1441
+ quality: z.enum(GPT_QUALITY_TIERS).optional().describe('GPT Image only. UNEVEN ladder: max=4× high, xhigh~1.8×. medium drafts; default high.'),
1442
+ backgroundMode: z.enum(GPT_BACKGROUNDS).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
1448
1443
  aspectRatio: zEnum(IMAGE_ASPECT_RATIOS).optional().describe(`Pick from the use case: cinematic 16:9 · TikTok/Reels 9:16 · IG square 1:1 · ultra-wide 21:9. 1:1 costs most on GPT Image. Per model: ${describeAspectRatios(IMAGE_MODELS)}`),
1449
1444
  count: z.number().int().min(1).max(10).optional().describe('Up to 10 with projectId; headless caps at 4.'),
1450
1445
  referenceImageUrls: z.array(z.string().url()).max(14).optional().describe('Headless (no projectId) nano-banana-2 only. With a projectId, upload via slates_upload_reference_image. Label every image role in the prompt.'),
@@ -1481,25 +1476,10 @@ export const generateImage = {
1481
1476
  }
1482
1477
  const resolution = input.resolution;
1483
1478
  const imageModel = input.model ?? 'nano-banana-2';
1484
- // 🚨 SEEDREAM HAS A 3k CLASS TOO, AND THIS GUARD USED TO DENY IT.
1485
- // It read `!== 'gpt-image-2'` and rejected every other model at 3k — but
1486
- // `seedream-5-lite` declares ['2k','3k','4k'] in the desktop registry and
1487
- // has a real `seedream-5-lite` cost key, so the op was refusing a request
1488
- // the desktop would have served. Pre-existing; found by the 2026-09-09
1489
- // audit, not introduced by the 2.5 swap.
1490
- //
1491
- // The ladder itself is owned by MODEL_REGISTRY in slate/src/shared/pricing.ts
1492
- // and is not readable from here, so THREE_K_IMAGE_MODELS is a MIRROR — declared
1493
- // and exported below so `pricing-consistency-check.mjs` compares it against
1494
- // the desktop registry's own `imageResolutions`. It used to be an inline
1495
- // literal with a comment admitting nothing checked it, which is how it came
1496
- // to deny `seedream-5-lite` a class the desktop had always served.
1497
- if (resolution === '3k' && !THREE_K_IMAGE_MODELS.includes(imageModel)) {
1498
- return ok({
1499
- requires_clarification: true,
1500
- missing: ['resolution'],
1501
- message: `3k (1440p) exists on ${THREE_K_IMAGE_MODELS.join(', ')} — pick 1k/2k/4k for ${imageModel}.`,
1502
- });
1479
+ const resolutions = MODEL_CAPABILITIES[imageModel]?.imageResolutions ?? [];
1480
+ if (resolution && !resolutions.includes(resolution)) {
1481
+ return ok({ requires_clarification: true, missing: ['resolution'],
1482
+ message: `${imageModel} accepts ${resolutions.join(', ')}.` });
1503
1483
  }
1504
1484
  // Only nano-banana-2 has a headless path — everything else routes through
1505
1485
  // the desktop generation pipeline, which needs a project.
@@ -1571,7 +1551,8 @@ export const generateImage = {
1571
1551
  const costKey = imageCostKey(imageModel, resolution, input.quality, input.aspectRatio ?? '1:1');
1572
1552
  const cloud = ctx.cloud();
1573
1553
  const registry = await cloud.get('/api/agent/models');
1574
- const entry = registry.models.find((m) => m.model === costKey);
1554
+ const entry = registry.models.find((m) => m.model === costKey) ??
1555
+ (await cloud.get(`/api/agent/models?costKey=${encodeURIComponent(costKey)}`)).models[0];
1575
1556
  if (!entry)
1576
1557
  throw new Error(`Model not in registry: ${costKey}`);
1577
1558
  const totalCents = creditCost(entry) * (input.count ?? 1);
@@ -1730,6 +1711,17 @@ export const generateImage = {
1730
1711
  params: {
1731
1712
  prompt: input.prompt,
1732
1713
  aspect_ratio: input.aspectRatio ?? '1:1',
1714
+ // 🚨 THE RESOLUTION HAS TO BE ON THE WIRE, BECAUSE IT IS IN THE KEY.
1715
+ // `imageCostKey` above bills `nano-banana-2-{resolution}`, and until
1716
+ // 2026-09-10 this body omitted the field entirely — so fal rendered at
1717
+ // its own default while we charged for whatever rung the caller asked
1718
+ // for. The enum is fal's (`1K`/`2K`/`4K`), the same one the desktop's
1719
+ // `buildFalNB2Request` sends, and the proxy now recovers the rung from it
1720
+ // rather than trusting the key (`lib/fal-image-keys.ts`).
1721
+ // `3k` cannot reach here — the capability guard above refuses a rung
1722
+ // nano-banana-2 does not declare — so the map is deliberately partial
1723
+ // rather than carrying a rung this model has no key for.
1724
+ resolution: { '1k': '1K', '2k': '2K', '4k': '4K' }[resolution] ?? '2K',
1733
1725
  // 🚨 THE HEADLESS PATH IS THE ONE PLACE WE ASK FAL FOR A BATCH, so it
1734
1726
  // is the one place a provider's own `num_images` ceiling binds — and
1735
1727
  // Nano Banana's is 4 (fal schema, read 2026-09-09), against the op's
@@ -1737,6 +1729,9 @@ export const generateImage = {
1737
1729
  // separate single-image generations, where no batch ceiling exists.
1738
1730
  // Guarded above rather than clamped here: silently making 4 when 10
1739
1731
  // were asked for would bill 4 and look like a partial failure.
1732
+ //
1733
+ // The proxy bills this COUNT (it multiplies the single-image key by it),
1734
+ // which is what the confirm gate above has always quoted.
1740
1735
  num_images: input.count ?? 1,
1741
1736
  ...(hasReferenceImages
1742
1737
  ? { image_urls: input.referenceImageUrls }
@@ -1815,8 +1810,8 @@ export const editImage = {
1815
1810
  editModel: z.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2-5-flare', 'gpt-image-2-5-sunburst', 'flux-2-max', 'seedream-5-lite']).optional(),
1816
1811
  referenceAssetIds: z.array(z.string().uuid()).max(13).optional().describe('Nano-Banana only (NB Pro 13, NB2 Lite 3).'),
1817
1812
  resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('3k = GPT Image/seedream-5-lite; nano-banana-2-lite is 1k only.'),
1818
- quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
1819
- backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
1813
+ quality: z.enum(GPT_QUALITY_TIERS).optional().describe('GPT Image tier; default high.'),
1814
+ backgroundMode: z.enum(GPT_BACKGROUNDS).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
1820
1815
  aspectRatio: z.string().optional(),
1821
1816
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
1822
1817
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
@@ -1847,7 +1842,8 @@ export const editImage = {
1847
1842
  : imageCostKey(editModel, resolution, input.quality, input.aspectRatio);
1848
1843
  const cloud = ctx.cloud();
1849
1844
  const registry = await cloud.get('/api/agent/models');
1850
- const entry = registry.models.find((m) => m.model === costKey);
1845
+ const entry = registry.models.find((m) => m.model === costKey) ??
1846
+ (await cloud.get(`/api/agent/models?costKey=${encodeURIComponent(costKey)}`)).models[0];
1851
1847
  if (!entry)
1852
1848
  throw new Error(`Model not in registry: ${costKey}`);
1853
1849
  const totalCents = creditCost(entry);
@@ -1991,6 +1987,11 @@ export function videoCostKey(input) {
1991
1987
  // Mirrors minimaxCreditKey() in slate/src/shared/pricing.ts.
1992
1988
  if (MINIMAX_MODELS.has(input.model)) {
1993
1989
  const res = input.videoResolution ?? defaultVideoResolutionFor(input.model);
1990
+ if (input.model === 'minimax-h3-max') {
1991
+ const tokens = minimaxMaxReferenceTokens({ imagePixels: (input.referenceImages ?? 0) * MINIMAX_MAX_REFERENCE.normalizedImageEdge ** 2,
1992
+ videoSeconds: input.videoRefSeconds ?? 0, audioSeconds: input.audioRefSeconds ?? 0, resolution: res ?? '' });
1993
+ return `${input.model}-${res}-${input.duration}s${tokens ? `-rt${tokens}` : ''}`;
1994
+ }
1994
1995
  const k = minimaxRefSurchargeCount(input.model, input.referenceImages);
1995
1996
  return `${input.model}-${res}-${input.duration}s${k > 0 ? `-ref${k}` : ''}`;
1996
1997
  }
@@ -2214,9 +2215,10 @@ function resolveVideoModel(raw) {
2214
2215
  'kling-v3.0-omni-pro': 'kling-v3.0-omni',
2215
2216
  'seedance-2.0': 'seedance-2',
2216
2217
  'seedance-2-0': 'seedance-2',
2217
- // ⚠️ The 2.5 spellings must resolve to 2.5, and the BARE `seedance` must keep
2218
- // resolving to 2.0 — 2.0 is the default video model and holds 1080p/4K, which
2219
- // 2.5 does not have at all.
2218
+ // ⚠️ The 2.5 spellings must resolve to 2.5, and the BARE `seedance` keeps
2219
+ // resolving to 2.0: every published CLI and MCP build that sends it expects
2220
+ // the 4K seat, and 2.5 rejects 4K. The default video model is 2.5 (tier in
2221
+ // MODEL_FACTS, 2026-09-13); an agent names it as `seedance-2.5`.
2220
2222
  'seedance-2.5': 'seedance-2.5',
2221
2223
  'seedance-25': 'seedance-2.5',
2222
2224
  'seedance-2-5': 'seedance-2.5',
@@ -2355,9 +2357,9 @@ export const generateVideo = {
2355
2357
  // The capacity sentences are DERIVED from MODEL_FACTS (see
2356
2358
  // multimodalRefSummary) rather than hand-typed, so a cap change in one
2357
2359
  // place cannot leave a stale number in a description an LLM reads.
2358
- videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS, cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to the vref key (input+output seconds) — pass videoReferenceSecondsEach. Over the cap is REFUSED, never trimmed.`),
2360
+ videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS, cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; per-model caps are on audioReferenceAssetIds. Billing switches to the vref key (input+output seconds) — pass videoReferenceSecondsEach. Over the cap is REFUSED, never trimmed.`),
2359
2361
  videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each clip\'s duration in seconds. Feeds the vref cost key; the server re-probes and corrects an understated value upward.'),
2360
- audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO, cited as "audio 1", "audio 2"… in the order given. No billing surcharge. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
2362
+ audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO, cited as "audio 1", "audio 2"… in the order given. ${multimodalRefModels().join(' / ')} only. No billing surcharge. ${multimodalRefModels().map(multimodalRefSummary).join(' ')}`),
2361
2363
  audioReferenceSpokenText: z.array(z.string()).optional().describe('The exact words in each reference clip — same order and length as audioReferenceAssetIds, "" for a clip with no speech. The model RE-TRANSCRIBES a take rather than using it verbatim, so audio decides voice/accent/timing and only this decides the WORDS. Omit it and the words are a guess.'),
2362
2364
  sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
2363
2365
  audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
@@ -2681,25 +2683,27 @@ export const generateVideo = {
2681
2683
  }
2682
2684
  const cloud = ctx.cloud();
2683
2685
  const registry = await cloud.get('/api/agent/models');
2684
- const costKey = videoCostKey({
2685
- model: input.model,
2686
- duration: input.duration,
2687
- videoResolution: input.videoResolution,
2688
- sound: input.sound,
2689
- seedanceFace: input.seedanceFace,
2690
- seedanceRealFace: input.seedanceRealFace,
2691
- // MiniMax H3 base: reference images past the fifth are a PAID key
2692
- // dimension. Counted from the same four arrays the request sends, so the
2693
- // quote and the server's own re-derivation see the same number.
2694
- referenceImages: minimaxRefImageCount(input),
2695
- // Σ ceil(d - 0.05) over every reference clip, both shapes — the same
2696
- // expression the desktop's estimateCost and the handler's key builder
2697
- // use. Quoting only the singular would understate a multi-clip call.
2698
- videoRefSeconds: (input.videoReferenceAssetId && input.videoReferenceSeconds
2699
- ? Math.ceil(input.videoReferenceSeconds - 0.05)
2700
- : 0) +
2701
- (input.videoReferenceSecondsEach ?? []).reduce((n, d) => n + (d > 0 ? Math.ceil(d - 0.05) : 0), 0),
2702
- });
2686
+ const costKey = input.model === 'minimax-h3-max'
2687
+ ? (await ctx.desktop().get('/agent/generation/video-quote', { request: JSON.stringify(input) })).costKey
2688
+ : videoCostKey({
2689
+ model: input.model,
2690
+ duration: input.duration,
2691
+ videoResolution: input.videoResolution,
2692
+ sound: input.sound,
2693
+ seedanceFace: input.seedanceFace,
2694
+ seedanceRealFace: input.seedanceRealFace,
2695
+ // MiniMax H3 base: reference images past the fifth are a PAID key
2696
+ // dimension. Counted from the same four arrays the request sends, so the
2697
+ // quote and the server's own re-derivation see the same number.
2698
+ referenceImages: minimaxRefImageCount(input),
2699
+ // Σ ceil(d - 0.05) over every reference clip, both shapes — the same
2700
+ // expression the desktop's estimateCost and the handler's key builder
2701
+ // use. Quoting only the singular would understate a multi-clip call.
2702
+ videoRefSeconds: (input.videoReferenceAssetId && input.videoReferenceSeconds
2703
+ ? Math.ceil(input.videoReferenceSeconds - 0.05)
2704
+ : 0) +
2705
+ (input.videoReferenceSecondsEach ?? []).reduce((n, d) => n + (d > 0 ? Math.ceil(d - 0.05) : 0), 0),
2706
+ });
2703
2707
  // Hard consent gate, checked before any spend: the real-face route is
2704
2708
  // consent-attested by design (the desktop enforces it too).
2705
2709
  if (input.seedanceRealFace && !input.realFaceConsent) {
@@ -2709,7 +2713,8 @@ export const generateVideo = {
2709
2713
  message: 'Real-person generation needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
2710
2714
  });
2711
2715
  }
2712
- const entry = registry.models.find((m) => m.model === costKey);
2716
+ const entry = registry.models.find((m) => m.model === costKey) ??
2717
+ (await cloud.get(`/api/agent/models?costKey=${encodeURIComponent(costKey)}`)).models[0];
2713
2718
  if (!entry) {
2714
2719
  throw new Error(`Model variant not in registry: ${costKey}. ` +
2715
2720
  // Prefixes DERIVED from the model list, so a new family cannot be
@@ -3067,7 +3072,8 @@ export const generateAudio = {
3067
3072
  durationSeconds: seconds,
3068
3073
  characters: input.model === TTS_MODEL ? input.prompt.length : undefined,
3069
3074
  });
3070
- const entry = registry.models.find((m) => m.model === costKey);
3075
+ const entry = registry.models.find((m) => m.model === costKey) ??
3076
+ (await cloud.get(`/api/agent/models?costKey=${encodeURIComponent(costKey)}`)).models[0];
3071
3077
  if (!entry) {
3072
3078
  throw new Error(`Audio variant not in registry: ${costKey}. Available audio models: ${AUDIO_MODELS.join(' | ')}.`);
3073
3079
  }
@@ -3179,7 +3185,8 @@ export const generateLipSync = {
3179
3185
  }
3180
3186
  const cloud = ctx.cloud();
3181
3187
  const registry = await cloud.get('/api/agent/models');
3182
- const entry = registry.models.find((m) => m.model === costKey);
3188
+ const entry = registry.models.find((m) => m.model === costKey) ??
3189
+ (await cloud.get(`/api/agent/models?costKey=${encodeURIComponent(costKey)}`)).models[0];
3183
3190
  if (!entry)
3184
3191
  throw new Error(`Model variant not in registry: ${costKey}`);
3185
3192
  const totalCents = creditCost(entry);
@@ -3275,7 +3282,8 @@ export const generateMotionTransfer = {
3275
3282
  const costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
3276
3283
  const cloud = ctx.cloud();
3277
3284
  const registry = await cloud.get('/api/agent/models');
3278
- const entry = registry.models.find((m) => m.model === costKey);
3285
+ const entry = registry.models.find((m) => m.model === costKey) ??
3286
+ (await cloud.get(`/api/agent/models?costKey=${encodeURIComponent(costKey)}`)).models[0];
3279
3287
  if (!entry)
3280
3288
  throw new Error(`Model variant not in registry: ${costKey}`);
3281
3289
  const totalCents = creditCost(entry);
@@ -3454,7 +3462,8 @@ export const editVideo = {
3454
3462
  : klingEditCostKey(model, billedSeconds);
3455
3463
  const cloud = ctx.cloud();
3456
3464
  const registry = await cloud.get('/api/agent/models');
3457
- const entry = registry.models.find((m) => m.model === costKey);
3465
+ const entry = registry.models.find((m) => m.model === costKey) ??
3466
+ (await cloud.get(`/api/agent/models?costKey=${encodeURIComponent(costKey)}`)).models[0];
3458
3467
  if (!entry)
3459
3468
  throw new Error(`Model variant not in registry: ${costKey}`);
3460
3469
  const totalCents = creditCost(entry);
@@ -4229,8 +4238,9 @@ function shotParamsShape(described) {
4229
4238
  duration: d(z.number().int().min(1).max(360).optional(), 'Seconds for video or duration-based audio; TTS uses text length.'),
4230
4239
  videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the model when the Shot is saved.'),
4231
4240
  imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
4232
- gptQuality: d(z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional(), 'GPT Image 2.5 only.'),
4233
- gptBackground: d(z.enum(['auto', 'transparent', 'opaque']).optional(), 'GPT Image only.'),
4241
+ gptQuality: d(z.enum(GPT_QUALITY_TIERS).optional(), 'GPT Image 2.5 only.'),
4242
+ gptBackground: d(z.enum(GPT_BACKGROUNDS).optional(), 'GPT Image only.'),
4243
+ detachedVoiceCharacterIds: z.array(z.string()).optional().describe('Character voices removed from this recipe.'),
4234
4244
  imageQuantity: d(z.number().int().min(1).max(10).optional(), 'Image models only.'),
4235
4245
  negativePrompt: z.string().optional(),
4236
4246
  sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
@@ -4467,7 +4477,7 @@ function shotCostKey(detail) {
4467
4477
  const duration = fires?.duration ?? p.duration;
4468
4478
  if (!duration)
4469
4479
  return null;
4470
- const billed = (d) => (d > 0 ? Math.ceil(d - 0.05) : 0);
4480
+ const billed = (d) => model === 'minimax-h3-max' ? Math.max(0, d) : (d > 0 ? Math.ceil(d - 0.05) : 0);
4471
4481
  return videoCostKey({
4472
4482
  model: model,
4473
4483
  duration,
@@ -4480,6 +4490,7 @@ function shotCostKey(detail) {
4480
4490
  // of these read 0 there. That is why a listing quote is announced as a
4481
4491
  // floor and `slates_get_shot` is the exact one.
4482
4492
  referenceImages: (detail.references ?? []).filter((r) => r.kind === 'image').length,
4493
+ audioRefSeconds: (detail.references ?? []).filter((r) => r.kind === 'audio').reduce((n, r) => n + (r.durationSeconds ?? 0), 0),
4483
4494
  videoRefSeconds: (detail.references ?? [])
4484
4495
  .filter((r) => r.kind === 'video')
4485
4496
  .reduce((n, r) => n + billed(r.durationSeconds ?? 0), 0),
@@ -4499,6 +4510,14 @@ function shotCostKey(detail) {
4499
4510
  }
4500
4511
  return null;
4501
4512
  }
4513
+ /** Dynamic reference keys are priced by the same server calculator that debits them. */
4514
+ async function loadDynamicPrice(ctx, byKey, key) {
4515
+ if (!key || byKey.has(key))
4516
+ return;
4517
+ const response = await ctx.cloud().get(`/api/agent/models?costKey=${encodeURIComponent(key)}`);
4518
+ for (const row of response.models)
4519
+ byKey.set(row.model, creditCost(row));
4520
+ }
4502
4521
  /** Credits for one Shot, and how many generations it fires.
4503
4522
  *
4504
4523
  * `imageQuantity` multiplies IMAGE models only — the same condition the
@@ -4760,6 +4779,8 @@ export const listShots = {
4760
4779
  // auditing.
4761
4780
  const registry = await ctx.cloud().get('/api/agent/models');
4762
4781
  const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
4782
+ for (const row of rows)
4783
+ await loadDynamicPrice(ctx, byKey, shotCostKey(row));
4763
4784
  let total = 0;
4764
4785
  let unpriced = 0;
4765
4786
  const shots = rows.map((s) => {
@@ -4811,6 +4832,7 @@ export const getShot = {
4811
4832
  const r = await desktop.get('/agent/shots/get', { id: input.shotId });
4812
4833
  const registry = await ctx.cloud().get('/api/agent/models');
4813
4834
  const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
4835
+ await loadDynamicPrice(ctx, byKey, shotCostKey(r.shot));
4814
4836
  const q = shotQuote(r.shot, byKey);
4815
4837
  return ok({ ...r.shot, cost_key: q.key, credits: q.credits }, `"${r.shot.name || 'Untitled'}" — ${r.shot.model ?? 'no model set'}, ` +
4816
4838
  (q.key && !r.shot.blocked
@@ -4894,6 +4916,7 @@ export const generateFromShots = {
4894
4916
  }
4895
4917
  const registry = await ctx.cloud().get('/api/agent/models');
4896
4918
  const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
4919
+ await Promise.all(details.map((d) => loadDynamicPrice(ctx, byKey, shotCostKey(d))));
4897
4920
  const quotes = details.map((d) => ({ detail: d, ...shotQuote(d, byKey) }));
4898
4921
  const total = quotes.reduce((n, q) => n + q.credits, 0);
4899
4922
  const largest = quotes.reduce((m, q) => (q.credits > m ? q.credits : m), 0);
@@ -84,7 +84,70 @@ export interface VoiceCloneCapability {
84
84
  max: number;
85
85
  };
86
86
  }
87
+ export declare const GPT_QUALITY_TIERS: readonly ["low", "medium", "high", "xhigh", "max"];
88
+ export type GptQuality = (typeof GPT_QUALITY_TIERS)[number];
89
+ export declare const GPT_BACKGROUNDS: readonly ["auto", "transparent", "opaque"];
90
+ export type GptBackground = (typeof GPT_BACKGROUNDS)[number];
91
+ export type ImageResolution = '1k' | '2k' | '3k' | '4k';
92
+ export declare const GPT_IMAGE_25_SIZES: Record<string, Record<string, {
93
+ width: number;
94
+ height: number;
95
+ }>>;
96
+ /**
97
+ * fal's named ~1MP presets per aspect ratio, with custom dims where fal has no
98
+ * preset. The `1k` rung of every non-GPT image model resolves through this.
99
+ */
100
+ export declare const FAL_1MP_SIZES: Record<string, string | {
101
+ width: number;
102
+ height: number;
103
+ }>;
104
+ /** Pixel dims for a megapixel target at an aspect ratio, rounded to multiples of 8. */
105
+ export declare function computeFalDimensions(aspectRatio: string, targetMP: number): {
106
+ width: number;
107
+ height: number;
108
+ };
109
+ /**
110
+ * The `image_size` a non-GPT fal image request carries, for one model × aspect ×
111
+ * resolution rung.
112
+ *
113
+ * 🚨 THIS IS A BILLING INPUT, WHICH IS WHY IT LIVES HERE (moved out of
114
+ * slate/src/main/api/fal.ts, 2026-09-10). The resolution rung is a segment of
115
+ * every image cost key, and nothing in the REQUEST names it — fal is told pixel
116
+ * dimensions, not "2k". The proxy therefore recovers the rung by running this
117
+ * function over the model's declared `imageResolutions` × `aspectRatios` and
118
+ * matching the body's `image_size`, exactly as it recovers a GPT Image rung from
119
+ * `GPT_IMAGE_25_SIZES`. A second copy of this arithmetic would mean the desktop
120
+ * and the server could disagree about what a request is worth, silently.
121
+ *
122
+ * GPT Image does NOT come through here — that family carries explicit pixel
123
+ * classes in `GPT_IMAGE_25_SIZES` and an explicit `quality` rung.
124
+ */
125
+ export declare function falImageSize(model: string, aspectRatio?: string, resolution?: string): string | {
126
+ width: number;
127
+ height: number;
128
+ };
87
129
  export interface ModelCapability {
130
+ imageResolutions?: ImageResolution[];
131
+ /**
132
+ * Images ONE request may ask the provider for in a single batch.
133
+ *
134
+ * 🚨 ABSENT MEANS ONE, AND THAT IS A BILLING BOUND, NOT A HINT. A cost key
135
+ * prices a single image, so a request that returns N images has to be charged
136
+ * N times — the proxy multiplies the debit by the batch size and refuses a
137
+ * batch larger than this (`slates-api/src/lib/fal-image-keys.ts`). Only a model
138
+ * with a READ provider ceiling gets a number here; everything else fans out as
139
+ * separate generations, which is what the desktop does for every model.
140
+ */
141
+ maxBatchImages?: number;
142
+ /** Per-file reference-audio bounds, independently of the combined cap. */
143
+ referenceVideoDuration?: {
144
+ min: number;
145
+ max: number;
146
+ };
147
+ referenceAudioDuration?: {
148
+ min: number;
149
+ max: number;
150
+ };
88
151
  aspectRatios: AspectRatio[];
89
152
  /** Provider-keyed overrides. `fal` is the one that matters — see AGENT_ROUTE_PROVIDER. */
90
153
  providerAspectRatios?: Record<string, AspectRatio[]>;
@@ -185,4 +248,21 @@ export declare function describeVideoResolutions(models: readonly string[]): str
185
248
  export declare function describeDurations(models: readonly string[]): string;
186
249
  /** e.g. "seedance-2: 9 · seedance-2.5: 30 · omni-flash: 7 · seedance-2.5-edit: 0 (prompt + source clip only)" */
187
250
  export declare function describeReferenceImageCaps(models: readonly string[]): string;
251
+ /** H3 Max reference accounting, fal's worked tables read 2026-09-09.
252
+ * https://fal.ai/models/minimax/h3-max/reference-to-video
253
+ * 1080p video-reference pricing is unpublished; never infer it from output rates.
254
+ */
255
+ export declare const MINIMAX_MAX_REFERENCE: {
256
+ readonly freeTokens: 4096;
257
+ readonly imagePixelsPerToken: 1024;
258
+ readonly normalizedImageEdge: 1024;
259
+ readonly audioTokensPerSecond: 80;
260
+ readonly videoTokensPerSecond: Partial<Record<VideoResolution, number>>;
261
+ };
262
+ export declare function minimaxMaxReferenceTokens(input: {
263
+ imagePixels: number;
264
+ videoSeconds: number;
265
+ audioSeconds: number;
266
+ resolution: string;
267
+ }): number;
188
268
  //# sourceMappingURL=model-capabilities.d.ts.map