@slatesvideo/shared 0.5.6 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,9 @@ import { z } from 'zod';
13
13
  import { SlatesCloudClient } from '../clients/cloud.js';
14
14
  import { SlatesDesktopClient } from '../clients/desktop.js';
15
15
  import { SKILLS } from '../skills/content.js';
16
+ // Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
17
+ // "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
18
+ import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } from '../prompts/model-facts.js';
16
19
  export function defaultContext() {
17
20
  return {
18
21
  cloud: () => new SlatesCloudClient(),
@@ -136,8 +139,7 @@ export const estimateGenerationCost = {
136
139
  input: z.object({
137
140
  model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
138
141
  quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
139
- duration: z.number().int().min(1).max(360).optional().describe('Seconds. Video 3-15 (cost scales linearly; required with a video base id). Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22. Ignored by eleven-v3 (per-character) and suno (flat).'),
140
- characters: z.number().int().min(1).max(5000).optional().describe('eleven-v3 only — script length in characters, billed in 100-char buckets rounded up.'),
142
+ duration: z.number().int().min(1).max(360).optional().describe('Seconds. Video 3-15 (cost scales linearly; required with a video base id). Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22 — required with either audio base id.'),
141
143
  videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Video only. Seedance defaults to 1080p.'),
142
144
  resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
143
145
  quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
@@ -156,13 +158,13 @@ export const estimateGenerationCost = {
156
158
  if (img)
157
159
  key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
158
160
  }
159
- // 2a) audio base id → seconds (seed-audio, eleven-sfx), characters
160
- // (eleven-v3), or flat (suno). Runs BEFORE the video resolver: it is
161
+ // 2a) audio base id → seconds. Both surfaces bill per second, so a
162
+ // duration is always required. Runs BEFORE the video resolver: it is
161
163
  // forgiving by design and "seed-audio" would otherwise be mistaken for
162
164
  // a seedance spelling.
163
165
  if (!key && AUDIO_MODELS.includes(input.model)) {
164
166
  const m = input.model;
165
- if ((m === 'seed-audio' || m === 'eleven-sfx') && !input.duration) {
167
+ if (!input.duration) {
166
168
  return ok({
167
169
  requires_clarification: true,
168
170
  missing: ['duration'],
@@ -171,13 +173,6 @@ export const estimateGenerationCost = {
171
173
  : 'Sound Effects bills per second — pass duration in seconds (1-22).',
172
174
  });
173
175
  }
174
- if (m === 'eleven-v3' && !input.characters) {
175
- return ok({
176
- requires_clarification: true,
177
- missing: ['characters'],
178
- message: 'Eleven v3 bills per 100 characters of script, rounded up — pass characters (the length of the text you intend to speak).',
179
- });
180
- }
181
176
  // REFUSE an out-of-range duration rather than quoting the clamped price.
182
177
  // audioCostKey clamps (it has to — it mirrors the desktop, which clamps),
183
178
  // so without this an agent asking for 2s of Seed Audio would be handed a
@@ -187,20 +182,17 @@ export const estimateGenerationCost = {
187
182
  const audioBounds = {
188
183
  'seed-audio': { min: SEED_AUDIO_MIN_SECONDS, max: SEED_AUDIO_MAX_SECONDS },
189
184
  'eleven-sfx': { min: ELEVEN_SFX_MIN_SECONDS, max: ELEVEN_SFX_MAX_SECONDS },
190
- suno: { min: SUNO_MIN_SECONDS, max: SUNO_MAX_SECONDS },
191
185
  };
192
186
  const bounds = audioBounds[m];
193
- if (bounds && input.duration != null && (input.duration < bounds.min || input.duration > bounds.max)) {
187
+ if (input.duration < bounds.min || input.duration > bounds.max) {
194
188
  return ok({
195
189
  requires_clarification: true,
196
190
  missing: ['duration'],
197
191
  message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
198
- (m === 'suno'
199
- ? 'Suno duration is FREE and flat-priced, so any value in range costs the same.'
200
- : 'Re-ask with a duration in range.'),
192
+ 'Re-ask with a duration in range.',
201
193
  });
202
194
  }
203
- key = audioCostKey({ model: m, durationSeconds: input.duration, characters: input.characters });
195
+ key = audioCostKey({ model: m, durationSeconds: input.duration });
204
196
  }
205
197
  // 2b) Kling O3 edit base id + duration (ceiled source-clip length)
206
198
  if (!key && (input.model === 'kling-v3.0-omni-edit' || input.model === 'kling-v3.0-omni-pro-edit')) {
@@ -231,7 +223,12 @@ export const estimateGenerationCost = {
231
223
  duration,
232
224
  videoResolution: input.videoResolution ??
233
225
  resolved.videoResolution ??
234
- (resolved.model.startsWith('seedance') ? '1080p' : undefined),
226
+ // Seedance quotes are resolution-scaled, so a missing resolution has
227
+ // to fall back to the model's OWN default — 2.5 has no 1080p at all,
228
+ // and a blanket '1080p' here quoted a key that does not exist.
229
+ (resolved.model === 'seedance-2.5' ? '720p'
230
+ : resolved.model.startsWith('seedance') ? '1080p'
231
+ : undefined),
235
232
  sound: input.sound ?? resolved.sound,
236
233
  seedanceFace: input.seedanceFace ?? resolved.seedanceFace,
237
234
  seedanceRealFace: input.seedanceRealFace,
@@ -240,7 +237,7 @@ export const estimateGenerationCost = {
240
237
  }
241
238
  const perCredits = key != null ? byKey.get(key) : undefined;
242
239
  if (key == null || perCredits == null) {
243
- throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/characters/resolution params, or use slates_list_available_models with a filter.`);
240
+ throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
244
241
  }
245
242
  const qty = input.quantity ?? 1;
246
243
  const totalCredits = perCredits * qty;
@@ -1323,6 +1320,11 @@ export const VIDEO_MODELS = [
1323
1320
  'veo-3.1-fast',
1324
1321
  'veo-3.1-standard',
1325
1322
  'seedance-2',
1323
+ // Seedance 2.5 is a SECOND SEAT, not a replacement: 30s takes, 30 image
1324
+ // references, audio-only references — and 480p/720p ONLY. 2.0 keeps the
1325
+ // ladder to native 4K and stays the default. Its EDIT row is not here; edit
1326
+ // models live on slates_edit_video, same as the Kling and Omni Flash ones.
1327
+ 'seedance-2.5',
1326
1328
  'omni-flash',
1327
1329
  ];
1328
1330
  // Model → registry cost-key. Each provider's keys ship with their own
@@ -1346,16 +1348,24 @@ const KLING_TIER_MAP = {
1346
1348
  // asserts this builder byte-matches the desktop's klingCreditKey/seedanceCreditKey.
1347
1349
  export function videoCostKey(input) {
1348
1350
  if (input.model.startsWith('seedance')) {
1349
- // Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (face × vref ×
1350
- // res × duration). AI-face route bills the `-face-` key (~45% over faceless);
1351
- // consented real-person route bills the premium `-realface-` key (fal partner
1352
- // endpoint). A reference video flips to `-vref-{res}-{T}s`, T = in + out (6..30).
1353
- const res = input.videoResolution ?? '1080p';
1351
+ // Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
1352
+ // × vref × res × duration). AI-face route bills the `-face-` key (~45% over
1353
+ // faceless); consented real-person route bills the premium `-realface-` key
1354
+ // (fal partner endpoint). A reference video flips to `-vref-{res}-{T}s`,
1355
+ // T = in + out.
1356
+ //
1357
+ // ⚠️ EVERY BOUND HERE IS VERSION-SCOPED. 2.5 is 480p/720p only, runs to 30s,
1358
+ // and takes references to 30s combined — so its vref total reaches 60, DOUBLE
1359
+ // 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would quote a real key at a
1360
+ // fraction of the real bill.
1361
+ const v25 = input.model.startsWith('seedance-2.5');
1362
+ const res = input.videoResolution ?? (v25 ? '720p' : '1080p');
1354
1363
  const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
1355
1364
  const vrefSecs = input.videoRefSeconds ?? 0;
1356
1365
  if (vrefSecs > 0) {
1357
1366
  // ceil(x - 0.05) matches the server's probe rounding — quote = bill.
1358
- const total = Math.min(30, Math.max(6, Math.ceil(vrefSecs - 0.05) + input.duration));
1367
+ const maxTotal = v25 ? SEEDANCE_25_VREF_MAX_TOTAL : SEEDANCE_20_VREF_MAX_TOTAL;
1368
+ const total = Math.min(maxTotal, Math.max(6, Math.ceil(vrefSecs - 0.05) + input.duration));
1359
1369
  return `${input.model}${face}-vref-${res}-${total}s`;
1360
1370
  }
1361
1371
  return `${input.model}${face}-${res}-${input.duration}s`;
@@ -1405,11 +1415,30 @@ export function klingEditCostKey(model, duration) {
1405
1415
  export function omniFlashEditCostKey(duration) {
1406
1416
  return `omni-flash-edit-${duration}s`;
1407
1417
  }
1418
+ /** Max billed (input + output) seconds on a Seedance video-reference gen.
1419
+ * 2.0: refs 2–15s + output 4–15s. 2.5: refs to 30s + output to 30s. */
1420
+ const SEEDANCE_20_VREF_MAX_TOTAL = 30;
1421
+ const SEEDANCE_25_VREF_MAX_TOTAL = 60;
1422
+ /** Seedance 2.5 source-clip bounds for the edit task type. */
1423
+ export const SEEDANCE_25_EDIT_MIN_SECONDS = 4;
1424
+ export const SEEDANCE_25_EDIT_MAX_SECONDS = 30;
1425
+ // Seedance 2.5 video-edit cost key — mirrors seedanceCreditKey() in
1426
+ // slate/src/shared/pricing.ts for the edit row (must byte-match; checked by the
1427
+ // slates-api pricing-consistency script). Unlike the Kling and Omni Flash edit
1428
+ // keys this one carries a RESOLUTION and a FACE ROUTE, because Seedance's edit
1429
+ // price moves with both. Duration is the CEILED source-clip length: the edit
1430
+ // task type forces `duration: -1` on the wire, so the source clip is the only
1431
+ // honest quote.
1432
+ export function seedanceEditCostKey(input) {
1433
+ const res = input.videoResolution ?? '720p';
1434
+ const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
1435
+ return `seedance-2.5-edit${face}-${res}-${input.duration}s`;
1436
+ }
1408
1437
  // ── Audio ───────────────────────────────────────────────────────
1409
1438
  // Exported: the exact `model` ids slates_generate_audio accepts — consumed
1410
1439
  // by the desktop Studio Agent system prompt (SSOT; never restate these ids
1411
1440
  // in prose that can drift). Mirrors VIDEO_MODELS for the third media type.
1412
- export const AUDIO_MODELS = ['seed-audio', 'eleven-v3', 'eleven-sfx', 'suno'];
1441
+ export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx'];
1413
1442
  /**
1414
1443
  * Per-surface bounds and defaults.
1415
1444
  *
@@ -1434,9 +1463,6 @@ export const SEED_AUDIO_DEFAULT_SECONDS = 15;
1434
1463
  export const ELEVEN_SFX_MIN_SECONDS = 1;
1435
1464
  export const ELEVEN_SFX_MAX_SECONDS = 22;
1436
1465
  export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
1437
- export const ELEVEN_V3_MAX_CHARACTERS = 5000;
1438
- export const SUNO_MIN_SECONDS = 10;
1439
- export const SUNO_MAX_SECONDS = 360;
1440
1466
  /**
1441
1467
  * Byte-for-byte the desktop's `clampAudioDuration` in slate/src/shared/pricing.ts,
1442
1468
  * INCLUDING the non-finite arm — that one matters: `Math.max(min, NaN)` is NaN,
@@ -1468,25 +1494,10 @@ export function audioCostKey(input) {
1468
1494
  const secs = clampAudioSeconds(input.durationSeconds ?? SEED_AUDIO_DEFAULT_SECONDS, SEED_AUDIO_MIN_SECONDS, SEED_AUDIO_MAX_SECONDS, SEED_AUDIO_DEFAULT_SECONDS);
1469
1495
  return `seed-audio-${secs}s`;
1470
1496
  }
1471
- if (input.model === 'eleven-v3') {
1472
- // 100-character buckets, always rounded UP — never under-bill a read.
1473
- // Mirrors ttsBuckets() in slate/src/shared/pricing.ts, non-finite arm included.
1474
- const chars = input.characters ?? 0;
1475
- const buckets = Number.isFinite(chars)
1476
- ? Math.min(ELEVEN_V3_MAX_CHARACTERS / 100, Math.max(1, Math.ceil(chars / 100)))
1477
- : 1;
1478
- return `eleven-v3-tts-${buckets}00c`;
1479
- }
1480
1497
  if (input.model === 'eleven-sfx') {
1481
1498
  const secs = clampAudioSeconds(input.durationSeconds ?? ELEVEN_SFX_DEFAULT_SECONDS, ELEVEN_SFX_MIN_SECONDS, ELEVEN_SFX_MAX_SECONDS, ELEVEN_SFX_DEFAULT_SECONDS);
1482
1499
  return `eleven-sfx-${secs}s`;
1483
1500
  }
1484
- if (input.model === 'suno') {
1485
- // Flat across every Suno model AND every duration up to 360s — measured
1486
- // against the live sunoapi.org balance 2026-07-31 (12 credits for a
1487
- // default call and for duration=240 alike). One call returns TWO songs.
1488
- return 'suno-generate';
1489
- }
1490
1501
  throw new Error(`Unknown audio model: ${input.model}`);
1491
1502
  }
1492
1503
  /**
@@ -1533,6 +1544,13 @@ function resolveVideoModel(raw) {
1533
1544
  'kling-v3.0-omni-pro': 'kling-v3.0-omni',
1534
1545
  'seedance-2.0': 'seedance-2',
1535
1546
  'seedance-2-0': 'seedance-2',
1547
+ // ⚠️ The 2.5 spellings must resolve to 2.5, and the BARE `seedance` must keep
1548
+ // resolving to 2.0 — 2.0 is the default video model and holds 1080p/4K, which
1549
+ // 2.5 does not have at all.
1550
+ 'seedance-2.5': 'seedance-2.5',
1551
+ 'seedance-25': 'seedance-2.5',
1552
+ 'seedance-2-5': 'seedance-2.5',
1553
+ 'seedance2.5': 'seedance-2.5',
1536
1554
  seedance: 'seedance-2',
1537
1555
  'veo-3.1': 'veo-3.1-fast',
1538
1556
  'veo-3': 'veo-3.1-fast',
@@ -1555,6 +1573,11 @@ function promptingSkillFor(model) {
1555
1573
  return 'slates-prompting-kling-v3';
1556
1574
  if (model.startsWith('veo'))
1557
1575
  return 'slates-prompting-veo-3';
1576
+ // 2.5 BEFORE the generic seedance test — "seedance-2.5" also starts with
1577
+ // "seedance", and falling through hands 2.0's guide to a model with different
1578
+ // limits, a different resolution ladder and an extra task type.
1579
+ if (model.startsWith('seedance-2.5'))
1580
+ return 'slates-prompting-seedance-2-5';
1558
1581
  if (model.startsWith('seedance'))
1559
1582
  return 'slates-prompting-seedance';
1560
1583
  if (model.startsWith('omni-flash'))
@@ -1563,23 +1586,30 @@ function promptingSkillFor(model) {
1563
1586
  }
1564
1587
  export const generateVideo = {
1565
1588
  id: 'slates_generate_video',
1566
- description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
1589
+ description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
1567
1590
  input: z.object({
1568
1591
  prompt: z.string().min(1).max(4000),
1569
- model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
1592
+ model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | seedance-2.5 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (4-30s takes, 30 image refs, audio-only refs — but 480p/720p ONLY, so stay on seedance-2 whenever resolution matters), Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
1570
1593
  projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
1571
1594
  aspectRatio: z.enum(['1:1', '16:9', '9:16', '4:3', '3:4', '21:9', '9:21', '4:5', '5:4', '2:3', '3:2']).optional().describe('Veo locks to 16:9 — passing anything else will be ignored or fail. Kling/Seedance support all.'),
1572
- duration: z.number().int().min(3).max(15).optional().describe('Seconds. Kling: 5-15. Veo: 4, 6, or 8 only (4K only at 8s). Seedance: 4-15. Omni Flash: 3-10. Default 5 if omitted but always be explicit (cost scales linearly).'),
1573
- videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive). Veo: 720p/1080p same price, 4K more (8s only).'),
1595
+ duration: z.number().int().min(3).max(30).optional().describe('Seconds. Kling: 5-15. Veo: 4, 6, or 8 only (4K only at 8s). Seedance 2: 4-15. Seedance 2.5: 4-30 (the only model that reaches 30). Omni Flash: 3-10. Default 5 if omitted but always be explicit — cost scales linearly, and a 30s seedance-2.5 take is several hundred credits.'),
1596
+ videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance 2: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive, and Pro-only). Seedance 2.5: 480p/720p ONLY (default 720p) — it has no 1080p and no 4K on any provider, so asking for one is rejected, not downgraded. Veo: 720p/1080p same price, 4K more (8s only).'),
1574
1597
  firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video: asset UUID or badge code ("IMG-A8") — codes resolve against the project at call time, so a code the user just spoke is always safe to pass.'),
1575
1598
  lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code). Veo and Seedance only. Pairs with firstFrameAssetId for guided transitions.'),
1576
- ingredientAssetIds: z.array(z.string()).max(9).optional().describe('Visual reference / ingredient assets (UUIDs or badge codes) for Kling Omni, Seedance, or Omni Flash. Up to 9 (Seedance), 4 (Kling), or 7 (Omni Flash, combined across all ref params).'),
1599
+ ingredientAssetIds: z.array(z.string()).max(30).optional().describe('Visual reference / ingredient assets (UUIDs or badge codes) for Kling Omni, Seedance, or Omni Flash. Up to 30 (Seedance 2.5), 9 (Seedance 2), 4 (Kling), or 7 (Omni Flash, combined across all ref params). More is not better: 2-4 strong references beat both extremes, and past 4 reference PEOPLE output stability drops on Seedance regardless of the cap.'),
1577
1600
  characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets (UUIDs or badge codes) — keeps a character consistent across the shot.'),
1578
1601
  environmentAssetIds: z.array(z.string()).optional().describe('Environment reference assets (UUIDs or badge codes) — keeps a location/setting consistent across the shot.'),
1579
1602
  styleAssetIds: z.array(z.string()).optional().describe('Style reference assets (UUIDs or badge codes) — locks the visual style of the shot.'),
1580
- videoReferenceAssetId: z.string().optional().describe('Seedance ONLY: an existing VIDEO asset (UUID or badge code) to use as a reference — edit/relocate a clip, or MOTION TRANSFER (pair with a subject in ingredientAssetIds and a prompt like "the character from image 1 performs the motion from video 1"). 2-15s. Billing switches to input+output seconds (the vref key) — pass videoReferenceSeconds so the quote is right. If the clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Ignored by Kling/Veo.'),
1581
- videoReferenceSeconds: z.number().optional().describe('REQUIRED with videoReferenceAssetId: the reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — a video-reference gen bills combined input+output seconds; the server re-derives this by probing the clip, so an understated value just gets corrected upward.'),
1582
- audioReferenceAssetId: z.string().optional().describe('Seedance ONLY: an AUDIO asset (UUID or badge code), ≤15s, used as a reference — e.g. lip-sync a character to this audio ("the character in image 1 speaks the dialogue from audio 1"). No billing surcharge (Seedance audio is included). Requires at least one image or video reference alongside.'),
1603
+ videoReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into videoReferenceAssetIds; prefer that for anything new. A single VIDEO asset (UUID or badge code) used as a reference. Kept working forever: installed CLI and MCP builds send this shape.'),
1604
+ videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach. Required with videoReferenceAssetId: that clip\'s duration in seconds.'),
1605
+ audioReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into audioReferenceAssetIds; prefer that. A single AUDIO asset (UUID or badge code) used as a reference.'),
1606
+ // ── Multimodal references, the plural surface ──
1607
+ // The capacity sentences are DERIVED from MODEL_FACTS (see
1608
+ // multimodalRefSummary) rather than hand-typed, so a cap change in one
1609
+ // place cannot leave a stale number in a description an LLM reads.
1610
+ videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS (UUIDs or badge codes) read alongside the images and audio in the same generation — own-footage restyle, MOTION TRANSFER ("the character from image 1 performs the motion from video 1"), or dialogue conditioning. Cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; ignored elsewhere. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to combined input+output seconds (the vref key) — pass videoReferenceSecondsEach so the quote is right. If any clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Over the cap is REFUSED, never trimmed: a dropped clip would already have been priced in.`),
1611
+ videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — the bill is Σceil(each) + output seconds. The server re-derives this by probing every uploaded clip, so an understated value just gets corrected upward.'),
1612
+ audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO clips (UUIDs or badge codes) read alongside the images and video — e.g. lip-sync a character to a line ("the character in image 1 speaks the dialogue from audio 1"). Cited as "audio 1", "audio 2"… in the order given. No billing surcharge (Seedance audio is included). ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
1583
1613
  sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
1584
1614
  audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
1585
1615
  generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
@@ -1648,7 +1678,10 @@ export const generateVideo = {
1648
1678
  message: `Omni Flash supports 3-10 seconds (you passed ${input.duration}s). Pick a duration in that range.`,
1649
1679
  });
1650
1680
  }
1651
- if (input.lastFrameAssetId || input.videoReferenceAssetId || input.audioReferenceAssetId) {
1681
+ if (input.lastFrameAssetId ||
1682
+ input.videoReferenceAssetId || input.audioReferenceAssetId ||
1683
+ (input.videoReferenceAssetIds?.length ?? 0) > 0 ||
1684
+ (input.audioReferenceAssetIds?.length ?? 0) > 0) {
1652
1685
  return ok({
1653
1686
  requires_clarification: true,
1654
1687
  missing: [],
@@ -1709,6 +1742,10 @@ export const generateVideo = {
1709
1742
  refInputs.push({ ref: input.videoReferenceAssetId, role: 'video reference' });
1710
1743
  if (input.audioReferenceAssetId)
1711
1744
  refInputs.push({ ref: input.audioReferenceAssetId, role: 'audio reference' });
1745
+ for (const r of input.videoReferenceAssetIds ?? [])
1746
+ refInputs.push({ ref: r, role: 'video reference' });
1747
+ for (const r of input.audioReferenceAssetIds ?? [])
1748
+ refInputs.push({ ref: r, role: 'audio reference' });
1712
1749
  const resolvedRefs = await resolveAssetRefs(ctx, input.projectId, refInputs.map((r) => r.ref));
1713
1750
  const rid = (v) => v ? (resolvedRefs.get(v)?.id ?? v) : v;
1714
1751
  const rids = (a) => a?.map((v) => resolvedRefs.get(v)?.id ?? v);
@@ -1716,6 +1753,8 @@ export const generateVideo = {
1716
1753
  input.lastFrameAssetId = rid(input.lastFrameAssetId);
1717
1754
  input.videoReferenceAssetId = rid(input.videoReferenceAssetId);
1718
1755
  input.audioReferenceAssetId = rid(input.audioReferenceAssetId);
1756
+ input.videoReferenceAssetIds = rids(input.videoReferenceAssetIds);
1757
+ input.audioReferenceAssetIds = rids(input.audioReferenceAssetIds);
1719
1758
  input.ingredientAssetIds = rids(input.ingredientAssetIds);
1720
1759
  input.characterAssetIds = rids(input.characterAssetIds);
1721
1760
  input.environmentAssetIds = rids(input.environmentAssetIds);
@@ -1724,13 +1763,59 @@ export const generateVideo = {
1724
1763
  // A video reference bills combined input+output seconds (the vref key) —
1725
1764
  // the quote needs the clip's length. The server probes the uploaded ref
1726
1765
  // and corrects the key anyway, so this only gates quote accuracy.
1727
- if (input.videoReferenceAssetId && input.model.startsWith('seedance') && !input.videoReferenceSeconds) {
1766
+ // 🚨 Seedance 2.5 PROMPT-INTENT PRE-FLIGHT. With references attached, the
1767
+ // provider sorts a request into one of five task types from the roles PLUS
1768
+ // the prompt's intent, then fails a mismatch ASYNCHRONOUSLY — after the task
1769
+ // queues and credits are reserved. Catch it before the spend, NAME the words,
1770
+ // and never rewrite the prompt (the prompt-transparency invariant).
1771
+ if (input.model === 'seedance-2.5') {
1772
+ // 🚨 EVERY reference shape counts, singular AND plural. The classifier
1773
+ // engages on the presence of reference ROLES, and the plural arrays are
1774
+ // the surface new callers use — listing only the deprecated singular ones
1775
+ // meant the modern call path got no warning at all, which is precisely
1776
+ // backwards.
1777
+ const hasRefs = !!input.firstFrameAssetId || !!input.lastFrameAssetId ||
1778
+ (input.ingredientAssetIds?.length ?? 0) > 0 ||
1779
+ (input.characterAssetIds?.length ?? 0) > 0 ||
1780
+ (input.environmentAssetIds?.length ?? 0) > 0 ||
1781
+ (input.styleAssetIds?.length ?? 0) > 0 ||
1782
+ !!input.videoReferenceAssetId || !!input.audioReferenceAssetId ||
1783
+ (input.videoReferenceAssetIds?.length ?? 0) > 0 ||
1784
+ (input.audioReferenceAssetIds?.length ?? 0) > 0;
1785
+ // Trigger words are DERIVED from the model-facts SSOT, not retyped here —
1786
+ // a local literal had already lost 'continue the story'.
1787
+ const hits = hasRefs ? seedanceTaskIntentWords(input.prompt) : [];
1788
+ if (hits.length > 0 && !input.confirm) {
1789
+ return ok({
1790
+ requires_clarification: true,
1791
+ missing: ['prompt'],
1792
+ message: `Seedance 2.5 reads ${hits.map((w) => `"${w}"`).join(', ')} in this prompt as an EDIT or EXTEND instruction and may run it as a video edit, ` +
1793
+ `which fails after the job has queued. If you mean to edit an existing clip, call slates_edit_video with model "seedance-2.5-edit". ` +
1794
+ `If you mean a fresh shot, describe the finished frame instead of an instruction to change one ("the workshop bench, clear and uncluttered" rather than "remove the tripod"). ` +
1795
+ `Pass confirm=true to send it as written.`,
1796
+ });
1797
+ }
1798
+ }
1799
+ // The quote needs a length for EVERY reference clip, in both shapes. A
1800
+ // missing one doesn't fail the generation (the server probes and corrects
1801
+ // upward) — it silently under-quotes, which is worse than asking.
1802
+ const isSeedance = input.model.startsWith('seedance');
1803
+ if (input.videoReferenceAssetId && isSeedance && !input.videoReferenceSeconds) {
1728
1804
  return ok({
1729
1805
  requires_clarification: true,
1730
1806
  missing: ['videoReferenceSeconds'],
1731
1807
  message: 'A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSeconds (the reference clip\'s duration, shown in slates_list_assets) so the pre-flight quote matches the bill.',
1732
1808
  });
1733
1809
  }
1810
+ const pluralRefCount = input.videoReferenceAssetIds?.length ?? 0;
1811
+ if (pluralRefCount > 0 && isSeedance &&
1812
+ (input.videoReferenceSecondsEach?.length ?? 0) !== pluralRefCount) {
1813
+ return ok({
1814
+ requires_clarification: true,
1815
+ missing: ['videoReferenceSecondsEach'],
1816
+ message: `A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSecondsEach with exactly ${pluralRefCount} duration${pluralRefCount === 1 ? '' : 's'}, in the same order as videoReferenceAssetIds (durations are shown in slates_list_assets), so the pre-flight quote matches the bill.`,
1817
+ });
1818
+ }
1734
1819
  const cloud = ctx.cloud();
1735
1820
  const registry = await cloud.get('/api/agent/models');
1736
1821
  const costKey = videoCostKey({
@@ -1740,7 +1825,13 @@ export const generateVideo = {
1740
1825
  sound: input.sound,
1741
1826
  seedanceFace: input.seedanceFace,
1742
1827
  seedanceRealFace: input.seedanceRealFace,
1743
- videoRefSeconds: input.videoReferenceAssetId ? input.videoReferenceSeconds : 0,
1828
+ // Σ ceil(d - 0.05) over every reference clip, both shapes — the same
1829
+ // expression the desktop's estimateCost and the handler's key builder
1830
+ // use. Quoting only the singular would understate a multi-clip call.
1831
+ videoRefSeconds: (input.videoReferenceAssetId && input.videoReferenceSeconds
1832
+ ? Math.ceil(input.videoReferenceSeconds - 0.05)
1833
+ : 0) +
1834
+ (input.videoReferenceSecondsEach ?? []).reduce((n, d) => n + (d > 0 ? Math.ceil(d - 0.05) : 0), 0),
1744
1835
  });
1745
1836
  // Hard consent gate, checked before any spend: the real-face route is
1746
1837
  // consent-attested by design (the desktop enforces it too).
@@ -1834,8 +1925,13 @@ export const generateVideo = {
1834
1925
  characterAssetIds: input.characterAssetIds ?? [],
1835
1926
  environmentAssetIds: input.environmentAssetIds ?? [],
1836
1927
  styleAssetIds: input.styleAssetIds ?? [],
1928
+ // Both shapes on the wire. The route merges and dedupes them, so an old
1929
+ // client sending only the singular and a new one sending only the plural
1930
+ // reach the identical handler params.
1837
1931
  videoReferenceAssetId: input.videoReferenceAssetId,
1838
1932
  audioReferenceAssetId: input.audioReferenceAssetId,
1933
+ videoReferenceAssetIds: input.videoReferenceAssetIds,
1934
+ audioReferenceAssetIds: input.audioReferenceAssetIds,
1839
1935
  sound: input.sound,
1840
1936
  audioLanguage: input.audioLanguage,
1841
1937
  generateMusic: input.generateMusic,
@@ -1884,30 +1980,28 @@ export const generateVideo = {
1884
1980
  // ── Generate audio ──────────────────────────────────────────────
1885
1981
  export const generateAudio = {
1886
1982
  id: 'slates_generate_audio',
1887
- description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Four surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s), eleven-v3 (verbatim text-to-speech in a named voice), eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop), suno (full music; every call returns TWO songs for one flat price). Which surface for which job: read the slates-model-selection skill. ' +
1983
+ description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Two surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s) and eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop). Which surface for which job: read the slates-model-selection skill. ' +
1888
1984
  '🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
1889
- 'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-suno). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
1985
+ 'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
1890
1986
  'projectId is REQUIRED (no headless path). Cost > 17 credits returns requires_confirm — pass confirm=true after explicit user OK. No skill files installed? Call slates_get_prompting_guide first.',
1891
1987
  input: z.object({
1892
1988
  projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
1893
1989
  model: z
1894
1990
  .enum(AUDIO_MODELS)
1895
- .describe('Audio surface. seed-audio = scene/ambience/beds (default choice), eleven-v3 = exact-script voiceover, eleven-sfx = one precise effect, suno = music. Routing doctrine: slates-model-selection skill.'),
1991
+ .describe('Audio surface. seed-audio = scene/ambience/beds/dialogue (default choice), eleven-sfx = one precise effect or a seamless loop. Routing doctrine: slates-model-selection skill.'),
1896
1992
  prompt: z
1897
1993
  .string()
1898
1994
  .min(1)
1899
1995
  .max(5000)
1900
- .describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-v3: the SCRIPT, spoken verbatim — never put stage directions here. eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars. suno: a description in default mode, or the EXACT LYRICS when customMode=true and instrumental=false.'),
1996
+ .describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars.'),
1901
1997
  durationSeconds: z
1902
1998
  .number()
1903
1999
  .optional()
1904
- .describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic. suno 10-360, FREE (a 6-minute track costs the same as a default one) but only accepted on sunoModel=V5_5 with customMode=true. Ignored by eleven-v3, which bills per 100 characters of text.'),
2000
+ .describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.'),
1905
2001
  voice: z
1906
2002
  .string()
1907
2003
  .optional()
1908
- .describe('seed-audio: a preset voice id (e.g. "cedric_en_zh") — leave unset to let the scene cast itself, which is usually right for background dialogue. eleven-v3: a preset name (Rachel default; Aria, Roger, Sarah, Laura, Charlie, George, Callum, River, Liam, Charlotte, Alice, Matilda, Will, Jessica, Eric, Chris, Brian, Daniel, Lily, Bill). Pick one and keep it for the whole piece. No voice cloning on this route.'),
1909
- stability: z.number().min(0).max(1).optional().describe('eleven-v3 only. 0-1, default 0.5. Lower = more expressive and more variable take-to-take; higher = flatter and repeatable. Raise it for long narration.'),
1910
- languageCode: z.string().optional().describe('eleven-v3 only — ISO 639-1 code to force a language when the text is ambiguous or code-switched.'),
2004
+ .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
1911
2005
  speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
1912
2006
  volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
1913
2007
  pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
@@ -1923,36 +2017,22 @@ export const generateAudio = {
1923
2017
  .string()
1924
2018
  .optional()
1925
2019
  .describe('seed-audio only — ONE image asset to score what is in frame. MUTUALLY EXCLUSIVE with audioReferenceAssetIds.'),
1926
- sunoModel: z.enum(['V4', 'V4_5', 'V4_5PLUS', 'V4_5ALL', 'V5', 'V5_5']).optional().describe('suno only — wire model id (underscored). Default V5. Cost is FLAT across every version. duration needs V5_5.'),
1927
- customMode: z
1928
- .boolean()
1929
- .optional()
1930
- .describe('suno only. false (default) = prompt is a ≤500-char DESCRIPTION and lyrics get written for you. true = style + title required, and prompt becomes the EXACT LYRICS, sung as written. Putting a description in the prompt while customMode=true wastes a full generation.'),
1931
- instrumental: z.boolean().optional().describe('suno only — score with no vocals. Usually right for a film bed: an unasked-for vocal fights dialogue.'),
1932
- style: z.string().optional().describe('suno only — genre + era + instrumentation + tempo ("90s trip-hop, dusty breakbeat, Rhodes, 85 bpm"). Required in customMode. Steer here, not by piling adjectives into the prompt.'),
1933
- title: z.string().optional().describe('suno only — track title. Required in customMode.'),
1934
- negativeTags: z.string().optional().describe('suno only — comma-separated things to keep out ("brass, EDM drop, male vocal").'),
1935
- vocalGender: z.enum(['m', 'f']).optional().describe('suno only — the WIRE values m/f, not "male"/"female".'),
1936
- styleWeight: z.number().min(0).max(1).optional().describe('suno only — 0-1, how hard the track hugs `style`. Higher = more genre-obedient and more generic; lower = more room to surprise you.'),
1937
- weirdnessConstraint: z.number().min(0).max(1).optional().describe('suno only — 0-1 experimentation dial. Low is safe and on-brief; high wanders. Leave unset for a film bed.'),
1938
- background: z.boolean().optional().describe(BACKGROUND_DESCRIBE + ' Recommended for suno (2-3 min renders).'),
2020
+ background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
1939
2021
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate after explicit user OK.'),
1940
2022
  }),
1941
2023
  run: async (input, ctx) => {
1942
2024
  // ── Per-surface clarification + constraint gates ──
1943
2025
  const cfgDefaults = {
1944
- 'seed-audio': 15,
1945
- 'eleven-sfx': 4,
1946
- 'eleven-v3': undefined,
1947
- suno: undefined,
2026
+ 'seed-audio': SEED_AUDIO_DEFAULT_SECONDS,
2027
+ 'eleven-sfx': ELEVEN_SFX_DEFAULT_SECONDS,
1948
2028
  };
1949
2029
  const seconds = input.durationSeconds ?? cfgDefaults[input.model];
1950
2030
  if (input.model === 'seed-audio') {
1951
- if (seconds == null || seconds < 3 || seconds > SEED_AUDIO_MAX_SECONDS) {
2031
+ if (seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
1952
2032
  return ok({
1953
2033
  requires_clarification: true,
1954
2034
  missing: ['durationSeconds'],
1955
- message: `Seed Audio needs a durationSeconds of 3-${SEED_AUDIO_MAX_SECONDS}. It has NO duration parameter — the number is written into the prompt AND is what the user is billed, so it must be a deliberate choice. Ask the user how long the bed should be (a few seconds longer than the clip it sits under, so the edit has handles).`,
2035
+ message: `Seed Audio needs a durationSeconds of ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS}. It has NO duration parameter — the number is written into the prompt AND is what the user is billed, so it must be a deliberate choice. Ask the user how long the bed should be (a few seconds longer than the clip it sits under, so the edit has handles).`,
1956
2036
  });
1957
2037
  }
1958
2038
  if ((input.audioReferenceAssetIds?.length ?? 0) > 0 && input.imageReferenceAssetId) {
@@ -1960,29 +2040,17 @@ export const generateAudio = {
1960
2040
  }
1961
2041
  }
1962
2042
  if (input.model === 'eleven-sfx') {
1963
- if (seconds == null || seconds < 1 || seconds > ELEVEN_SFX_MAX_SECONDS) {
2043
+ if (seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
1964
2044
  return ok({
1965
2045
  requires_clarification: true,
1966
2046
  missing: ['durationSeconds'],
1967
- message: `Sound Effects needs a durationSeconds of 1-${ELEVEN_SFX_MAX_SECONDS}. It is billed per second and is never left for the model to pick (that would make the charge non-deterministic). Roughly: 0.5-1s for an impact, 2-4s for a whoosh, 8-22s for a loopable bed.`,
2047
+ message: `Sound Effects needs a durationSeconds of ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. It is billed per second and is never left for the model to pick (that would make the charge non-deterministic). Roughly: 0.5-1s for an impact, 2-4s for a whoosh, 8-22s for a loopable bed.`,
1968
2048
  });
1969
2049
  }
1970
2050
  if (input.prompt.length > 450) {
1971
2051
  throw new Error(`Sound Effects accepts up to 450 characters — this prompt is ${input.prompt.length}.`);
1972
2052
  }
1973
2053
  }
1974
- if (input.model === 'eleven-v3' && input.prompt.length > ELEVEN_V3_MAX_CHARACTERS) {
1975
- throw new Error(`Eleven v3 accepts up to ${ELEVEN_V3_MAX_CHARACTERS} characters — this script is ${input.prompt.length}.`);
1976
- }
1977
- if (input.model === 'suno' && input.customMode === true) {
1978
- if (!input.style || !input.title) {
1979
- return ok({
1980
- requires_clarification: true,
1981
- missing: [...(input.style ? [] : ['style']), ...(input.title ? [] : ['title'])],
1982
- message: 'Suno custom mode requires style and title. Remember that in custom mode the prompt field is the EXACT LYRICS (unless instrumental=true, where it is ignored) — if you meant to describe a mood, use customMode=false instead.',
1983
- });
1984
- }
1985
- }
1986
2054
  await ctx.desktop().requireCapability('audio-generation', 'audio generation');
1987
2055
  // ── Resolve asset refs at CALL time (UUIDs or badge codes) ──
1988
2056
  const refInputs = [];
@@ -2002,7 +2070,6 @@ export const generateAudio = {
2002
2070
  const costKey = audioCostKey({
2003
2071
  model: input.model,
2004
2072
  durationSeconds: seconds,
2005
- characters: input.prompt.length,
2006
2073
  });
2007
2074
  const entry = registry.models.find((m) => m.model === costKey);
2008
2075
  if (!entry) {
@@ -2020,7 +2087,6 @@ export const generateAudio = {
2020
2087
  (input.model === 'seed-audio'
2021
2088
  ? `You are billed for the ${seconds}s you requested regardless of the returned length. `
2022
2089
  : '') +
2023
- (input.model === 'suno' ? 'This returns TWO songs for that one price. ' : '') +
2024
2090
  'Confirm with the user, then call again with confirm: true.',
2025
2091
  });
2026
2092
  }
@@ -2036,8 +2102,6 @@ export const generateAudio = {
2036
2102
  prompt: input.prompt,
2037
2103
  durationSeconds: seconds,
2038
2104
  voice: input.voice,
2039
- stability: input.stability,
2040
- languageCode: input.languageCode,
2041
2105
  speed: input.speed,
2042
2106
  volume: input.volume,
2043
2107
  pitch: input.pitch,
@@ -2046,15 +2110,6 @@ export const generateAudio = {
2046
2110
  promptInfluence: input.promptInfluence,
2047
2111
  audioReferenceAssetIds: (input.audioReferenceAssetIds ?? []).map((r) => rid(r)),
2048
2112
  imageReferenceAssetId: rid(input.imageReferenceAssetId),
2049
- sunoModel: input.sunoModel,
2050
- customMode: input.customMode,
2051
- instrumental: input.instrumental,
2052
- style: input.style,
2053
- title: input.title,
2054
- negativeTags: input.negativeTags,
2055
- vocalGender: input.vocalGender,
2056
- styleWeight: input.styleWeight,
2057
- weirdnessConstraint: input.weirdnessConstraint,
2058
2113
  background: input.background,
2059
2114
  });
2060
2115
  if (!result.success)
@@ -2062,10 +2117,8 @@ export const generateAudio = {
2062
2117
  if (result.background) {
2063
2118
  return backgroundSubmitted(`${input.model} audio generation`, [result.generationId].filter(Boolean), { model: input.model, variant: costKey, projectId: input.projectId, cost_credits: totalCents }, refEcho);
2064
2119
  }
2065
- const siblings = result.siblingAssets ?? [];
2066
2120
  return {
2067
2121
  text: `Generated ${input.model} audio into project ${input.projectId} for ${fmtCredits(totalCents)}` +
2068
- (siblings.length > 0 ? ` (${siblings.length + 1} tracks — Suno returns two variations)` : '') +
2069
2122
  `. Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
2070
2123
  (refEcho ? ` ${refEcho}` : ''),
2071
2124
  data: {
@@ -2076,7 +2129,6 @@ export const generateAudio = {
2076
2129
  cost_cents: totalCents,
2077
2130
  cost_credits: totalCents,
2078
2131
  asset: result.asset,
2079
- ...(siblings.length > 0 ? { siblingAssets: siblings } : {}),
2080
2132
  generationId: result.generationId,
2081
2133
  },
2082
2134
  };
@@ -2085,27 +2137,19 @@ export const generateAudio = {
2085
2137
  // ── Generate lip-sync ───────────────────────────────────────────
2086
2138
  export const generateLipSync = {
2087
2139
  id: 'slates_generate_lip_sync',
2088
- description: 'Lip-sync a still image (avatar) or a video clip to audio. Two engines — route per slates-model-selection: (1) engine=kling (default, cheap utility lane): sourceType=video re-syncs a clip (~$0.11 / 5s); sourceType=image animates a still avatar (avatar-standard ~$0.42 / 5s; avatar-pro ~$0.86 / 5s). Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. (2) engine=seedance-2 (premium single-pass): the speech is generated IN the video itself — natural delivery, a video source keeps its OWN voice (native voice clone), audio included; ttsText becomes the spoken line (no voice/speed params), or an uploaded ≤15s audio file drives the speech as a reference. Seedance sources must be 2-15s videos or images; bills seedance keys (video sources bill input+output seconds — pass sourceSeconds). Faces route via seedanceFace (default true) / seedanceRealFace+realFaceConsent for real people. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
2140
+ description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip (~$0.11 / 5s); sourceType=image animates a still avatar (avatar-standard ~$0.42 / 5s; avatar-pro ~$0.86 / 5s). Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
2089
2141
  input: z.object({
2090
2142
  projectId: z.string().uuid().describe('Slates project the source asset lives in. The new lip-synced video lands here.'),
2091
2143
  sourceAssetId: z.string().uuid().describe('Asset id of the still image (avatar flow) or video clip (lip-sync flow). Must already exist in the project — use slates_upload_reference_image or slates_generate_image / slates_generate_video first if needed.'),
2092
2144
  sourceType: z.enum(['image', 'video']).describe('"image" = animate a still portrait (avatar). "video" = re-sync an existing talking-head clip. Determines pricing — be deliberate.'),
2093
- audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText (on Seedance the line is spoken natively in the generation). "upload" = use the file at audioFilePath (absolute path on the user\'s machine; ≤15s on Seedance).'),
2145
+ audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText. "upload" = use the file at audioFilePath (absolute path on the user\'s machine).'),
2094
2146
  ttsText: z.string().min(1).max(2000).optional().describe('Required when audioMethod=tts. The exact words the avatar/clip will speak.'),
2095
- ttsVoice: z.string().optional().describe('Kling engine only — voice id (e.g. "oversea_male1"). See slates-prompting-lip-sync skill for the voice catalog. Ignored on Seedance (a video source keeps its own voice; otherwise describe the voice in ttsText context).'),
2096
- ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling engine only — TTS language. Default EN.'),
2097
- ttsSpeed: z.number().min(0.5).max(2).optional().describe('Kling engine only — TTS speech rate. Default 1.0. Range 0.5-2.0.'),
2147
+ ttsVoice: z.string().optional().describe('Voice id (e.g. "oversea_male1"). See slates-prompting-lip-sync skill for the voice catalog.'),
2148
+ ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('TTS language. Default EN.'),
2149
+ ttsSpeed: z.number().min(0.5).max(2).optional().describe('TTS speech rate. Default 1.0. Range 0.5-2.0.'),
2098
2150
  audioFilePath: z.string().optional().describe('Required when audioMethod=upload. Absolute path to the audio file on the user\'s machine (mp3, wav, m4a).'),
2099
- avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('Kling engine, image-source only. avatar-standard (~14 credits/5s) for general use. avatar-pro (~29 credits/5s) for sharper face fidelity.'),
2100
- klingProvider: z.enum(['fal', 'kling']).optional().describe('Kling engine only — provider routing. Leave unset: all agent generations bill Slates credits (BYOK is retired).'),
2101
- engine: z.enum(['kling', 'seedance-2']).optional().describe('Default kling (cheap utility). seedance-2 = premium single-pass: natural speech generated in the video, voice cloned from a video source, audio included. Credits only.'),
2102
- videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
2103
- aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
2104
- seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the source (default TRUE for lip-sync; the faceless route would reject it). Bills the -face key.'),
2105
- seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the source shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
2106
- realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
2107
- sourceSeconds: z.number().optional().describe('Seedance engine + sourceType=video: the source clip\'s duration in seconds (from the asset listing). Feeds the vref cost key (input+output billing).'),
2108
- audioSeconds: z.number().optional().describe('Seedance engine + audioMethod=upload: the audio file\'s duration in seconds — sets the output length (4-15s).'),
2151
+ avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('Image-source only. avatar-standard (~14 credits/5s) for general use. avatar-pro (~29 credits/5s) for sharper face fidelity.'),
2152
+ klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. Leave unset: all agent generations bill Slates credits (BYOK is retired).'),
2109
2153
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2110
2154
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required for avatar-pro.'),
2111
2155
  }),
@@ -2124,41 +2168,8 @@ export const generateLipSync = {
2124
2168
  message: 'audioMethod=upload requires audioFilePath. Pass an absolute path to the audio file on the user\'s machine.',
2125
2169
  });
2126
2170
  }
2127
- const isSeedance = input.engine === 'seedance-2';
2128
2171
  let costKey;
2129
- let seedanceDuration = 0;
2130
- if (isSeedance) {
2131
- // Consent gate before any spend, mirroring slates_generate_video.
2132
- if (input.seedanceRealFace && !input.realFaceConsent) {
2133
- return ok({
2134
- requires_clarification: true,
2135
- missing: ['realFaceConsent'],
2136
- message: 'Real-person lip-sync needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
2137
- });
2138
- }
2139
- if (input.sourceType === 'video' && !input.sourceSeconds) {
2140
- return ok({
2141
- requires_clarification: true,
2142
- missing: ['sourceSeconds'],
2143
- message: 'Seedance lip-sync on a video source bills combined input+output seconds. Pass sourceSeconds (the clip\'s duration from slates_list_assets, must be 2-15s).',
2144
- });
2145
- }
2146
- const clamp = (n) => Math.min(15, Math.max(4, Math.ceil(n)));
2147
- seedanceDuration = clamp(input.sourceType === 'video' && input.sourceSeconds
2148
- ? input.sourceSeconds
2149
- : input.audioSeconds ?? (input.ttsText ? input.ttsText.length / 13 : 5));
2150
- costKey = videoCostKey({
2151
- model: 'seedance-2',
2152
- duration: seedanceDuration,
2153
- videoResolution: input.videoResolution ?? '1080p',
2154
- // Lip-sync sources are faces by definition — face route unless
2155
- // explicitly disabled or escalated to realface.
2156
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
2157
- seedanceRealFace: input.seedanceRealFace === true,
2158
- videoRefSeconds: input.sourceType === 'video' ? input.sourceSeconds ?? 0 : 0,
2159
- });
2160
- }
2161
- else if (input.sourceType === 'video') {
2172
+ if (input.sourceType === 'video') {
2162
2173
  costKey = 'kling-lip-sync-video-5s';
2163
2174
  }
2164
2175
  else {
@@ -2187,7 +2198,7 @@ export const generateLipSync = {
2187
2198
  estimated_cents: totalCents,
2188
2199
  estimated_credits: totalCents,
2189
2200
  source_ref: sourceRef,
2190
- message: `Cost: ${fmtCredits(totalCents)} for ${isSeedance ? `${seedanceDuration}s Seedance` : '5s'} lip-sync (${costKey}). ` +
2201
+ message: `Cost: ${fmtCredits(totalCents)} for 5s lip-sync (${costKey}). ` +
2191
2202
  `Source: ${sourceRef}. ${audioPreview}. ` +
2192
2203
  `Re-call with confirm=true after the user explicitly OKs the spend. ` +
2193
2204
  `When discussing with the user, refer to the source by its code (matches the gallery badge).`,
@@ -2210,28 +2221,12 @@ export const generateLipSync = {
2210
2221
  klingProvider: input.klingProvider,
2211
2222
  estimatedCost: totalCents,
2212
2223
  background: input.background,
2213
- // Seedance engine passthrough — the desktop delegates to the seedance
2214
- // ref-to-video path (vref billing, face cascade, consent gate). The
2215
- // durations ride along so the desktop bills exactly what was quoted.
2216
- ...(isSeedance
2217
- ? {
2218
- lipSyncEngine: 'seedance-2',
2219
- duration: seedanceDuration,
2220
- videoResolution: input.videoResolution,
2221
- aspectRatio: input.aspectRatio,
2222
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
2223
- seedanceRealFace: input.seedanceRealFace === true,
2224
- realFaceConsent: input.realFaceConsent === true,
2225
- sourceDurationSeconds: input.sourceSeconds,
2226
- audioDurationSeconds: input.audioSeconds,
2227
- }
2228
- : {}),
2229
2224
  });
2230
2225
  if (!result.success)
2231
2226
  throw new Error(result.error ?? 'Lip-sync generation failed');
2232
2227
  if (result.background) {
2233
2228
  const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
2234
- return backgroundSubmitted(`${isSeedance ? `${seedanceDuration}s` : '5s'} lip-sync (${costKey})`, ids, {
2229
+ return backgroundSubmitted(`5s lip-sync (${costKey})`, ids, {
2235
2230
  variant: costKey,
2236
2231
  projectId: input.projectId,
2237
2232
  sourceAssetId: input.sourceAssetId,
@@ -2240,7 +2235,7 @@ export const generateLipSync = {
2240
2235
  });
2241
2236
  }
2242
2237
  return {
2243
- text: `Generated ${isSeedance ? `${seedanceDuration}s` : '5s'} lip-sync (${costKey}) into project ${input.projectId} ` +
2238
+ text: `Generated 5s lip-sync (${costKey}) into project ${input.projectId} ` +
2244
2239
  `for ${fmtCredits(totalCents)}. ` +
2245
2240
  (input.audioMethod === 'tts'
2246
2241
  ? `Spoken: "${(input.ttsText ?? '').slice(0, 60)}${(input.ttsText ?? '').length > 60 ? '...' : ''}"`
@@ -2261,58 +2256,21 @@ export const generateLipSync = {
2261
2256
  // ── Generate motion transfer ────────────────────────────────────
2262
2257
  export const generateMotionTransfer = {
2263
2258
  id: 'slates_generate_motion_transfer',
2264
- description: 'Transfer the motion from a reference video onto a target image character. Two engines — route per slates-model-selection: (1) kling-mc-std ($0.95 / 5s) / kling-mc-pro ($1.26 / 5s) — the cheap utility lane, structured skeleton/depth retargeting, always 5s. (2) motionModel=seedance-2 — the PREMIUM lane: single-pass generation with the driving clip as a native conditioning signal (better motion fidelity + native audio), prompt-driven (write what the character does, e.g. "the character from image 1 performs the exact motion from video 1"), bills input+output seconds on seedance vref keys (pass sourceVideoSeconds; driving clip must be 2-15s). Faces: seedanceFace defaults true; a REAL person needs seedanceRealFace+realFaceConsent (premium route). REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. All tiers hit the >$0.50 confirm gate.',
2259
+ description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std ($0.95 / 5s) or kling-mc-pro ($1.26 / 5s), structured skeleton/depth retargeting, always 5s. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. Both tiers hit the >$0.50 confirm gate.',
2265
2260
  input: z.object({
2266
2261
  projectId: z.string().uuid().describe('Slates project. Both source and target assets must live here.'),
2267
- sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Seedance engine: 2-15s clips only.'),
2262
+ sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s.'),
2268
2263
  targetImageAssetId: z.string().uuid().describe('Asset id of the target image (the character that will perform the motion). Must already exist in the project.'),
2269
- motionModel: z.enum(['kling-mc-std', 'kling-mc-pro', 'seedance-2']).optional().describe('kling-mc-std (~32 credits) general motion; kling-mc-pro (~42 credits) cleaner anatomy — default. seedance-2 = premium single-pass lane (prompt-driven, native audio, input+output-second billing) — pick when motion fidelity or audio matters.'),
2270
- characterOrientation: z.enum(['video', 'image']).optional().describe('Kling only. "video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
2271
- prompt: z.string().optional().describe('Kling: optional refinement. Seedance: THE driver — describe what the character does with the motion from the clip (ordinal references: "the character from image 1 performs the motion from video 1"). A sensible default recipe is used if omitted. Read slates-prompting-motion-transfer.'),
2272
- klingProvider: z.enum(['fal', 'kling']).optional().describe('Kling engine only — provider routing. "fal" (default) uses Slates credits.'),
2273
- duration: z.number().int().min(4).max(15).optional().describe('Seedance engine only — output duration in seconds (4-15). Defaults to the driving clip\'s length.'),
2274
- videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
2275
- aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
2276
- seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the clip/image (default TRUE for motion transfer). Bills the -face key.'),
2277
- seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the driving clip/subject shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
2278
- realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
2279
- sourceVideoSeconds: z.number().optional().describe('Seedance engine: the driving clip\'s duration in seconds (from the asset listing, 2-15s). Feeds the vref cost key (input+output billing).'),
2264
+ motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std (~32 credits) general motion; kling-mc-pro (~42 credits) cleaner anatomy — default.'),
2265
+ characterOrientation: z.enum(['video', 'image']).optional().describe('"video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
2266
+ prompt: z.string().optional().describe('Optional refinement. Read slates-prompting-motion-transfer.'),
2267
+ klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. "fal" (default) uses Slates credits.'),
2280
2268
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2281
2269
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required — both tiers exceed.'),
2282
2270
  }),
2283
2271
  async run(input, ctx) {
2284
2272
  const motionModel = input.motionModel ?? 'kling-mc-pro';
2285
- const isSeedance = motionModel === 'seedance-2';
2286
- let costKey;
2287
- let seedanceDuration = 0;
2288
- if (isSeedance) {
2289
- if (input.seedanceRealFace && !input.realFaceConsent) {
2290
- return ok({
2291
- requires_clarification: true,
2292
- missing: ['realFaceConsent'],
2293
- message: 'Real-person motion transfer needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
2294
- });
2295
- }
2296
- if (!input.sourceVideoSeconds) {
2297
- return ok({
2298
- requires_clarification: true,
2299
- missing: ['sourceVideoSeconds'],
2300
- message: 'Seedance motion transfer bills combined input+output seconds. Pass sourceVideoSeconds (the driving clip\'s duration from slates_list_assets, must be 2-15s).',
2301
- });
2302
- }
2303
- seedanceDuration = input.duration ?? Math.min(15, Math.max(4, Math.ceil(input.sourceVideoSeconds)));
2304
- costKey = videoCostKey({
2305
- model: 'seedance-2',
2306
- duration: seedanceDuration,
2307
- videoResolution: input.videoResolution ?? '1080p',
2308
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
2309
- seedanceRealFace: input.seedanceRealFace === true,
2310
- videoRefSeconds: input.sourceVideoSeconds,
2311
- });
2312
- }
2313
- else {
2314
- costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
2315
- }
2273
+ const costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
2316
2274
  const cloud = ctx.cloud();
2317
2275
  const registry = await cloud.get('/api/agent/models');
2318
2276
  const entry = registry.models.find((m) => m.model === costKey);
@@ -2336,9 +2294,9 @@ export const generateMotionTransfer = {
2336
2294
  estimated_credits: totalCents,
2337
2295
  source_ref: source,
2338
2296
  target_ref: target,
2339
- message: `Cost: ${fmtCredits(totalCents)} for ${isSeedance ? `${seedanceDuration}s Seedance motion transfer` : `5s ${motionModel}`} (${costKey}). ` +
2297
+ message: `Cost: ${fmtCredits(totalCents)} for 5s ${motionModel} (${costKey}). ` +
2340
2298
  `Transferring motion from ${source} onto ${target}. ` +
2341
- `Re-call with confirm=true after the user explicitly OKs the spend${isSeedance ? '' : ', or pick kling-mc-std to save ~10 credits'}. ` +
2299
+ `Re-call with confirm=true after the user explicitly OKs the spend, or pick kling-mc-std to save ~10 credits. ` +
2342
2300
  `When discussing with the user, refer to the assets by those codes — they'll match the gallery badges.`,
2343
2301
  });
2344
2302
  }
@@ -2356,27 +2314,12 @@ export const generateMotionTransfer = {
2356
2314
  klingProvider: input.klingProvider,
2357
2315
  estimatedCost: totalCents,
2358
2316
  background: input.background,
2359
- // Seedance engine passthrough — the desktop delegates to the seedance
2360
- // ref-to-video path (vref billing, face cascade, consent gate).
2361
- ...(isSeedance
2362
- ? {
2363
- duration: seedanceDuration,
2364
- videoResolution: input.videoResolution,
2365
- aspectRatio: input.aspectRatio,
2366
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
2367
- seedanceRealFace: input.seedanceRealFace === true,
2368
- realFaceConsent: input.realFaceConsent === true,
2369
- // Ride the caller-supplied clip duration through — the asset row's
2370
- // duration can be null for imported clips.
2371
- sourceVideoDurationSeconds: input.sourceVideoSeconds,
2372
- }
2373
- : {}),
2374
2317
  });
2375
2318
  if (!result.success)
2376
2319
  throw new Error(result.error ?? 'Motion transfer generation failed');
2377
2320
  if (result.background) {
2378
2321
  const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
2379
- return backgroundSubmitted(`${isSeedance ? `${seedanceDuration}s` : '5s'} motion transfer (${motionModel})`, ids, {
2322
+ return backgroundSubmitted(`5s motion transfer (${motionModel})`, ids, {
2380
2323
  variant: costKey,
2381
2324
  motionModel,
2382
2325
  projectId: input.projectId,
@@ -2387,7 +2330,7 @@ export const generateMotionTransfer = {
2387
2330
  });
2388
2331
  }
2389
2332
  return {
2390
- text: `Generated ${isSeedance ? `${seedanceDuration}s` : '5s'} motion transfer (${motionModel}) into project ${input.projectId} ` +
2333
+ text: `Generated 5s motion transfer (${motionModel}) into project ${input.projectId} ` +
2391
2334
  `for ${fmtCredits(totalCents)}.` +
2392
2335
  (input.prompt ? ` Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` : ''),
2393
2336
  data: {
@@ -2407,15 +2350,17 @@ export const generateMotionTransfer = {
2407
2350
  // ── Edit video (Kling O3 video-to-video) ────────────────────────
2408
2351
  export const editVideo = {
2409
2352
  id: 'slates_edit_video',
2410
- description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements) or omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
2353
+ description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat), or seedance-2.5-edit (4–30s clips — the ONLY engine that takes a clip over 15s; 480p/720p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds — always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
2411
2354
  input: z.object({
2412
2355
  projectId: z.string().uuid().describe('Project the source clip lives in.'),
2413
2356
  sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
2414
2357
  prompt: z.string().min(1).max(2500).describe('The change, not the whole scene — e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
2415
- model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs.'),
2358
+ model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit', 'seedance-2.5-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs. seedance-2.5-edit (480p/720p, 4–30s) is the ONLY engine that accepts a clip longer than 15s; it also takes NO character/style refs on this op.'),
2416
2359
  characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY — rejected on omni-flash-edit.'),
2417
2360
  styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
2418
- keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit output carries its own audio).'),
2361
+ keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
2362
+ videoResolution: z.enum(['480p', '720p']).optional().describe('seedance-2.5-edit ONLY (default 720p). Ignored by the Kling and Omni Flash engines, whose output follows the source clip.'),
2363
+ seedanceFace: z.boolean().optional().describe('seedance-2.5-edit ONLY: set true when a CHARACTER FACE is visible in the clip. Faceless edits run on BytePlus; faces are blocked there and must route to the relaxed provider, which costs ~35% more. A face edit submitted without this flag is rejected by the provider, not silently downgraded.'),
2419
2364
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2420
2365
  confirm: z.boolean().optional().describe('Set true to bypass the cost confirm gate after the user OKs the spend.'),
2421
2366
  }),
@@ -2427,10 +2372,13 @@ export const editVideo = {
2427
2372
  }
2428
2373
  const model = input.model ?? 'kling-v3.0-omni-edit';
2429
2374
  const isOmniFlashEdit = model === 'omni-flash-edit';
2430
- // Kling edit: 3–15s source clips; Omni Flash edit: 3–10s.
2431
- const maxClipSeconds = isOmniFlashEdit ? 10 : 15;
2432
- if (isOmniFlashEdit && ((input.characterAssetIds?.length ?? 0) > 0 || (input.styleAssetIds?.length ?? 0) > 0)) {
2433
- throw new Error('omni-flash-edit is prompt-only — it takes no character/style reference images. Drop the refs, or switch to kling-v3.0-omni-edit which supports elements.');
2375
+ const isSeedanceEdit = model === 'seedance-2.5-edit';
2376
+ // Kling edit: 3–15s source clips; Omni Flash edit: 3–10s; Seedance 2.5
2377
+ // edit: 4–30s — the only engine that takes a clip over 15s.
2378
+ const minClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MIN_SECONDS : 3;
2379
+ const maxClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MAX_SECONDS : isOmniFlashEdit ? 10 : 15;
2380
+ if ((isOmniFlashEdit || isSeedanceEdit) && ((input.characterAssetIds?.length ?? 0) > 0 || (input.styleAssetIds?.length ?? 0) > 0)) {
2381
+ throw new Error(`${model} takes the prompt and the source clip only on this op — no character/style reference images. Drop the refs, or switch to kling-v3.0-omni-edit which supports elements.`);
2434
2382
  }
2435
2383
  // Resolve refs (UUIDs or badge codes) against the project AT CALL TIME.
2436
2384
  const refInputs = [
@@ -2459,12 +2407,24 @@ export const editVideo = {
2459
2407
  if (!Number.isFinite(clipSeconds) || clipSeconds <= 0) {
2460
2408
  throw new Error('Source clip has no recorded duration — cannot quote the edit. Re-import the clip or pick another.');
2461
2409
  }
2462
- if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds < 2.95) {
2463
- throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts 3–${maxClipSeconds}s. ` +
2464
- `Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip.`);
2465
- }
2466
- const billedSeconds = Math.min(maxClipSeconds, Math.max(3, Math.ceil(clipSeconds - 0.05)));
2467
- const costKey = isOmniFlashEdit ? omniFlashEditCostKey(billedSeconds) : klingEditCostKey(model, billedSeconds);
2410
+ if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds < minClipSeconds - 0.05) {
2411
+ throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts ${minClipSeconds}–${maxClipSeconds}s. ` +
2412
+ `Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip` +
2413
+ (maxClipSeconds < SEEDANCE_25_EDIT_MAX_SECONDS ? ', or switch to seedance-2.5-edit which accepts up to 30s' : '') +
2414
+ '.');
2415
+ }
2416
+ const billedSeconds = Math.min(maxClipSeconds, Math.max(minClipSeconds, Math.ceil(clipSeconds - 0.05)));
2417
+ // Three engines, three key shapes — only Seedance's carries a resolution and
2418
+ // a face route, because only its price moves with them.
2419
+ const costKey = isSeedanceEdit
2420
+ ? seedanceEditCostKey({
2421
+ duration: billedSeconds,
2422
+ videoResolution: input.videoResolution ?? '720p',
2423
+ seedanceFace: input.seedanceFace === true,
2424
+ })
2425
+ : isOmniFlashEdit
2426
+ ? omniFlashEditCostKey(billedSeconds)
2427
+ : klingEditCostKey(model, billedSeconds);
2468
2428
  const cloud = ctx.cloud();
2469
2429
  const registry = await cloud.get('/api/agent/models');
2470
2430
  const entry = registry.models.find((m) => m.model === costKey);
@@ -2503,6 +2463,8 @@ export const editVideo = {
2503
2463
  characterAssetIds,
2504
2464
  styleAssetIds,
2505
2465
  keepAudio: input.keepAudio !== false,
2466
+ videoResolution: input.videoResolution,
2467
+ seedanceFace: input.seedanceFace,
2506
2468
  background: input.background,
2507
2469
  });
2508
2470
  if (!result.success)
@@ -3105,6 +3067,53 @@ export const updateFrame = {
3105
3067
  }));
3106
3068
  },
3107
3069
  };
3070
+ /**
3071
+ * Batch form of `slates_update_frame`.
3072
+ *
3073
+ * Writing a scene's worth of frames — motion prompts, shot labels — is ONE
3074
+ * logical edit. Doing it as N `slates_update_frame` calls costs N LLM
3075
+ * round-trips and makes the desktop refetch the whole storyboard N times for a
3076
+ * single intent. This is also where the storyboard's deleted "generate motion
3077
+ * prompts" button's capability went: the agent can be told "redo scene 3,
3078
+ * handheld", which that fixed-shape button never could.
3079
+ *
3080
+ * Hits `POST /agent/frames/batch-update`, which validates every id BEFORE the
3081
+ * first write and emits one broadcast for the batch.
3082
+ */
3083
+ export const batchUpdateFrames = {
3084
+ id: 'slates_batch_update_frames',
3085
+ description: 'Update MANY frames in one call — motion prompts, shot labels, notes, asset binding, scene/position, frameType. Prefer this over repeated slates_update_frame when writing a scene or a whole storyboard: it is one round-trip and one UI refresh. Every id is validated before anything is written, so the batch never lands half-applied.',
3086
+ input: z.object({
3087
+ updates: z
3088
+ .array(z.object({
3089
+ frameId: z.string().uuid(),
3090
+ shotLabel: z.string().optional(),
3091
+ notes: z.string().optional(),
3092
+ assetId: z.string().uuid().nullable().optional(),
3093
+ sceneId: z.string().uuid().nullable().optional(),
3094
+ position: z.number().int().min(0).optional(),
3095
+ frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
3096
+ motionPrompt: z.string().nullable().optional(),
3097
+ }))
3098
+ .min(1),
3099
+ }),
3100
+ async run(input, ctx) {
3101
+ return ok(await ctx.desktop().post('/agent/frames/batch-update', {
3102
+ updates: input.updates.map((u) => ({
3103
+ id: u.frameId,
3104
+ data: {
3105
+ shotLabel: u.shotLabel,
3106
+ notes: u.notes,
3107
+ assetId: u.assetId,
3108
+ sceneId: u.sceneId,
3109
+ position: u.position,
3110
+ frameType: u.frameType,
3111
+ motionPrompt: u.motionPrompt,
3112
+ },
3113
+ })),
3114
+ }));
3115
+ },
3116
+ };
3108
3117
  export const deleteFrame = {
3109
3118
  id: 'slates_delete_frame',
3110
3119
  description: 'Delete a frame from its scene (the referenced asset is untouched).',
@@ -3152,13 +3161,30 @@ function resolveGuideTopic(topic) {
3152
3161
  // Audio — seed-audio BEFORE the seedance check: "seed-audio" also starts
3153
3162
  // with "seed", and falling through would hand the video guide to the audio
3154
3163
  // model (the exact class of aliasing bug this comment block warns about).
3155
- if (t.startsWith('seed-audio') || t === 'seed audio' || t === 'audio')
3164
+ // Speech, dialogue and scratch VO all live on Seed Audio now — the TTS
3165
+ // surface is gone, so "tts"/"voiceover" must NOT land on the ElevenLabs
3166
+ // guide, which is SFX-only.
3167
+ if (t.startsWith('seed-audio') ||
3168
+ t === 'seed audio' ||
3169
+ t === 'audio' ||
3170
+ t === 'tts' ||
3171
+ t === 'voiceover' ||
3172
+ t === 'dialogue') {
3156
3173
  return 'slates-prompting-seed-audio';
3157
- if (t.startsWith('eleven') || t.startsWith('elevenlabs') || t === 'tts' || t === 'sfx' || t === 'sound-effects' || t === 'sound effects' || t === 'voiceover') {
3174
+ }
3175
+ if (t.startsWith('eleven') || t.startsWith('elevenlabs') || t === 'sfx' || t === 'sound-effects' || t === 'sound effects') {
3158
3176
  return 'slates-prompting-elevenlabs';
3159
3177
  }
3160
- if (t.startsWith('suno') || t === 'music')
3161
- return 'slates-prompting-suno';
3178
+ // ⚠️ 2.5 BEFORE the generic `seedance` prefix — the same ordering trap as
3179
+ // seed-audio-before-seedance above. Falling through silently hands the 2.0
3180
+ // guide to 2.5, whose limits, resolutions and task types are all different.
3181
+ if (t.startsWith('seedance-2.5') ||
3182
+ t.startsWith('seedance-25') ||
3183
+ t.startsWith('seedance-2-5') ||
3184
+ t.startsWith('seedance2.5') ||
3185
+ t === 'seedance 2.5') {
3186
+ return 'slates-prompting-seedance-2-5';
3187
+ }
3162
3188
  if (t.startsWith('seedance'))
3163
3189
  return 'slates-prompting-seedance';
3164
3190
  if (t.startsWith('avatar-') || t.includes('lip-sync'))
@@ -3185,7 +3211,7 @@ export const getPromptingGuide = {
3185
3211
  topic: z
3186
3212
  .string()
3187
3213
  .min(1)
3188
- .describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-suno, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
3214
+ .describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
3189
3215
  }),
3190
3216
  async run(input) {
3191
3217
  const resolved = resolveGuideTopic(input.topic);
@@ -3274,6 +3300,7 @@ export const ALL_OPERATIONS = [
3274
3300
  deleteScene,
3275
3301
  reorderScenes,
3276
3302
  updateFrame,
3303
+ batchUpdateFrames,
3277
3304
  deleteFrame,
3278
3305
  getPromptingGuide,
3279
3306
  ];