@slatesvideo/shared 0.5.5 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. package/dist/index.d.ts +2 -2
  2. package/dist/index.js +6 -3
  3. package/dist/operations/index.d.ts +106 -19
  4. package/dist/operations/index.js +624 -172
  5. package/dist/prompts/character-sheet.d.ts +2 -1
  6. package/dist/prompts/character-sheet.js +82 -13
  7. package/dist/prompts/model-facts.d.ts +43 -1
  8. package/dist/prompts/model-facts.js +104 -2
  9. package/dist/prompts/prompting-tips.d.ts +1 -1
  10. package/dist/prompts/prompting-tips.js +208 -5
  11. package/dist/prompts/reference-composer.d.ts +21 -3
  12. package/dist/prompts/reference-composer.js +80 -10
  13. package/dist/prompts/reference-rules.d.ts +19 -2
  14. package/dist/prompts/reference-rules.js +18 -1
  15. package/dist/skills/content.js +8 -5
  16. package/exports/slates-prompt-builder/generated/SKILL.md +2 -2
  17. package/exports/slates-prompt-builder/generated/reference-character.md +9 -4
  18. package/exports/slates-prompt-builder/generated/reference-seedance.md +12 -1
  19. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +11 -11
  20. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  21. package/package.json +1 -1
  22. package/skills/slates-character-identity.md +9 -4
  23. package/skills/slates-model-selection.md +39 -11
  24. package/skills/slates-prompting-elevenlabs.md +69 -0
  25. package/skills/slates-prompting-lip-sync.md +12 -14
  26. package/skills/slates-prompting-motion-transfer.md +18 -14
  27. package/skills/slates-prompting-seed-audio.md +110 -0
  28. package/skills/slates-prompting-seedance-2-5.md +215 -0
  29. package/skills/slates-prompting-seedance.md +17 -1
@@ -13,6 +13,9 @@ import { z } from 'zod';
13
13
  import { SlatesCloudClient } from '../clients/cloud.js';
14
14
  import { SlatesDesktopClient } from '../clients/desktop.js';
15
15
  import { SKILLS } from '../skills/content.js';
16
+ // Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
17
+ // "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
18
+ import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } from '../prompts/model-facts.js';
16
19
  export function defaultContext() {
17
20
  return {
18
21
  cloud: () => new SlatesCloudClient(),
@@ -120,8 +123,11 @@ export const listAvailableModels = {
120
123
  const table = models.map((m) => `${m.model} ${creditCost(m)}`).join('\n');
121
124
  return {
122
125
  text: `${models.length} COST keys (credits per generation)${input.filter ? ` matching "${input.filter}"` : ''}. ` +
123
- `NOTE: these are billing keys for cost lookup ONLY — the \`model\` param on slates_generate_video takes a BASE id ` +
124
- `(kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | veo-3.1-fast | veo-3.1-standard) with duration/videoResolution as separate params:\n` +
126
+ `NOTE: these are billing keys for cost lookup ONLY — the \`model\` param on the generate ops takes a BASE id. ` +
127
+ // Derived from the SSOT arrays, not restated: a hand-written list here
128
+ // drifts the moment a model lands (it already omitted omni-flash).
129
+ `slates_generate_video: ${VIDEO_MODELS.join(' | ')} (duration/videoResolution as separate params). ` +
130
+ `slates_generate_audio: ${AUDIO_MODELS.join(' | ')} (durationSeconds as a separate param):\n` +
125
131
  table,
126
132
  data: { count: models.length },
127
133
  };
@@ -133,7 +139,7 @@ export const estimateGenerationCost = {
133
139
  input: z.object({
134
140
  model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
135
141
  quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
136
- duration: z.number().int().min(3).max(15).optional().describe('Video only — seconds. Cost scales linearly; required with a video base id.'),
142
+ duration: z.number().int().min(1).max(360).optional().describe('Seconds. Video 3-15 (cost scales linearly; required with a video base id). Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22 — required with either audio base id.'),
137
143
  videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Video only. Seedance defaults to 1080p.'),
138
144
  resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
139
145
  quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
@@ -152,6 +158,42 @@ export const estimateGenerationCost = {
152
158
  if (img)
153
159
  key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
154
160
  }
161
+ // 2a) audio base id → seconds. Both surfaces bill per second, so a
162
+ // duration is always required. Runs BEFORE the video resolver: it is
163
+ // forgiving by design and "seed-audio" would otherwise be mistaken for
164
+ // a seedance spelling.
165
+ if (!key && AUDIO_MODELS.includes(input.model)) {
166
+ const m = input.model;
167
+ if (!input.duration) {
168
+ return ok({
169
+ requires_clarification: true,
170
+ missing: ['duration'],
171
+ message: m === 'seed-audio'
172
+ ? 'Seed Audio cost scales with the REQUESTED duration — and that duration is what the user is billed regardless of what comes back (it has no duration parameter; the number is written into the prompt). Pass duration in seconds (3-120).'
173
+ : 'Sound Effects bills per second — pass duration in seconds (1-22).',
174
+ });
175
+ }
176
+ // REFUSE an out-of-range duration rather than quoting the clamped price.
177
+ // audioCostKey clamps (it has to — it mirrors the desktop, which clamps),
178
+ // so without this an agent asking for 2s of Seed Audio would be handed a
179
+ // real 3s price and no hint that 2s is not a thing it can order. The
180
+ // generate op gates the same way; the two must agree or the quote is a
181
+ // promise the generation refuses to keep.
182
+ const audioBounds = {
183
+ 'seed-audio': { min: SEED_AUDIO_MIN_SECONDS, max: SEED_AUDIO_MAX_SECONDS },
184
+ 'eleven-sfx': { min: ELEVEN_SFX_MIN_SECONDS, max: ELEVEN_SFX_MAX_SECONDS },
185
+ };
186
+ const bounds = audioBounds[m];
187
+ if (input.duration < bounds.min || input.duration > bounds.max) {
188
+ return ok({
189
+ requires_clarification: true,
190
+ missing: ['duration'],
191
+ message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
192
+ 'Re-ask with a duration in range.',
193
+ });
194
+ }
195
+ key = audioCostKey({ model: m, durationSeconds: input.duration });
196
+ }
155
197
  // 2b) Kling O3 edit base id + duration (ceiled source-clip length)
156
198
  if (!key && (input.model === 'kling-v3.0-omni-edit' || input.model === 'kling-v3.0-omni-pro-edit')) {
157
199
  if (!input.duration) {
@@ -181,7 +223,12 @@ export const estimateGenerationCost = {
181
223
  duration,
182
224
  videoResolution: input.videoResolution ??
183
225
  resolved.videoResolution ??
184
- (resolved.model.startsWith('seedance') ? '1080p' : undefined),
226
+ // Seedance quotes are resolution-scaled, so a missing resolution has
227
+ // to fall back to the model's OWN default — 2.5 has no 1080p at all,
228
+ // and a blanket '1080p' here quoted a key that does not exist.
229
+ (resolved.model === 'seedance-2.5' ? '720p'
230
+ : resolved.model.startsWith('seedance') ? '1080p'
231
+ : undefined),
185
232
  sound: input.sound ?? resolved.sound,
186
233
  seedanceFace: input.seedanceFace ?? resolved.seedanceFace,
187
234
  seedanceRealFace: input.seedanceRealFace,
@@ -190,7 +237,7 @@ export const estimateGenerationCost = {
190
237
  }
191
238
  const perCredits = key != null ? byKey.get(key) : undefined;
192
239
  if (key == null || perCredits == null) {
193
- throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
240
+ throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
194
241
  }
195
242
  const qty = input.quantity ?? 1;
196
243
  const totalCredits = perCredits * qty;
@@ -439,16 +486,16 @@ export const getAssetVideoFrames = {
439
486
  };
440
487
  export const uploadReferenceImage = {
441
488
  id: 'slates_upload_reference_image',
442
- description: 'Add a reference image OR video clip to a Slates project. Pass either filePath (absolute path to a local file) or dataUrl (base64 data: URL) — exactly one. Set type:"video" on a filePath import to bring in a clip (the user\'s own footage to edit/relocate/trim); imported videos are probed on ingest, so duration + dimensions are available immediately. Default type is "image". dataUrl is image-only.',
489
+ description: 'Add a reference image, video clip, or audio file to a Slates project from disk. Pass either filePath (absolute path to a local file) or dataUrl (base64 data: URL) — exactly one. Set type:"video" to bring in a clip (the user\'s own footage to edit/relocate/trim) or type:"audio" for music/VO/SFX they already have; both are probed on ingest, so duration (and for video, dimensions) are available immediately, and audio gets its waveform. Default type is "image". dataUrl is image-only.',
443
490
  input: z
444
491
  .object({
445
492
  projectId: z.string().uuid(),
446
493
  filePath: z.string().optional(),
447
494
  dataUrl: z.string().optional(),
448
495
  type: z
449
- .enum(['image', 'video'])
496
+ .enum(['image', 'video', 'audio'])
450
497
  .optional()
451
- .describe('Asset kind for a filePath import — "image" (default) or "video". A dataUrl is always an image.'),
498
+ .describe('Asset kind for a filePath import — "image" (default), "video", or "audio". A dataUrl is always an image.'),
452
499
  })
453
500
  .refine((d) => !!d.filePath !== !!d.dataUrl, {
454
501
  message: 'Pass exactly one of filePath or dataUrl',
@@ -463,8 +510,8 @@ export const uploadReferenceImage = {
463
510
  });
464
511
  return ok(r);
465
512
  }
466
- if (input.type === 'video') {
467
- throw new Error('dataUrl uploads are image-only — pass a filePath to import a video clip.');
513
+ if (input.type === 'video' || input.type === 'audio') {
514
+ throw new Error(`dataUrl uploads are image-only — pass a filePath to import ${input.type === 'audio' ? 'an audio file' : 'a video clip'}.`);
468
515
  }
469
516
  const r = await desktop.post('/agent/assets/upload-base64', {
470
517
  projectId: input.projectId,
@@ -506,6 +553,82 @@ export const moveAssetsToFolder = {
506
553
  return ok(await ctx.desktop().post('/agent/folders/move-assets', input));
507
554
  },
508
555
  };
556
+ export const moveAssetsToProject = {
557
+ id: 'slates_move_assets_to_project',
558
+ description: 'Move assets (images, videos, audio) out of one project into another. The media files move on disk into the destination project folder — this is a real re-home, not a copy. Moved assets leave whatever gallery folder they were in and are issued fresh badge codes in the destination.',
559
+ input: z.object({
560
+ sourceProjectId: z.string().uuid(),
561
+ // Badge codes ("IMG-A8") resolve against sourceProjectId at call time.
562
+ assetIds: z.array(z.string().min(1)).min(1),
563
+ targetProjectId: z.string().uuid(),
564
+ }),
565
+ async run(input, ctx) {
566
+ if (input.sourceProjectId === input.targetProjectId) {
567
+ throw new Error('sourceProjectId and targetProjectId are the same — nothing to move.');
568
+ }
569
+ const resolved = await resolveAssetRefs(ctx, input.sourceProjectId, input.assetIds);
570
+ const assetIds = input.assetIds.map((ref) => resolved.get(ref)?.id ?? ref);
571
+ const result = await ctx.desktop().post('/agent/assets/move-to-project', {
572
+ assetIds,
573
+ targetProjectId: input.targetProjectId,
574
+ // Sent so the route can ENFORCE it. resolveAssetRefs only validates
575
+ // badge codes against the source project — a raw UUID passes straight
576
+ // through, so without this a caller could move an asset out of a
577
+ // project it never named.
578
+ sourceProjectId: input.sourceProjectId,
579
+ });
580
+ // Timeline clips key media by PATH, so the move repointed edits in OTHER
581
+ // projects at files that now live inside the destination — and deleting the
582
+ // destination deletes those files. Say it in the TEXT, not just the data:
583
+ // an agent summarising this result must be able to pass the warning on.
584
+ const refs = result.externalClipRefs ?? [];
585
+ if (refs.length > 0) {
586
+ const clips = refs.reduce((n, r) => n + r.clipCount, 0);
587
+ const where = refs.map((r) => `"${r.projectName}"`).join(', ');
588
+ return ok(result, `${JSON.stringify(result)}\n\n⚠️ ${clips} timeline clip(s) in ${where} now reference the moved ` +
589
+ `file(s) inside the destination project. Deleting the destination project would break those ` +
590
+ `edits. Tell the user — this is the one cross-project reference Slates allows.`);
591
+ }
592
+ return ok(result);
593
+ },
594
+ };
595
+ export const copyAssetsToProject = {
596
+ id: 'slates_copy_assets_to_project',
597
+ description: 'Copy assets (images, videos, audio) from one project into another. The originals stay exactly where they are — new files, new rows, new badge codes in the destination. Use this instead of slates_move_assets_to_project when the asset is already in use where it lives: an image that is a character/environment/style identity or sits in a storyboard frame CANNOT be moved out (that would leave the other project pointing at a file it no longer owns), but it can always be copied. Lineage is not copied; the copy starts clean.',
598
+ input: z.object({
599
+ sourceProjectId: z.string().uuid(),
600
+ // Badge codes ("IMG-A8") resolve against sourceProjectId at call time.
601
+ assetIds: z.array(z.string().min(1)).min(1),
602
+ targetProjectId: z.string().uuid(),
603
+ }),
604
+ async run(input, ctx) {
605
+ if (input.sourceProjectId === input.targetProjectId) {
606
+ throw new Error('sourceProjectId and targetProjectId are the same — nothing to copy.');
607
+ }
608
+ const resolved = await resolveAssetRefs(ctx, input.sourceProjectId, input.assetIds);
609
+ const assetIds = input.assetIds.map((ref) => resolved.get(ref)?.id ?? ref);
610
+ return ok(await ctx.desktop().post('/agent/assets/copy-to-project', {
611
+ assetIds,
612
+ targetProjectId: input.targetProjectId,
613
+ // Enforced route-side for the same reason as the move op: resolveAssetRefs
614
+ // only validates badge codes, so a raw UUID would otherwise let a caller
615
+ // copy out of a project it never named.
616
+ sourceProjectId: input.sourceProjectId,
617
+ }));
618
+ },
619
+ };
620
+ export const moveEntityToProject = {
621
+ id: 'slates_move_entity_to_project',
622
+ description: "Move a character, environment, or style into another project, taking every image it references with it. This is the fix when slates_move_assets_to_project refuses an asset because an entity still uses it: the identity image can't leave on its own, but the whole entity can. Refuses (rather than cascading) if one of its images is ALSO used by something else — copy the images instead in that case. Storyboard frames are not movable this way; moving a frame's image out would empty the shot.",
623
+ input: z.object({
624
+ kind: z.enum(['character', 'environment', 'style']),
625
+ entityId: z.string().uuid(),
626
+ targetProjectId: z.string().uuid(),
627
+ }),
628
+ async run(input, ctx) {
629
+ return ok(await ctx.desktop().post('/agent/entities/move-to-project', input));
630
+ },
631
+ };
509
632
  // ── Characters ──────────────────────────────────────────────────
510
633
  export const listCharacters = {
511
634
  id: 'slates_list_characters',
@@ -1197,6 +1320,11 @@ export const VIDEO_MODELS = [
1197
1320
  'veo-3.1-fast',
1198
1321
  'veo-3.1-standard',
1199
1322
  'seedance-2',
1323
+ // Seedance 2.5 is a SECOND SEAT, not a replacement: 30s takes, 30 image
1324
+ // references, audio-only references — and 480p/720p ONLY. 2.0 keeps the
1325
+ // ladder to native 4K and stays the default. Its EDIT row is not here; edit
1326
+ // models live on slates_edit_video, same as the Kling and Omni Flash ones.
1327
+ 'seedance-2.5',
1200
1328
  'omni-flash',
1201
1329
  ];
1202
1330
  // Model → registry cost-key. Each provider's keys ship with their own
@@ -1220,16 +1348,24 @@ const KLING_TIER_MAP = {
1220
1348
  // asserts this builder byte-matches the desktop's klingCreditKey/seedanceCreditKey.
1221
1349
  export function videoCostKey(input) {
1222
1350
  if (input.model.startsWith('seedance')) {
1223
- // Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (face × vref ×
1224
- // res × duration). AI-face route bills the `-face-` key (~45% over faceless);
1225
- // consented real-person route bills the premium `-realface-` key (fal partner
1226
- // endpoint). A reference video flips to `-vref-{res}-{T}s`, T = in + out (6..30).
1227
- const res = input.videoResolution ?? '1080p';
1351
+ // Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
1352
+ // × vref × res × duration). AI-face route bills the `-face-` key (~45% over
1353
+ // faceless); consented real-person route bills the premium `-realface-` key
1354
+ // (fal partner endpoint). A reference video flips to `-vref-{res}-{T}s`,
1355
+ // T = in + out.
1356
+ //
1357
+ // ⚠️ EVERY BOUND HERE IS VERSION-SCOPED. 2.5 is 480p/720p only, runs to 30s,
1358
+ // and takes references to 30s combined — so its vref total reaches 60, DOUBLE
1359
+ // 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would quote a real key at a
1360
+ // fraction of the real bill.
1361
+ const v25 = input.model.startsWith('seedance-2.5');
1362
+ const res = input.videoResolution ?? (v25 ? '720p' : '1080p');
1228
1363
  const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
1229
1364
  const vrefSecs = input.videoRefSeconds ?? 0;
1230
1365
  if (vrefSecs > 0) {
1231
1366
  // ceil(x - 0.05) matches the server's probe rounding — quote = bill.
1232
- const total = Math.min(30, Math.max(6, Math.ceil(vrefSecs - 0.05) + input.duration));
1367
+ const maxTotal = v25 ? SEEDANCE_25_VREF_MAX_TOTAL : SEEDANCE_20_VREF_MAX_TOTAL;
1368
+ const total = Math.min(maxTotal, Math.max(6, Math.ceil(vrefSecs - 0.05) + input.duration));
1233
1369
  return `${input.model}${face}-vref-${res}-${total}s`;
1234
1370
  }
1235
1371
  return `${input.model}${face}-${res}-${input.duration}s`;
@@ -1279,6 +1415,91 @@ export function klingEditCostKey(model, duration) {
1279
1415
  export function omniFlashEditCostKey(duration) {
1280
1416
  return `omni-flash-edit-${duration}s`;
1281
1417
  }
1418
+ /** Max billed (input + output) seconds on a Seedance video-reference gen.
1419
+ * 2.0: refs 2–15s + output 4–15s. 2.5: refs to 30s + output to 30s. */
1420
+ const SEEDANCE_20_VREF_MAX_TOTAL = 30;
1421
+ const SEEDANCE_25_VREF_MAX_TOTAL = 60;
1422
+ /** Seedance 2.5 source-clip bounds for the edit task type. */
1423
+ export const SEEDANCE_25_EDIT_MIN_SECONDS = 4;
1424
+ export const SEEDANCE_25_EDIT_MAX_SECONDS = 30;
1425
+ // Seedance 2.5 video-edit cost key — mirrors seedanceCreditKey() in
1426
+ // slate/src/shared/pricing.ts for the edit row (must byte-match; checked by the
1427
+ // slates-api pricing-consistency script). Unlike the Kling and Omni Flash edit
1428
+ // keys this one carries a RESOLUTION and a FACE ROUTE, because Seedance's edit
1429
+ // price moves with both. Duration is the CEILED source-clip length: the edit
1430
+ // task type forces `duration: -1` on the wire, so the source clip is the only
1431
+ // honest quote.
1432
+ export function seedanceEditCostKey(input) {
1433
+ const res = input.videoResolution ?? '720p';
1434
+ const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
1435
+ return `seedance-2.5-edit${face}-${res}-${input.duration}s`;
1436
+ }
1437
+ // ── Audio ───────────────────────────────────────────────────────
1438
+ // Exported: the exact `model` ids slates_generate_audio accepts — consumed
1439
+ // by the desktop Studio Agent system prompt (SSOT; never restate these ids
1440
+ // in prose that can drift). Mirrors VIDEO_MODELS for the third media type.
1441
+ export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx'];
1442
+ /**
1443
+ * Per-surface bounds and defaults.
1444
+ *
1445
+ * 🚨 THESE FOUR NUMBERS PER SURFACE LIVE IN THREE REPOS. A change is a
1446
+ * three-site edit, every time:
1447
+ * 1. HERE (`audioCostKey`, the agent's pre-flight quote)
1448
+ * 2. `slate/src/shared/pricing.ts` → MODEL_REGISTRY `audio.durationSeconds`
1449
+ * (min/max/default), read by `clampAudioDuration` + `audioCreditKey`
1450
+ * 3. `slates-api/src/lib/audio-keys.ts` → the server's fail-closed bounds
1451
+ * `slates-api/scripts/pricing-consistency-check.mjs` §4 asserts 1 and 2 agree
1452
+ * at EVERY value including out-of-range ones; the gate check covers 3.
1453
+ *
1454
+ * The MINs used to be missing here, and the clamp floor was a hardcoded 1. That
1455
+ * made `slates_estimate_generation_cost({model:'seed-audio', duration:2})`
1456
+ * quote a real `seed-audio-2s` price for a generation the desktop would bill as
1457
+ * 3s and the proxy would REJECT outright. Same for an omitted duration, which
1458
+ * quoted `seed-audio-1s` against the desktop's `seed-audio-15s`.
1459
+ */
1460
+ export const SEED_AUDIO_MIN_SECONDS = 3;
1461
+ export const SEED_AUDIO_MAX_SECONDS = 120;
1462
+ export const SEED_AUDIO_DEFAULT_SECONDS = 15;
1463
+ export const ELEVEN_SFX_MIN_SECONDS = 1;
1464
+ export const ELEVEN_SFX_MAX_SECONDS = 22;
1465
+ export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
1466
+ /**
1467
+ * Byte-for-byte the desktop's `clampAudioDuration` in slate/src/shared/pricing.ts,
1468
+ * INCLUDING the non-finite arm — that one matters: `Math.max(min, NaN)` is NaN,
1469
+ * so without it a NaN duration produces the key `seed-audio-NaNs` here while the
1470
+ * desktop quotes the default. Divergence at a value neither side can bill is
1471
+ * still divergence; the checker sweeps for it.
1472
+ */
1473
+ function clampAudioSeconds(value, min, max, fallback) {
1474
+ if (!Number.isFinite(value))
1475
+ return fallback;
1476
+ return Math.min(max, Math.max(min, Math.ceil(value)));
1477
+ }
1478
+ function clampInt(value, min, max) {
1479
+ return Math.min(max, Math.max(min, Math.ceil(value)));
1480
+ }
1481
+ // Exported for scripts/pricing-consistency-check.mjs (slates-api repo), which
1482
+ // asserts this builder byte-matches the desktop's audioCreditKey().
1483
+ //
1484
+ // Seed Audio has NO duration parameter — length is driven by the prompt text.
1485
+ // We bill the REQUESTED duration (shape B, locked 2026-07-31): the desktop
1486
+ // injects "... N seconds" into the prompt and bills seed-audio-{N}s, so
1487
+ // display == billing with no amendment to the pricing law. The server probes
1488
+ // the returned audio.duration afterwards and logs SEED AUDIO BILLING DRIFT.
1489
+ export function audioCostKey(input) {
1490
+ if (input.model === 'seed-audio') {
1491
+ // `?? default` before the clamp, not `?? 0` — the desktop resolves a missing
1492
+ // duration to the registry DEFAULT, and a quote that doesn't match what the
1493
+ // desktop would bill is the whole bug class this mirrors away.
1494
+ const secs = clampAudioSeconds(input.durationSeconds ?? SEED_AUDIO_DEFAULT_SECONDS, SEED_AUDIO_MIN_SECONDS, SEED_AUDIO_MAX_SECONDS, SEED_AUDIO_DEFAULT_SECONDS);
1495
+ return `seed-audio-${secs}s`;
1496
+ }
1497
+ if (input.model === 'eleven-sfx') {
1498
+ const secs = clampAudioSeconds(input.durationSeconds ?? ELEVEN_SFX_DEFAULT_SECONDS, ELEVEN_SFX_MIN_SECONDS, ELEVEN_SFX_MAX_SECONDS, ELEVEN_SFX_DEFAULT_SECONDS);
1499
+ return `eleven-sfx-${secs}s`;
1500
+ }
1501
+ throw new Error(`Unknown audio model: ${input.model}`);
1502
+ }
1282
1503
  /**
1283
1504
  * Forgiving model-id resolver. Agents routinely paste registry COST keys
1284
1505
  * ("kling-v3-standard-8s", "seedance-2-1080p-8s") into the `model` param —
@@ -1323,6 +1544,13 @@ function resolveVideoModel(raw) {
1323
1544
  'kling-v3.0-omni-pro': 'kling-v3.0-omni',
1324
1545
  'seedance-2.0': 'seedance-2',
1325
1546
  'seedance-2-0': 'seedance-2',
1547
+ // ⚠️ The 2.5 spellings must resolve to 2.5, and the BARE `seedance` must keep
1548
+ // resolving to 2.0 — 2.0 is the default video model and holds 1080p/4K, which
1549
+ // 2.5 does not have at all.
1550
+ 'seedance-2.5': 'seedance-2.5',
1551
+ 'seedance-25': 'seedance-2.5',
1552
+ 'seedance-2-5': 'seedance-2.5',
1553
+ 'seedance2.5': 'seedance-2.5',
1326
1554
  seedance: 'seedance-2',
1327
1555
  'veo-3.1': 'veo-3.1-fast',
1328
1556
  'veo-3': 'veo-3.1-fast',
@@ -1345,6 +1573,11 @@ function promptingSkillFor(model) {
1345
1573
  return 'slates-prompting-kling-v3';
1346
1574
  if (model.startsWith('veo'))
1347
1575
  return 'slates-prompting-veo-3';
1576
+ // 2.5 BEFORE the generic seedance test — "seedance-2.5" also starts with
1577
+ // "seedance", and falling through hands 2.0's guide to a model with different
1578
+ // limits, a different resolution ladder and an extra task type.
1579
+ if (model.startsWith('seedance-2.5'))
1580
+ return 'slates-prompting-seedance-2-5';
1348
1581
  if (model.startsWith('seedance'))
1349
1582
  return 'slates-prompting-seedance';
1350
1583
  if (model.startsWith('omni-flash'))
@@ -1353,23 +1586,30 @@ function promptingSkillFor(model) {
1353
1586
  }
1354
1587
  export const generateVideo = {
1355
1588
  id: 'slates_generate_video',
1356
- description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
1589
+ description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
1357
1590
  input: z.object({
1358
1591
  prompt: z.string().min(1).max(4000),
1359
- model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
1592
+ model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | seedance-2.5 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (4-30s takes, 30 image refs, audio-only refs — but 480p/720p ONLY, so stay on seedance-2 whenever resolution matters), Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
1360
1593
  projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
1361
1594
  aspectRatio: z.enum(['1:1', '16:9', '9:16', '4:3', '3:4', '21:9', '9:21', '4:5', '5:4', '2:3', '3:2']).optional().describe('Veo locks to 16:9 — passing anything else will be ignored or fail. Kling/Seedance support all.'),
1362
- duration: z.number().int().min(3).max(15).optional().describe('Seconds. Kling: 5-15. Veo: 4, 6, or 8 only (4K only at 8s). Seedance: 4-15. Omni Flash: 3-10. Default 5 if omitted but always be explicit (cost scales linearly).'),
1363
- videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive). Veo: 720p/1080p same price, 4K more (8s only).'),
1595
+ duration: z.number().int().min(3).max(30).optional().describe('Seconds. Kling: 5-15. Veo: 4, 6, or 8 only (4K only at 8s). Seedance 2: 4-15. Seedance 2.5: 4-30 (the only model that reaches 30). Omni Flash: 3-10. Default 5 if omitted but always be explicit — cost scales linearly, and a 30s seedance-2.5 take is several hundred credits.'),
1596
+ videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance 2: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive, and Pro-only). Seedance 2.5: 480p/720p ONLY (default 720p) — it has no 1080p and no 4K on any provider, so asking for one is rejected, not downgraded. Veo: 720p/1080p same price, 4K more (8s only).'),
1364
1597
  firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video: asset UUID or badge code ("IMG-A8") — codes resolve against the project at call time, so a code the user just spoke is always safe to pass.'),
1365
1598
  lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code). Veo and Seedance only. Pairs with firstFrameAssetId for guided transitions.'),
1366
- ingredientAssetIds: z.array(z.string()).max(9).optional().describe('Visual reference / ingredient assets (UUIDs or badge codes) for Kling Omni, Seedance, or Omni Flash. Up to 9 (Seedance), 4 (Kling), or 7 (Omni Flash, combined across all ref params).'),
1599
+ ingredientAssetIds: z.array(z.string()).max(30).optional().describe('Visual reference / ingredient assets (UUIDs or badge codes) for Kling Omni, Seedance, or Omni Flash. Up to 30 (Seedance 2.5), 9 (Seedance 2), 4 (Kling), or 7 (Omni Flash, combined across all ref params). More is not better: 2-4 strong references beat both extremes, and past 4 reference PEOPLE output stability drops on Seedance regardless of the cap.'),
1367
1600
  characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets (UUIDs or badge codes) — keeps a character consistent across the shot.'),
1368
1601
  environmentAssetIds: z.array(z.string()).optional().describe('Environment reference assets (UUIDs or badge codes) — keeps a location/setting consistent across the shot.'),
1369
1602
  styleAssetIds: z.array(z.string()).optional().describe('Style reference assets (UUIDs or badge codes) — locks the visual style of the shot.'),
1370
- videoReferenceAssetId: z.string().optional().describe('Seedance ONLY: an existing VIDEO asset (UUID or badge code) to use as a reference — edit/relocate a clip, or MOTION TRANSFER (pair with a subject in ingredientAssetIds and a prompt like "the character from image 1 performs the motion from video 1"). 2-15s. Billing switches to input+output seconds (the vref key) — pass videoReferenceSeconds so the quote is right. If the clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Ignored by Kling/Veo.'),
1371
- videoReferenceSeconds: z.number().optional().describe('REQUIRED with videoReferenceAssetId: the reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — a video-reference gen bills combined input+output seconds; the server re-derives this by probing the clip, so an understated value just gets corrected upward.'),
1372
- audioReferenceAssetId: z.string().optional().describe('Seedance ONLY: an AUDIO asset (UUID or badge code), ≤15s, used as a reference — e.g. lip-sync a character to this audio ("the character in image 1 speaks the dialogue from audio 1"). No billing surcharge (Seedance audio is included). Requires at least one image or video reference alongside.'),
1603
+ videoReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into videoReferenceAssetIds; prefer that for anything new. A single VIDEO asset (UUID or badge code) used as a reference. Kept working forever: installed CLI and MCP builds send this shape.'),
1604
+ videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach. Required with videoReferenceAssetId: that clip\'s duration in seconds.'),
1605
+ audioReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into audioReferenceAssetIds; prefer that. A single AUDIO asset (UUID or badge code) used as a reference.'),
1606
+ // ── Multimodal references, the plural surface ──
1607
+ // The capacity sentences are DERIVED from MODEL_FACTS (see
1608
+ // multimodalRefSummary) rather than hand-typed, so a cap change in one
1609
+ // place cannot leave a stale number in a description an LLM reads.
1610
+ videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS (UUIDs or badge codes) read alongside the images and audio in the same generation — own-footage restyle, MOTION TRANSFER ("the character from image 1 performs the motion from video 1"), or dialogue conditioning. Cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; ignored elsewhere. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to combined input+output seconds (the vref key) — pass videoReferenceSecondsEach so the quote is right. If any clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Over the cap is REFUSED, never trimmed: a dropped clip would already have been priced in.`),
1611
+ videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — the bill is Σceil(each) + output seconds. The server re-derives this by probing every uploaded clip, so an understated value just gets corrected upward.'),
1612
+ audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO clips (UUIDs or badge codes) read alongside the images and video — e.g. lip-sync a character to a line ("the character in image 1 speaks the dialogue from audio 1"). Cited as "audio 1", "audio 2"… in the order given. No billing surcharge (Seedance audio is included). ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
1373
1613
  sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
1374
1614
  audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
1375
1615
  generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
@@ -1438,7 +1678,10 @@ export const generateVideo = {
1438
1678
  message: `Omni Flash supports 3-10 seconds (you passed ${input.duration}s). Pick a duration in that range.`,
1439
1679
  });
1440
1680
  }
1441
- if (input.lastFrameAssetId || input.videoReferenceAssetId || input.audioReferenceAssetId) {
1681
+ if (input.lastFrameAssetId ||
1682
+ input.videoReferenceAssetId || input.audioReferenceAssetId ||
1683
+ (input.videoReferenceAssetIds?.length ?? 0) > 0 ||
1684
+ (input.audioReferenceAssetIds?.length ?? 0) > 0) {
1442
1685
  return ok({
1443
1686
  requires_clarification: true,
1444
1687
  missing: [],
@@ -1499,6 +1742,10 @@ export const generateVideo = {
1499
1742
  refInputs.push({ ref: input.videoReferenceAssetId, role: 'video reference' });
1500
1743
  if (input.audioReferenceAssetId)
1501
1744
  refInputs.push({ ref: input.audioReferenceAssetId, role: 'audio reference' });
1745
+ for (const r of input.videoReferenceAssetIds ?? [])
1746
+ refInputs.push({ ref: r, role: 'video reference' });
1747
+ for (const r of input.audioReferenceAssetIds ?? [])
1748
+ refInputs.push({ ref: r, role: 'audio reference' });
1502
1749
  const resolvedRefs = await resolveAssetRefs(ctx, input.projectId, refInputs.map((r) => r.ref));
1503
1750
  const rid = (v) => v ? (resolvedRefs.get(v)?.id ?? v) : v;
1504
1751
  const rids = (a) => a?.map((v) => resolvedRefs.get(v)?.id ?? v);
@@ -1506,6 +1753,8 @@ export const generateVideo = {
1506
1753
  input.lastFrameAssetId = rid(input.lastFrameAssetId);
1507
1754
  input.videoReferenceAssetId = rid(input.videoReferenceAssetId);
1508
1755
  input.audioReferenceAssetId = rid(input.audioReferenceAssetId);
1756
+ input.videoReferenceAssetIds = rids(input.videoReferenceAssetIds);
1757
+ input.audioReferenceAssetIds = rids(input.audioReferenceAssetIds);
1509
1758
  input.ingredientAssetIds = rids(input.ingredientAssetIds);
1510
1759
  input.characterAssetIds = rids(input.characterAssetIds);
1511
1760
  input.environmentAssetIds = rids(input.environmentAssetIds);
@@ -1514,13 +1763,59 @@ export const generateVideo = {
1514
1763
  // A video reference bills combined input+output seconds (the vref key) —
1515
1764
  // the quote needs the clip's length. The server probes the uploaded ref
1516
1765
  // and corrects the key anyway, so this only gates quote accuracy.
1517
- if (input.videoReferenceAssetId && input.model.startsWith('seedance') && !input.videoReferenceSeconds) {
1766
+ // 🚨 Seedance 2.5 PROMPT-INTENT PRE-FLIGHT. With references attached, the
1767
+ // provider sorts a request into one of five task types from the roles PLUS
1768
+ // the prompt's intent, then fails a mismatch ASYNCHRONOUSLY — after the task
1769
+ // queues and credits are reserved. Catch it before the spend, NAME the words,
1770
+ // and never rewrite the prompt (the prompt-transparency invariant).
1771
+ if (input.model === 'seedance-2.5') {
1772
+ // 🚨 EVERY reference shape counts, singular AND plural. The classifier
1773
+ // engages on the presence of reference ROLES, and the plural arrays are
1774
+ // the surface new callers use — listing only the deprecated singular ones
1775
+ // meant the modern call path got no warning at all, which is precisely
1776
+ // backwards.
1777
+ const hasRefs = !!input.firstFrameAssetId || !!input.lastFrameAssetId ||
1778
+ (input.ingredientAssetIds?.length ?? 0) > 0 ||
1779
+ (input.characterAssetIds?.length ?? 0) > 0 ||
1780
+ (input.environmentAssetIds?.length ?? 0) > 0 ||
1781
+ (input.styleAssetIds?.length ?? 0) > 0 ||
1782
+ !!input.videoReferenceAssetId || !!input.audioReferenceAssetId ||
1783
+ (input.videoReferenceAssetIds?.length ?? 0) > 0 ||
1784
+ (input.audioReferenceAssetIds?.length ?? 0) > 0;
1785
+ // Trigger words are DERIVED from the model-facts SSOT, not retyped here —
1786
+ // a local literal had already lost 'continue the story'.
1787
+ const hits = hasRefs ? seedanceTaskIntentWords(input.prompt) : [];
1788
+ if (hits.length > 0 && !input.confirm) {
1789
+ return ok({
1790
+ requires_clarification: true,
1791
+ missing: ['prompt'],
1792
+ message: `Seedance 2.5 reads ${hits.map((w) => `"${w}"`).join(', ')} in this prompt as an EDIT or EXTEND instruction and may run it as a video edit, ` +
1793
+ `which fails after the job has queued. If you mean to edit an existing clip, call slates_edit_video with model "seedance-2.5-edit". ` +
1794
+ `If you mean a fresh shot, describe the finished frame instead of an instruction to change one ("the workshop bench, clear and uncluttered" rather than "remove the tripod"). ` +
1795
+ `Pass confirm=true to send it as written.`,
1796
+ });
1797
+ }
1798
+ }
1799
+ // The quote needs a length for EVERY reference clip, in both shapes. A
1800
+ // missing one doesn't fail the generation (the server probes and corrects
1801
+ // upward) — it silently under-quotes, which is worse than asking.
1802
+ const isSeedance = input.model.startsWith('seedance');
1803
+ if (input.videoReferenceAssetId && isSeedance && !input.videoReferenceSeconds) {
1518
1804
  return ok({
1519
1805
  requires_clarification: true,
1520
1806
  missing: ['videoReferenceSeconds'],
1521
1807
  message: 'A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSeconds (the reference clip\'s duration, shown in slates_list_assets) so the pre-flight quote matches the bill.',
1522
1808
  });
1523
1809
  }
1810
+ const pluralRefCount = input.videoReferenceAssetIds?.length ?? 0;
1811
+ if (pluralRefCount > 0 && isSeedance &&
1812
+ (input.videoReferenceSecondsEach?.length ?? 0) !== pluralRefCount) {
1813
+ return ok({
1814
+ requires_clarification: true,
1815
+ missing: ['videoReferenceSecondsEach'],
1816
+ message: `A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSecondsEach with exactly ${pluralRefCount} duration${pluralRefCount === 1 ? '' : 's'}, in the same order as videoReferenceAssetIds (durations are shown in slates_list_assets), so the pre-flight quote matches the bill.`,
1817
+ });
1818
+ }
1524
1819
  const cloud = ctx.cloud();
1525
1820
  const registry = await cloud.get('/api/agent/models');
1526
1821
  const costKey = videoCostKey({
@@ -1530,7 +1825,13 @@ export const generateVideo = {
1530
1825
  sound: input.sound,
1531
1826
  seedanceFace: input.seedanceFace,
1532
1827
  seedanceRealFace: input.seedanceRealFace,
1533
- videoRefSeconds: input.videoReferenceAssetId ? input.videoReferenceSeconds : 0,
1828
+ // Σ ceil(d - 0.05) over every reference clip, both shapes — the same
1829
+ // expression the desktop's estimateCost and the handler's key builder
1830
+ // use. Quoting only the singular would understate a multi-clip call.
1831
+ videoRefSeconds: (input.videoReferenceAssetId && input.videoReferenceSeconds
1832
+ ? Math.ceil(input.videoReferenceSeconds - 0.05)
1833
+ : 0) +
1834
+ (input.videoReferenceSecondsEach ?? []).reduce((n, d) => n + (d > 0 ? Math.ceil(d - 0.05) : 0), 0),
1534
1835
  });
1535
1836
  // Hard consent gate, checked before any spend: the real-face route is
1536
1837
  // consent-attested by design (the desktop enforces it too).
@@ -1624,8 +1925,13 @@ export const generateVideo = {
1624
1925
  characterAssetIds: input.characterAssetIds ?? [],
1625
1926
  environmentAssetIds: input.environmentAssetIds ?? [],
1626
1927
  styleAssetIds: input.styleAssetIds ?? [],
1928
+ // Both shapes on the wire. The route merges and dedupes them, so an old
1929
+ // client sending only the singular and a new one sending only the plural
1930
+ // reach the identical handler params.
1627
1931
  videoReferenceAssetId: input.videoReferenceAssetId,
1628
1932
  audioReferenceAssetId: input.audioReferenceAssetId,
1933
+ videoReferenceAssetIds: input.videoReferenceAssetIds,
1934
+ audioReferenceAssetIds: input.audioReferenceAssetIds,
1629
1935
  sound: input.sound,
1630
1936
  audioLanguage: input.audioLanguage,
1631
1937
  generateMusic: input.generateMusic,
@@ -1671,30 +1977,179 @@ export const generateVideo = {
1671
1977
  };
1672
1978
  },
1673
1979
  };
1980
+ // ── Generate audio ──────────────────────────────────────────────
1981
+ export const generateAudio = {
1982
+ id: 'slates_generate_audio',
1983
+ description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Two surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s) and eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop). Which surface for which job: read the slates-model-selection skill. ' +
1984
+ '🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
1985
+ 'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
1986
+ 'projectId is REQUIRED (no headless path). Cost > 17 credits returns requires_confirm — pass confirm=true after explicit user OK. No skill files installed? Call slates_get_prompting_guide first.',
1987
+ input: z.object({
1988
+ projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
1989
+ model: z
1990
+ .enum(AUDIO_MODELS)
1991
+ .describe('Audio surface. seed-audio = scene/ambience/beds/dialogue (default choice), eleven-sfx = one precise effect or a seamless loop. Routing doctrine: slates-model-selection skill.'),
1992
+ prompt: z
1993
+ .string()
1994
+ .min(1)
1995
+ .max(5000)
1996
+ .describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars.'),
1997
+ durationSeconds: z
1998
+ .number()
1999
+ .optional()
2000
+ .describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.'),
2001
+ voice: z
2002
+ .string()
2003
+ .optional()
2004
+ .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
2005
+ speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
2006
+ volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
2007
+ pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
2008
+ multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
2009
+ loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
2010
+ promptInfluence: z.number().min(0).max(1).optional().describe('eleven-sfx only — 0-1, default 0.3. Higher hugs your wording with less variation between takes.'),
2011
+ audioReferenceAssetIds: z
2012
+ .array(z.string())
2013
+ .max(3)
2014
+ .optional()
2015
+ .describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId — the API rejects both.'),
2016
+ imageReferenceAssetId: z
2017
+ .string()
2018
+ .optional()
2019
+ .describe('seed-audio only — ONE image asset to score what is in frame. MUTUALLY EXCLUSIVE with audioReferenceAssetIds.'),
2020
+ background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2021
+ confirm: z.boolean().optional().describe('Set true to bypass the confirm gate after explicit user OK.'),
2022
+ }),
2023
+ run: async (input, ctx) => {
2024
+ // ── Per-surface clarification + constraint gates ──
2025
+ const cfgDefaults = {
2026
+ 'seed-audio': SEED_AUDIO_DEFAULT_SECONDS,
2027
+ 'eleven-sfx': ELEVEN_SFX_DEFAULT_SECONDS,
2028
+ };
2029
+ const seconds = input.durationSeconds ?? cfgDefaults[input.model];
2030
+ if (input.model === 'seed-audio') {
2031
+ if (seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
2032
+ return ok({
2033
+ requires_clarification: true,
2034
+ missing: ['durationSeconds'],
2035
+ message: `Seed Audio needs a durationSeconds of ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS}. It has NO duration parameter — the number is written into the prompt AND is what the user is billed, so it must be a deliberate choice. Ask the user how long the bed should be (a few seconds longer than the clip it sits under, so the edit has handles).`,
2036
+ });
2037
+ }
2038
+ if ((input.audioReferenceAssetIds?.length ?? 0) > 0 && input.imageReferenceAssetId) {
2039
+ throw new Error('Seed Audio takes audio references OR one image reference, never both — the provider rejects the combination.');
2040
+ }
2041
+ }
2042
+ if (input.model === 'eleven-sfx') {
2043
+ if (seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
2044
+ return ok({
2045
+ requires_clarification: true,
2046
+ missing: ['durationSeconds'],
2047
+ message: `Sound Effects needs a durationSeconds of ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. It is billed per second and is never left for the model to pick (that would make the charge non-deterministic). Roughly: 0.5-1s for an impact, 2-4s for a whoosh, 8-22s for a loopable bed.`,
2048
+ });
2049
+ }
2050
+ if (input.prompt.length > 450) {
2051
+ throw new Error(`Sound Effects accepts up to 450 characters — this prompt is ${input.prompt.length}.`);
2052
+ }
2053
+ }
2054
+ await ctx.desktop().requireCapability('audio-generation', 'audio generation');
2055
+ // ── Resolve asset refs at CALL time (UUIDs or badge codes) ──
2056
+ const refInputs = [];
2057
+ for (const ref of input.audioReferenceAssetIds ?? [])
2058
+ refInputs.push({ ref, role: 'audio reference' });
2059
+ if (input.imageReferenceAssetId)
2060
+ refInputs.push({ ref: input.imageReferenceAssetId, role: 'image reference' });
2061
+ const resolvedRefs = refInputs.length > 0
2062
+ ? await resolveAssetRefs(ctx, input.projectId, refInputs.map((r) => r.ref))
2063
+ : new Map();
2064
+ const rid = (v) => (v ? (resolvedRefs.get(v)?.id ?? v) : v);
2065
+ const refEcho = refInputs.length > 0 ? describeResolvedRefs(refInputs, resolvedRefs) : '';
2066
+ // ── Cost — from the CLOUD registry, keyed by the SAME builder the desktop
2067
+ // and the proxy use. Never quote a price from memory. ──
2068
+ const cloud = ctx.cloud();
2069
+ const registry = await cloud.get('/api/agent/models');
2070
+ const costKey = audioCostKey({
2071
+ model: input.model,
2072
+ durationSeconds: seconds,
2073
+ });
2074
+ const entry = registry.models.find((m) => m.model === costKey);
2075
+ if (!entry) {
2076
+ throw new Error(`Audio variant not in registry: ${costKey}. Available audio models: ${AUDIO_MODELS.join(' | ')}.`);
2077
+ }
2078
+ const totalCents = creditCost(entry);
2079
+ if (!input.confirm && totalCents > CONFIRM_CREDITS) {
2080
+ return ok({
2081
+ requires_confirm: true,
2082
+ model: input.model,
2083
+ variant: costKey,
2084
+ cost_credits: totalCents,
2085
+ cost_display: fmtCredits(totalCents),
2086
+ message: `${input.model} audio will cost ${fmtCredits(totalCents)}. ` +
2087
+ (input.model === 'seed-audio'
2088
+ ? `You are billed for the ${seconds}s you requested regardless of the returned length. `
2089
+ : '') +
2090
+ 'Confirm with the user, then call again with confirm: true.',
2091
+ });
2092
+ }
2093
+ // ── Submit through the DESKTOP so the progress card, the project save,
2094
+ // and recovery all behave exactly like a UI-triggered run. ──
2095
+ const desktop = ctx.desktop();
2096
+ if (input.background) {
2097
+ await desktop.requireCapability('background-generation', 'background generation');
2098
+ }
2099
+ const result = await desktop.post('/agent/generation/audio', {
2100
+ projectId: input.projectId,
2101
+ model: input.model,
2102
+ prompt: input.prompt,
2103
+ durationSeconds: seconds,
2104
+ voice: input.voice,
2105
+ speed: input.speed,
2106
+ volume: input.volume,
2107
+ pitch: input.pitch,
2108
+ multilingual: input.multilingual,
2109
+ loop: input.loop,
2110
+ promptInfluence: input.promptInfluence,
2111
+ audioReferenceAssetIds: (input.audioReferenceAssetIds ?? []).map((r) => rid(r)),
2112
+ imageReferenceAssetId: rid(input.imageReferenceAssetId),
2113
+ background: input.background,
2114
+ });
2115
+ if (!result.success)
2116
+ throw new Error(result.error ?? 'Audio generation failed');
2117
+ if (result.background) {
2118
+ return backgroundSubmitted(`${input.model} audio generation`, [result.generationId].filter(Boolean), { model: input.model, variant: costKey, projectId: input.projectId, cost_credits: totalCents }, refEcho);
2119
+ }
2120
+ return {
2121
+ text: `Generated ${input.model} audio into project ${input.projectId} for ${fmtCredits(totalCents)}` +
2122
+ `. Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
2123
+ (refEcho ? ` ${refEcho}` : ''),
2124
+ data: {
2125
+ model: input.model,
2126
+ variant: costKey,
2127
+ projectId: input.projectId,
2128
+ durationSeconds: seconds,
2129
+ cost_cents: totalCents,
2130
+ cost_credits: totalCents,
2131
+ asset: result.asset,
2132
+ generationId: result.generationId,
2133
+ },
2134
+ };
2135
+ },
2136
+ };
1674
2137
  // ── Generate lip-sync ───────────────────────────────────────────
1675
2138
  export const generateLipSync = {
1676
2139
  id: 'slates_generate_lip_sync',
1677
- description: 'Lip-sync a still image (avatar) or a video clip to audio. Two engines — route per slates-model-selection: (1) engine=kling (default, cheap utility lane): sourceType=video re-syncs a clip (~$0.11 / 5s); sourceType=image animates a still avatar (avatar-standard ~$0.42 / 5s; avatar-pro ~$0.86 / 5s). Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. (2) engine=seedance-2 (premium single-pass): the speech is generated IN the video itself — natural delivery, a video source keeps its OWN voice (native voice clone), audio included; ttsText becomes the spoken line (no voice/speed params), or an uploaded ≤15s audio file drives the speech as a reference. Seedance sources must be 2-15s videos or images; bills seedance keys (video sources bill input+output seconds — pass sourceSeconds). Faces route via seedanceFace (default true) / seedanceRealFace+realFaceConsent for real people. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
2140
+ description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip (~$0.11 / 5s); sourceType=image animates a still avatar (avatar-standard ~$0.42 / 5s; avatar-pro ~$0.86 / 5s). Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
1678
2141
  input: z.object({
1679
2142
  projectId: z.string().uuid().describe('Slates project the source asset lives in. The new lip-synced video lands here.'),
1680
2143
  sourceAssetId: z.string().uuid().describe('Asset id of the still image (avatar flow) or video clip (lip-sync flow). Must already exist in the project — use slates_upload_reference_image or slates_generate_image / slates_generate_video first if needed.'),
1681
2144
  sourceType: z.enum(['image', 'video']).describe('"image" = animate a still portrait (avatar). "video" = re-sync an existing talking-head clip. Determines pricing — be deliberate.'),
1682
- audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText (on Seedance the line is spoken natively in the generation). "upload" = use the file at audioFilePath (absolute path on the user\'s machine; ≤15s on Seedance).'),
2145
+ audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText. "upload" = use the file at audioFilePath (absolute path on the user\'s machine).'),
1683
2146
  ttsText: z.string().min(1).max(2000).optional().describe('Required when audioMethod=tts. The exact words the avatar/clip will speak.'),
1684
- ttsVoice: z.string().optional().describe('Kling engine only — voice id (e.g. "oversea_male1"). See slates-prompting-lip-sync skill for the voice catalog. Ignored on Seedance (a video source keeps its own voice; otherwise describe the voice in ttsText context).'),
1685
- ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling engine only — TTS language. Default EN.'),
1686
- ttsSpeed: z.number().min(0.5).max(2).optional().describe('Kling engine only — TTS speech rate. Default 1.0. Range 0.5-2.0.'),
2147
+ ttsVoice: z.string().optional().describe('Voice id (e.g. "oversea_male1"). See slates-prompting-lip-sync skill for the voice catalog.'),
2148
+ ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('TTS language. Default EN.'),
2149
+ ttsSpeed: z.number().min(0.5).max(2).optional().describe('TTS speech rate. Default 1.0. Range 0.5-2.0.'),
1687
2150
  audioFilePath: z.string().optional().describe('Required when audioMethod=upload. Absolute path to the audio file on the user\'s machine (mp3, wav, m4a).'),
1688
- avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('Kling engine, image-source only. avatar-standard (~14 credits/5s) for general use. avatar-pro (~29 credits/5s) for sharper face fidelity.'),
1689
- klingProvider: z.enum(['fal', 'kling']).optional().describe('Kling engine only — provider routing. Leave unset: all agent generations bill Slates credits (BYOK is retired).'),
1690
- engine: z.enum(['kling', 'seedance-2']).optional().describe('Default kling (cheap utility). seedance-2 = premium single-pass: natural speech generated in the video, voice cloned from a video source, audio included. Credits only.'),
1691
- videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
1692
- aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
1693
- seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the source (default TRUE for lip-sync; the faceless route would reject it). Bills the -face key.'),
1694
- seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the source shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
1695
- realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
1696
- sourceSeconds: z.number().optional().describe('Seedance engine + sourceType=video: the source clip\'s duration in seconds (from the asset listing). Feeds the vref cost key (input+output billing).'),
1697
- audioSeconds: z.number().optional().describe('Seedance engine + audioMethod=upload: the audio file\'s duration in seconds — sets the output length (4-15s).'),
2151
+ avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('Image-source only. avatar-standard (~14 credits/5s) for general use. avatar-pro (~29 credits/5s) for sharper face fidelity.'),
2152
+ klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. Leave unset: all agent generations bill Slates credits (BYOK is retired).'),
1698
2153
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
1699
2154
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required for avatar-pro.'),
1700
2155
  }),
@@ -1713,41 +2168,8 @@ export const generateLipSync = {
1713
2168
  message: 'audioMethod=upload requires audioFilePath. Pass an absolute path to the audio file on the user\'s machine.',
1714
2169
  });
1715
2170
  }
1716
- const isSeedance = input.engine === 'seedance-2';
1717
2171
  let costKey;
1718
- let seedanceDuration = 0;
1719
- if (isSeedance) {
1720
- // Consent gate before any spend, mirroring slates_generate_video.
1721
- if (input.seedanceRealFace && !input.realFaceConsent) {
1722
- return ok({
1723
- requires_clarification: true,
1724
- missing: ['realFaceConsent'],
1725
- message: 'Real-person lip-sync needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
1726
- });
1727
- }
1728
- if (input.sourceType === 'video' && !input.sourceSeconds) {
1729
- return ok({
1730
- requires_clarification: true,
1731
- missing: ['sourceSeconds'],
1732
- message: 'Seedance lip-sync on a video source bills combined input+output seconds. Pass sourceSeconds (the clip\'s duration from slates_list_assets, must be 2-15s).',
1733
- });
1734
- }
1735
- const clamp = (n) => Math.min(15, Math.max(4, Math.ceil(n)));
1736
- seedanceDuration = clamp(input.sourceType === 'video' && input.sourceSeconds
1737
- ? input.sourceSeconds
1738
- : input.audioSeconds ?? (input.ttsText ? input.ttsText.length / 13 : 5));
1739
- costKey = videoCostKey({
1740
- model: 'seedance-2',
1741
- duration: seedanceDuration,
1742
- videoResolution: input.videoResolution ?? '1080p',
1743
- // Lip-sync sources are faces by definition — face route unless
1744
- // explicitly disabled or escalated to realface.
1745
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
1746
- seedanceRealFace: input.seedanceRealFace === true,
1747
- videoRefSeconds: input.sourceType === 'video' ? input.sourceSeconds ?? 0 : 0,
1748
- });
1749
- }
1750
- else if (input.sourceType === 'video') {
2172
+ if (input.sourceType === 'video') {
1751
2173
  costKey = 'kling-lip-sync-video-5s';
1752
2174
  }
1753
2175
  else {
@@ -1776,7 +2198,7 @@ export const generateLipSync = {
1776
2198
  estimated_cents: totalCents,
1777
2199
  estimated_credits: totalCents,
1778
2200
  source_ref: sourceRef,
1779
- message: `Cost: ${fmtCredits(totalCents)} for ${isSeedance ? `${seedanceDuration}s Seedance` : '5s'} lip-sync (${costKey}). ` +
2201
+ message: `Cost: ${fmtCredits(totalCents)} for 5s lip-sync (${costKey}). ` +
1780
2202
  `Source: ${sourceRef}. ${audioPreview}. ` +
1781
2203
  `Re-call with confirm=true after the user explicitly OKs the spend. ` +
1782
2204
  `When discussing with the user, refer to the source by its code (matches the gallery badge).`,
@@ -1799,28 +2221,12 @@ export const generateLipSync = {
1799
2221
  klingProvider: input.klingProvider,
1800
2222
  estimatedCost: totalCents,
1801
2223
  background: input.background,
1802
- // Seedance engine passthrough — the desktop delegates to the seedance
1803
- // ref-to-video path (vref billing, face cascade, consent gate). The
1804
- // durations ride along so the desktop bills exactly what was quoted.
1805
- ...(isSeedance
1806
- ? {
1807
- lipSyncEngine: 'seedance-2',
1808
- duration: seedanceDuration,
1809
- videoResolution: input.videoResolution,
1810
- aspectRatio: input.aspectRatio,
1811
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
1812
- seedanceRealFace: input.seedanceRealFace === true,
1813
- realFaceConsent: input.realFaceConsent === true,
1814
- sourceDurationSeconds: input.sourceSeconds,
1815
- audioDurationSeconds: input.audioSeconds,
1816
- }
1817
- : {}),
1818
2224
  });
1819
2225
  if (!result.success)
1820
2226
  throw new Error(result.error ?? 'Lip-sync generation failed');
1821
2227
  if (result.background) {
1822
2228
  const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
1823
- return backgroundSubmitted(`${isSeedance ? `${seedanceDuration}s` : '5s'} lip-sync (${costKey})`, ids, {
2229
+ return backgroundSubmitted(`5s lip-sync (${costKey})`, ids, {
1824
2230
  variant: costKey,
1825
2231
  projectId: input.projectId,
1826
2232
  sourceAssetId: input.sourceAssetId,
@@ -1829,7 +2235,7 @@ export const generateLipSync = {
1829
2235
  });
1830
2236
  }
1831
2237
  return {
1832
- text: `Generated ${isSeedance ? `${seedanceDuration}s` : '5s'} lip-sync (${costKey}) into project ${input.projectId} ` +
2238
+ text: `Generated 5s lip-sync (${costKey}) into project ${input.projectId} ` +
1833
2239
  `for ${fmtCredits(totalCents)}. ` +
1834
2240
  (input.audioMethod === 'tts'
1835
2241
  ? `Spoken: "${(input.ttsText ?? '').slice(0, 60)}${(input.ttsText ?? '').length > 60 ? '...' : ''}"`
@@ -1850,58 +2256,21 @@ export const generateLipSync = {
1850
2256
  // ── Generate motion transfer ────────────────────────────────────
1851
2257
  export const generateMotionTransfer = {
1852
2258
  id: 'slates_generate_motion_transfer',
1853
- description: 'Transfer the motion from a reference video onto a target image character. Two engines — route per slates-model-selection: (1) kling-mc-std ($0.95 / 5s) / kling-mc-pro ($1.26 / 5s) — the cheap utility lane, structured skeleton/depth retargeting, always 5s. (2) motionModel=seedance-2 — the PREMIUM lane: single-pass generation with the driving clip as a native conditioning signal (better motion fidelity + native audio), prompt-driven (write what the character does, e.g. "the character from image 1 performs the exact motion from video 1"), bills input+output seconds on seedance vref keys (pass sourceVideoSeconds; driving clip must be 2-15s). Faces: seedanceFace defaults true; a REAL person needs seedanceRealFace+realFaceConsent (premium route). REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. All tiers hit the >$0.50 confirm gate.',
2259
+ description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std ($0.95 / 5s) or kling-mc-pro ($1.26 / 5s), structured skeleton/depth retargeting, always 5s. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. Both tiers hit the >$0.50 confirm gate.',
1854
2260
  input: z.object({
1855
2261
  projectId: z.string().uuid().describe('Slates project. Both source and target assets must live here.'),
1856
- sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Seedance engine: 2-15s clips only.'),
2262
+ sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s.'),
1857
2263
  targetImageAssetId: z.string().uuid().describe('Asset id of the target image (the character that will perform the motion). Must already exist in the project.'),
1858
- motionModel: z.enum(['kling-mc-std', 'kling-mc-pro', 'seedance-2']).optional().describe('kling-mc-std (~32 credits) general motion; kling-mc-pro (~42 credits) cleaner anatomy — default. seedance-2 = premium single-pass lane (prompt-driven, native audio, input+output-second billing) — pick when motion fidelity or audio matters.'),
1859
- characterOrientation: z.enum(['video', 'image']).optional().describe('Kling only. "video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
1860
- prompt: z.string().optional().describe('Kling: optional refinement. Seedance: THE driver — describe what the character does with the motion from the clip (ordinal references: "the character from image 1 performs the motion from video 1"). A sensible default recipe is used if omitted. Read slates-prompting-motion-transfer.'),
1861
- klingProvider: z.enum(['fal', 'kling']).optional().describe('Kling engine only — provider routing. "fal" (default) uses Slates credits.'),
1862
- duration: z.number().int().min(4).max(15).optional().describe('Seedance engine only — output duration in seconds (4-15). Defaults to the driving clip\'s length.'),
1863
- videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
1864
- aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
1865
- seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the clip/image (default TRUE for motion transfer). Bills the -face key.'),
1866
- seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the driving clip/subject shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
1867
- realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
1868
- sourceVideoSeconds: z.number().optional().describe('Seedance engine: the driving clip\'s duration in seconds (from the asset listing, 2-15s). Feeds the vref cost key (input+output billing).'),
2264
+ motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std (~32 credits) general motion; kling-mc-pro (~42 credits) cleaner anatomy — default.'),
2265
+ characterOrientation: z.enum(['video', 'image']).optional().describe('"video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
2266
+ prompt: z.string().optional().describe('Optional refinement. Read slates-prompting-motion-transfer.'),
2267
+ klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. "fal" (default) uses Slates credits.'),
1869
2268
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
1870
2269
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required — both tiers exceed.'),
1871
2270
  }),
1872
2271
  async run(input, ctx) {
1873
2272
  const motionModel = input.motionModel ?? 'kling-mc-pro';
1874
- const isSeedance = motionModel === 'seedance-2';
1875
- let costKey;
1876
- let seedanceDuration = 0;
1877
- if (isSeedance) {
1878
- if (input.seedanceRealFace && !input.realFaceConsent) {
1879
- return ok({
1880
- requires_clarification: true,
1881
- missing: ['realFaceConsent'],
1882
- message: 'Real-person motion transfer needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
1883
- });
1884
- }
1885
- if (!input.sourceVideoSeconds) {
1886
- return ok({
1887
- requires_clarification: true,
1888
- missing: ['sourceVideoSeconds'],
1889
- message: 'Seedance motion transfer bills combined input+output seconds. Pass sourceVideoSeconds (the driving clip\'s duration from slates_list_assets, must be 2-15s).',
1890
- });
1891
- }
1892
- seedanceDuration = input.duration ?? Math.min(15, Math.max(4, Math.ceil(input.sourceVideoSeconds)));
1893
- costKey = videoCostKey({
1894
- model: 'seedance-2',
1895
- duration: seedanceDuration,
1896
- videoResolution: input.videoResolution ?? '1080p',
1897
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
1898
- seedanceRealFace: input.seedanceRealFace === true,
1899
- videoRefSeconds: input.sourceVideoSeconds,
1900
- });
1901
- }
1902
- else {
1903
- costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
1904
- }
2273
+ const costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
1905
2274
  const cloud = ctx.cloud();
1906
2275
  const registry = await cloud.get('/api/agent/models');
1907
2276
  const entry = registry.models.find((m) => m.model === costKey);
@@ -1925,9 +2294,9 @@ export const generateMotionTransfer = {
1925
2294
  estimated_credits: totalCents,
1926
2295
  source_ref: source,
1927
2296
  target_ref: target,
1928
- message: `Cost: ${fmtCredits(totalCents)} for ${isSeedance ? `${seedanceDuration}s Seedance motion transfer` : `5s ${motionModel}`} (${costKey}). ` +
2297
+ message: `Cost: ${fmtCredits(totalCents)} for 5s ${motionModel} (${costKey}). ` +
1929
2298
  `Transferring motion from ${source} onto ${target}. ` +
1930
- `Re-call with confirm=true after the user explicitly OKs the spend${isSeedance ? '' : ', or pick kling-mc-std to save ~10 credits'}. ` +
2299
+ `Re-call with confirm=true after the user explicitly OKs the spend, or pick kling-mc-std to save ~10 credits. ` +
1931
2300
  `When discussing with the user, refer to the assets by those codes — they'll match the gallery badges.`,
1932
2301
  });
1933
2302
  }
@@ -1945,27 +2314,12 @@ export const generateMotionTransfer = {
1945
2314
  klingProvider: input.klingProvider,
1946
2315
  estimatedCost: totalCents,
1947
2316
  background: input.background,
1948
- // Seedance engine passthrough — the desktop delegates to the seedance
1949
- // ref-to-video path (vref billing, face cascade, consent gate).
1950
- ...(isSeedance
1951
- ? {
1952
- duration: seedanceDuration,
1953
- videoResolution: input.videoResolution,
1954
- aspectRatio: input.aspectRatio,
1955
- seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
1956
- seedanceRealFace: input.seedanceRealFace === true,
1957
- realFaceConsent: input.realFaceConsent === true,
1958
- // Ride the caller-supplied clip duration through — the asset row's
1959
- // duration can be null for imported clips.
1960
- sourceVideoDurationSeconds: input.sourceVideoSeconds,
1961
- }
1962
- : {}),
1963
2317
  });
1964
2318
  if (!result.success)
1965
2319
  throw new Error(result.error ?? 'Motion transfer generation failed');
1966
2320
  if (result.background) {
1967
2321
  const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
1968
- return backgroundSubmitted(`${isSeedance ? `${seedanceDuration}s` : '5s'} motion transfer (${motionModel})`, ids, {
2322
+ return backgroundSubmitted(`5s motion transfer (${motionModel})`, ids, {
1969
2323
  variant: costKey,
1970
2324
  motionModel,
1971
2325
  projectId: input.projectId,
@@ -1976,7 +2330,7 @@ export const generateMotionTransfer = {
1976
2330
  });
1977
2331
  }
1978
2332
  return {
1979
- text: `Generated ${isSeedance ? `${seedanceDuration}s` : '5s'} motion transfer (${motionModel}) into project ${input.projectId} ` +
2333
+ text: `Generated 5s motion transfer (${motionModel}) into project ${input.projectId} ` +
1980
2334
  `for ${fmtCredits(totalCents)}.` +
1981
2335
  (input.prompt ? ` Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` : ''),
1982
2336
  data: {
@@ -1996,15 +2350,17 @@ export const generateMotionTransfer = {
1996
2350
  // ── Edit video (Kling O3 video-to-video) ────────────────────────
1997
2351
  export const editVideo = {
1998
2352
  id: 'slates_edit_video',
1999
- description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements) or omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
2353
+ description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat), or seedance-2.5-edit (4–30s clips — the ONLY engine that takes a clip over 15s; 480p/720p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds — always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
2000
2354
  input: z.object({
2001
2355
  projectId: z.string().uuid().describe('Project the source clip lives in.'),
2002
2356
  sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
2003
2357
  prompt: z.string().min(1).max(2500).describe('The change, not the whole scene — e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
2004
- model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs.'),
2358
+ model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit', 'seedance-2.5-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs. seedance-2.5-edit (480p/720p, 4–30s) is the ONLY engine that accepts a clip longer than 15s; it also takes NO character/style refs on this op.'),
2005
2359
  characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY — rejected on omni-flash-edit.'),
2006
2360
  styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
2007
- keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit output carries its own audio).'),
2361
+ keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
2362
+ videoResolution: z.enum(['480p', '720p']).optional().describe('seedance-2.5-edit ONLY (default 720p). Ignored by the Kling and Omni Flash engines, whose output follows the source clip.'),
2363
+ seedanceFace: z.boolean().optional().describe('seedance-2.5-edit ONLY: set true when a CHARACTER FACE is visible in the clip. Faceless edits run on BytePlus; faces are blocked there and must route to the relaxed provider, which costs ~35% more. A face edit submitted without this flag is rejected by the provider, not silently downgraded.'),
2008
2364
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2009
2365
  confirm: z.boolean().optional().describe('Set true to bypass the cost confirm gate after the user OKs the spend.'),
2010
2366
  }),
@@ -2016,10 +2372,13 @@ export const editVideo = {
2016
2372
  }
2017
2373
  const model = input.model ?? 'kling-v3.0-omni-edit';
2018
2374
  const isOmniFlashEdit = model === 'omni-flash-edit';
2019
- // Kling edit: 3–15s source clips; Omni Flash edit: 3–10s.
2020
- const maxClipSeconds = isOmniFlashEdit ? 10 : 15;
2021
- if (isOmniFlashEdit && ((input.characterAssetIds?.length ?? 0) > 0 || (input.styleAssetIds?.length ?? 0) > 0)) {
2022
- throw new Error('omni-flash-edit is prompt-only — it takes no character/style reference images. Drop the refs, or switch to kling-v3.0-omni-edit which supports elements.');
2375
+ const isSeedanceEdit = model === 'seedance-2.5-edit';
2376
+ // Kling edit: 3–15s source clips; Omni Flash edit: 3–10s; Seedance 2.5
2377
+ // edit: 4–30s — the only engine that takes a clip over 15s.
2378
+ const minClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MIN_SECONDS : 3;
2379
+ const maxClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MAX_SECONDS : isOmniFlashEdit ? 10 : 15;
2380
+ if ((isOmniFlashEdit || isSeedanceEdit) && ((input.characterAssetIds?.length ?? 0) > 0 || (input.styleAssetIds?.length ?? 0) > 0)) {
2381
+ throw new Error(`${model} takes the prompt and the source clip only on this op — no character/style reference images. Drop the refs, or switch to kling-v3.0-omni-edit which supports elements.`);
2023
2382
  }
2024
2383
  // Resolve refs (UUIDs or badge codes) against the project AT CALL TIME.
2025
2384
  const refInputs = [
@@ -2048,12 +2407,24 @@ export const editVideo = {
2048
2407
  if (!Number.isFinite(clipSeconds) || clipSeconds <= 0) {
2049
2408
  throw new Error('Source clip has no recorded duration — cannot quote the edit. Re-import the clip or pick another.');
2050
2409
  }
2051
- if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds < 2.95) {
2052
- throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts 3–${maxClipSeconds}s. ` +
2053
- `Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip.`);
2054
- }
2055
- const billedSeconds = Math.min(maxClipSeconds, Math.max(3, Math.ceil(clipSeconds - 0.05)));
2056
- const costKey = isOmniFlashEdit ? omniFlashEditCostKey(billedSeconds) : klingEditCostKey(model, billedSeconds);
2410
+ if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds < minClipSeconds - 0.05) {
2411
+ throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts ${minClipSeconds}–${maxClipSeconds}s. ` +
2412
+ `Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip` +
2413
+ (maxClipSeconds < SEEDANCE_25_EDIT_MAX_SECONDS ? ', or switch to seedance-2.5-edit which accepts up to 30s' : '') +
2414
+ '.');
2415
+ }
2416
+ const billedSeconds = Math.min(maxClipSeconds, Math.max(minClipSeconds, Math.ceil(clipSeconds - 0.05)));
2417
+ // Three engines, three key shapes — only Seedance's carries a resolution and
2418
+ // a face route, because only its price moves with them.
2419
+ const costKey = isSeedanceEdit
2420
+ ? seedanceEditCostKey({
2421
+ duration: billedSeconds,
2422
+ videoResolution: input.videoResolution ?? '720p',
2423
+ seedanceFace: input.seedanceFace === true,
2424
+ })
2425
+ : isOmniFlashEdit
2426
+ ? omniFlashEditCostKey(billedSeconds)
2427
+ : klingEditCostKey(model, billedSeconds);
2057
2428
  const cloud = ctx.cloud();
2058
2429
  const registry = await cloud.get('/api/agent/models');
2059
2430
  const entry = registry.models.find((m) => m.model === costKey);
@@ -2092,6 +2463,8 @@ export const editVideo = {
2092
2463
  characterAssetIds,
2093
2464
  styleAssetIds,
2094
2465
  keepAudio: input.keepAudio !== false,
2466
+ videoResolution: input.videoResolution,
2467
+ seedanceFace: input.seedanceFace,
2095
2468
  background: input.background,
2096
2469
  });
2097
2470
  if (!result.success)
@@ -2694,6 +3067,53 @@ export const updateFrame = {
2694
3067
  }));
2695
3068
  },
2696
3069
  };
3070
+ /**
3071
+ * Batch form of `slates_update_frame`.
3072
+ *
3073
+ * Writing a scene's worth of frames — motion prompts, shot labels — is ONE
3074
+ * logical edit. Doing it as N `slates_update_frame` calls costs N LLM
3075
+ * round-trips and makes the desktop refetch the whole storyboard N times for a
3076
+ * single intent. This is also where the storyboard's deleted "generate motion
3077
+ * prompts" button's capability went: the agent can be told "redo scene 3,
3078
+ * handheld", which that fixed-shape button never could.
3079
+ *
3080
+ * Hits `POST /agent/frames/batch-update`, which validates every id BEFORE the
3081
+ * first write and emits one broadcast for the batch.
3082
+ */
3083
+ export const batchUpdateFrames = {
3084
+ id: 'slates_batch_update_frames',
3085
+ description: 'Update MANY frames in one call — motion prompts, shot labels, notes, asset binding, scene/position, frameType. Prefer this over repeated slates_update_frame when writing a scene or a whole storyboard: it is one round-trip and one UI refresh. Every id is validated before anything is written, so the batch never lands half-applied.',
3086
+ input: z.object({
3087
+ updates: z
3088
+ .array(z.object({
3089
+ frameId: z.string().uuid(),
3090
+ shotLabel: z.string().optional(),
3091
+ notes: z.string().optional(),
3092
+ assetId: z.string().uuid().nullable().optional(),
3093
+ sceneId: z.string().uuid().nullable().optional(),
3094
+ position: z.number().int().min(0).optional(),
3095
+ frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
3096
+ motionPrompt: z.string().nullable().optional(),
3097
+ }))
3098
+ .min(1),
3099
+ }),
3100
+ async run(input, ctx) {
3101
+ return ok(await ctx.desktop().post('/agent/frames/batch-update', {
3102
+ updates: input.updates.map((u) => ({
3103
+ id: u.frameId,
3104
+ data: {
3105
+ shotLabel: u.shotLabel,
3106
+ notes: u.notes,
3107
+ assetId: u.assetId,
3108
+ sceneId: u.sceneId,
3109
+ position: u.position,
3110
+ frameType: u.frameType,
3111
+ motionPrompt: u.motionPrompt,
3112
+ },
3113
+ })),
3114
+ }));
3115
+ },
3116
+ };
2697
3117
  export const deleteFrame = {
2698
3118
  id: 'slates_delete_frame',
2699
3119
  description: 'Delete a frame from its scene (the referenced asset is untouched).',
@@ -2738,6 +3158,33 @@ function resolveGuideTopic(topic) {
2738
3158
  return 'slates-prompting-kling-v3';
2739
3159
  if (t.startsWith('kling-v3'))
2740
3160
  return 'slates-prompting-kling-v3';
3161
+ // Audio — seed-audio BEFORE the seedance check: "seed-audio" also starts
3162
+ // with "seed", and falling through would hand the video guide to the audio
3163
+ // model (the exact class of aliasing bug this comment block warns about).
3164
+ // Speech, dialogue and scratch VO all live on Seed Audio now — the TTS
3165
+ // surface is gone, so "tts"/"voiceover" must NOT land on the ElevenLabs
3166
+ // guide, which is SFX-only.
3167
+ if (t.startsWith('seed-audio') ||
3168
+ t === 'seed audio' ||
3169
+ t === 'audio' ||
3170
+ t === 'tts' ||
3171
+ t === 'voiceover' ||
3172
+ t === 'dialogue') {
3173
+ return 'slates-prompting-seed-audio';
3174
+ }
3175
+ if (t.startsWith('eleven') || t.startsWith('elevenlabs') || t === 'sfx' || t === 'sound-effects' || t === 'sound effects') {
3176
+ return 'slates-prompting-elevenlabs';
3177
+ }
3178
+ // ⚠️ 2.5 BEFORE the generic `seedance` prefix — the same ordering trap as
3179
+ // seed-audio-before-seedance above. Falling through silently hands the 2.0
3180
+ // guide to 2.5, whose limits, resolutions and task types are all different.
3181
+ if (t.startsWith('seedance-2.5') ||
3182
+ t.startsWith('seedance-25') ||
3183
+ t.startsWith('seedance-2-5') ||
3184
+ t.startsWith('seedance2.5') ||
3185
+ t === 'seedance 2.5') {
3186
+ return 'slates-prompting-seedance-2-5';
3187
+ }
2741
3188
  if (t.startsWith('seedance'))
2742
3189
  return 'slates-prompting-seedance';
2743
3190
  if (t.startsWith('avatar-') || t.includes('lip-sync'))
@@ -2764,7 +3211,7 @@ export const getPromptingGuide = {
2764
3211
  topic: z
2765
3212
  .string()
2766
3213
  .min(1)
2767
- .describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
3214
+ .describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
2768
3215
  }),
2769
3216
  async run(input) {
2770
3217
  const resolved = resolveGuideTopic(input.topic);
@@ -2796,6 +3243,9 @@ export const ALL_OPERATIONS = [
2796
3243
  listFolders,
2797
3244
  createFolder,
2798
3245
  moveAssetsToFolder,
3246
+ moveAssetsToProject,
3247
+ copyAssetsToProject,
3248
+ moveEntityToProject,
2799
3249
  listCharacters,
2800
3250
  createCharacter,
2801
3251
  setCharacterIdentity,
@@ -2810,6 +3260,7 @@ export const ALL_OPERATIONS = [
2810
3260
  addFrame,
2811
3261
  generateImage,
2812
3262
  generateVideo,
3263
+ generateAudio,
2813
3264
  generateLipSync,
2814
3265
  generateMotionTransfer,
2815
3266
  editVideo,
@@ -2849,6 +3300,7 @@ export const ALL_OPERATIONS = [
2849
3300
  deleteScene,
2850
3301
  reorderScenes,
2851
3302
  updateFrame,
3303
+ batchUpdateFrames,
2852
3304
  deleteFrame,
2853
3305
  getPromptingGuide,
2854
3306
  ];