@slatesvideo/shared 0.5.6 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +4 -1
- package/dist/operations/index.d.ts +43 -36
- package/dist/operations/index.js +289 -262
- package/dist/prompts/model-facts.d.ts +42 -0
- package/dist/prompts/model-facts.js +88 -18
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +108 -99
- package/dist/prompts/reference-composer.d.ts +21 -3
- package/dist/prompts/reference-composer.js +80 -10
- package/dist/skills/content.js +7 -7
- package/exports/slates-prompt-builder/generated/SKILL.md +2 -2
- package/exports/slates-prompt-builder/generated/reference-seedance.md +12 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +8 -8
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +1 -1
- package/skills/slates-model-selection.md +21 -19
- package/skills/slates-prompting-elevenlabs.md +16 -78
- package/skills/slates-prompting-lip-sync.md +12 -14
- package/skills/slates-prompting-motion-transfer.md +18 -14
- package/skills/slates-prompting-seed-audio.md +2 -2
- package/skills/slates-prompting-seedance-2-5.md +215 -0
- package/skills/slates-prompting-seedance.md +17 -1
- package/skills/slates-prompting-suno.md +0 -110
package/dist/operations/index.js
CHANGED
|
@@ -13,6 +13,9 @@ import { z } from 'zod';
|
|
|
13
13
|
import { SlatesCloudClient } from '../clients/cloud.js';
|
|
14
14
|
import { SlatesDesktopClient } from '../clients/desktop.js';
|
|
15
15
|
import { SKILLS } from '../skills/content.js';
|
|
16
|
+
// Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
|
|
17
|
+
// "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
|
|
18
|
+
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } from '../prompts/model-facts.js';
|
|
16
19
|
export function defaultContext() {
|
|
17
20
|
return {
|
|
18
21
|
cloud: () => new SlatesCloudClient(),
|
|
@@ -136,8 +139,7 @@ export const estimateGenerationCost = {
|
|
|
136
139
|
input: z.object({
|
|
137
140
|
model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
|
|
138
141
|
quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
|
|
139
|
-
duration: z.number().int().min(1).max(360).optional().describe('Seconds. Video 3-15 (cost scales linearly; required with a video base id). Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22
|
|
140
|
-
characters: z.number().int().min(1).max(5000).optional().describe('eleven-v3 only — script length in characters, billed in 100-char buckets rounded up.'),
|
|
142
|
+
duration: z.number().int().min(1).max(360).optional().describe('Seconds. Video 3-15 (cost scales linearly; required with a video base id). Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22 — required with either audio base id.'),
|
|
141
143
|
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Video only. Seedance defaults to 1080p.'),
|
|
142
144
|
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
|
|
143
145
|
quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
|
|
@@ -156,13 +158,13 @@ export const estimateGenerationCost = {
|
|
|
156
158
|
if (img)
|
|
157
159
|
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
|
|
158
160
|
}
|
|
159
|
-
// 2a) audio base id → seconds
|
|
160
|
-
//
|
|
161
|
+
// 2a) audio base id → seconds. Both surfaces bill per second, so a
|
|
162
|
+
// duration is always required. Runs BEFORE the video resolver: it is
|
|
161
163
|
// forgiving by design and "seed-audio" would otherwise be mistaken for
|
|
162
164
|
// a seedance spelling.
|
|
163
165
|
if (!key && AUDIO_MODELS.includes(input.model)) {
|
|
164
166
|
const m = input.model;
|
|
165
|
-
if (
|
|
167
|
+
if (!input.duration) {
|
|
166
168
|
return ok({
|
|
167
169
|
requires_clarification: true,
|
|
168
170
|
missing: ['duration'],
|
|
@@ -171,13 +173,6 @@ export const estimateGenerationCost = {
|
|
|
171
173
|
: 'Sound Effects bills per second — pass duration in seconds (1-22).',
|
|
172
174
|
});
|
|
173
175
|
}
|
|
174
|
-
if (m === 'eleven-v3' && !input.characters) {
|
|
175
|
-
return ok({
|
|
176
|
-
requires_clarification: true,
|
|
177
|
-
missing: ['characters'],
|
|
178
|
-
message: 'Eleven v3 bills per 100 characters of script, rounded up — pass characters (the length of the text you intend to speak).',
|
|
179
|
-
});
|
|
180
|
-
}
|
|
181
176
|
// REFUSE an out-of-range duration rather than quoting the clamped price.
|
|
182
177
|
// audioCostKey clamps (it has to — it mirrors the desktop, which clamps),
|
|
183
178
|
// so without this an agent asking for 2s of Seed Audio would be handed a
|
|
@@ -187,20 +182,17 @@ export const estimateGenerationCost = {
|
|
|
187
182
|
const audioBounds = {
|
|
188
183
|
'seed-audio': { min: SEED_AUDIO_MIN_SECONDS, max: SEED_AUDIO_MAX_SECONDS },
|
|
189
184
|
'eleven-sfx': { min: ELEVEN_SFX_MIN_SECONDS, max: ELEVEN_SFX_MAX_SECONDS },
|
|
190
|
-
suno: { min: SUNO_MIN_SECONDS, max: SUNO_MAX_SECONDS },
|
|
191
185
|
};
|
|
192
186
|
const bounds = audioBounds[m];
|
|
193
|
-
if (
|
|
187
|
+
if (input.duration < bounds.min || input.duration > bounds.max) {
|
|
194
188
|
return ok({
|
|
195
189
|
requires_clarification: true,
|
|
196
190
|
missing: ['duration'],
|
|
197
191
|
message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
|
|
198
|
-
|
|
199
|
-
? 'Suno duration is FREE and flat-priced, so any value in range costs the same.'
|
|
200
|
-
: 'Re-ask with a duration in range.'),
|
|
192
|
+
'Re-ask with a duration in range.',
|
|
201
193
|
});
|
|
202
194
|
}
|
|
203
|
-
key = audioCostKey({ model: m, durationSeconds: input.duration
|
|
195
|
+
key = audioCostKey({ model: m, durationSeconds: input.duration });
|
|
204
196
|
}
|
|
205
197
|
// 2b) Kling O3 edit base id + duration (ceiled source-clip length)
|
|
206
198
|
if (!key && (input.model === 'kling-v3.0-omni-edit' || input.model === 'kling-v3.0-omni-pro-edit')) {
|
|
@@ -231,7 +223,12 @@ export const estimateGenerationCost = {
|
|
|
231
223
|
duration,
|
|
232
224
|
videoResolution: input.videoResolution ??
|
|
233
225
|
resolved.videoResolution ??
|
|
234
|
-
|
|
226
|
+
// Seedance quotes are resolution-scaled, so a missing resolution has
|
|
227
|
+
// to fall back to the model's OWN default — 2.5 has no 1080p at all,
|
|
228
|
+
// and a blanket '1080p' here quoted a key that does not exist.
|
|
229
|
+
(resolved.model === 'seedance-2.5' ? '720p'
|
|
230
|
+
: resolved.model.startsWith('seedance') ? '1080p'
|
|
231
|
+
: undefined),
|
|
235
232
|
sound: input.sound ?? resolved.sound,
|
|
236
233
|
seedanceFace: input.seedanceFace ?? resolved.seedanceFace,
|
|
237
234
|
seedanceRealFace: input.seedanceRealFace,
|
|
@@ -240,7 +237,7 @@ export const estimateGenerationCost = {
|
|
|
240
237
|
}
|
|
241
238
|
const perCredits = key != null ? byKey.get(key) : undefined;
|
|
242
239
|
if (key == null || perCredits == null) {
|
|
243
|
-
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/
|
|
240
|
+
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
|
|
244
241
|
}
|
|
245
242
|
const qty = input.quantity ?? 1;
|
|
246
243
|
const totalCredits = perCredits * qty;
|
|
@@ -1323,6 +1320,11 @@ export const VIDEO_MODELS = [
|
|
|
1323
1320
|
'veo-3.1-fast',
|
|
1324
1321
|
'veo-3.1-standard',
|
|
1325
1322
|
'seedance-2',
|
|
1323
|
+
// Seedance 2.5 is a SECOND SEAT, not a replacement: 30s takes, 30 image
|
|
1324
|
+
// references, audio-only references — and 480p/720p ONLY. 2.0 keeps the
|
|
1325
|
+
// ladder to native 4K and stays the default. Its EDIT row is not here; edit
|
|
1326
|
+
// models live on slates_edit_video, same as the Kling and Omni Flash ones.
|
|
1327
|
+
'seedance-2.5',
|
|
1326
1328
|
'omni-flash',
|
|
1327
1329
|
];
|
|
1328
1330
|
// Model → registry cost-key. Each provider's keys ship with their own
|
|
@@ -1346,16 +1348,24 @@ const KLING_TIER_MAP = {
|
|
|
1346
1348
|
// asserts this builder byte-matches the desktop's klingCreditKey/seedanceCreditKey.
|
|
1347
1349
|
export function videoCostKey(input) {
|
|
1348
1350
|
if (input.model.startsWith('seedance')) {
|
|
1349
|
-
// Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (
|
|
1350
|
-
// res × duration). AI-face route bills the `-face-` key (~45% over
|
|
1351
|
-
// consented real-person route bills the premium `-realface-` key
|
|
1352
|
-
// endpoint). A reference video flips to `-vref-{res}-{T}s`,
|
|
1353
|
-
|
|
1351
|
+
// Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
|
|
1352
|
+
// × vref × res × duration). AI-face route bills the `-face-` key (~45% over
|
|
1353
|
+
// faceless); consented real-person route bills the premium `-realface-` key
|
|
1354
|
+
// (fal partner endpoint). A reference video flips to `-vref-{res}-{T}s`,
|
|
1355
|
+
// T = in + out.
|
|
1356
|
+
//
|
|
1357
|
+
// ⚠️ EVERY BOUND HERE IS VERSION-SCOPED. 2.5 is 480p/720p only, runs to 30s,
|
|
1358
|
+
// and takes references to 30s combined — so its vref total reaches 60, DOUBLE
|
|
1359
|
+
// 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would quote a real key at a
|
|
1360
|
+
// fraction of the real bill.
|
|
1361
|
+
const v25 = input.model.startsWith('seedance-2.5');
|
|
1362
|
+
const res = input.videoResolution ?? (v25 ? '720p' : '1080p');
|
|
1354
1363
|
const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
|
|
1355
1364
|
const vrefSecs = input.videoRefSeconds ?? 0;
|
|
1356
1365
|
if (vrefSecs > 0) {
|
|
1357
1366
|
// ceil(x - 0.05) matches the server's probe rounding — quote = bill.
|
|
1358
|
-
const
|
|
1367
|
+
const maxTotal = v25 ? SEEDANCE_25_VREF_MAX_TOTAL : SEEDANCE_20_VREF_MAX_TOTAL;
|
|
1368
|
+
const total = Math.min(maxTotal, Math.max(6, Math.ceil(vrefSecs - 0.05) + input.duration));
|
|
1359
1369
|
return `${input.model}${face}-vref-${res}-${total}s`;
|
|
1360
1370
|
}
|
|
1361
1371
|
return `${input.model}${face}-${res}-${input.duration}s`;
|
|
@@ -1405,11 +1415,30 @@ export function klingEditCostKey(model, duration) {
|
|
|
1405
1415
|
export function omniFlashEditCostKey(duration) {
|
|
1406
1416
|
return `omni-flash-edit-${duration}s`;
|
|
1407
1417
|
}
|
|
1418
|
+
/** Max billed (input + output) seconds on a Seedance video-reference gen.
|
|
1419
|
+
* 2.0: refs 2–15s + output 4–15s. 2.5: refs to 30s + output to 30s. */
|
|
1420
|
+
const SEEDANCE_20_VREF_MAX_TOTAL = 30;
|
|
1421
|
+
const SEEDANCE_25_VREF_MAX_TOTAL = 60;
|
|
1422
|
+
/** Seedance 2.5 source-clip bounds for the edit task type. */
|
|
1423
|
+
export const SEEDANCE_25_EDIT_MIN_SECONDS = 4;
|
|
1424
|
+
export const SEEDANCE_25_EDIT_MAX_SECONDS = 30;
|
|
1425
|
+
// Seedance 2.5 video-edit cost key — mirrors seedanceCreditKey() in
|
|
1426
|
+
// slate/src/shared/pricing.ts for the edit row (must byte-match; checked by the
|
|
1427
|
+
// slates-api pricing-consistency script). Unlike the Kling and Omni Flash edit
|
|
1428
|
+
// keys this one carries a RESOLUTION and a FACE ROUTE, because Seedance's edit
|
|
1429
|
+
// price moves with both. Duration is the CEILED source-clip length: the edit
|
|
1430
|
+
// task type forces `duration: -1` on the wire, so the source clip is the only
|
|
1431
|
+
// honest quote.
|
|
1432
|
+
export function seedanceEditCostKey(input) {
|
|
1433
|
+
const res = input.videoResolution ?? '720p';
|
|
1434
|
+
const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
|
|
1435
|
+
return `seedance-2.5-edit${face}-${res}-${input.duration}s`;
|
|
1436
|
+
}
|
|
1408
1437
|
// ── Audio ───────────────────────────────────────────────────────
|
|
1409
1438
|
// Exported: the exact `model` ids slates_generate_audio accepts — consumed
|
|
1410
1439
|
// by the desktop Studio Agent system prompt (SSOT; never restate these ids
|
|
1411
1440
|
// in prose that can drift). Mirrors VIDEO_MODELS for the third media type.
|
|
1412
|
-
export const AUDIO_MODELS = ['seed-audio', 'eleven-
|
|
1441
|
+
export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx'];
|
|
1413
1442
|
/**
|
|
1414
1443
|
* Per-surface bounds and defaults.
|
|
1415
1444
|
*
|
|
@@ -1434,9 +1463,6 @@ export const SEED_AUDIO_DEFAULT_SECONDS = 15;
|
|
|
1434
1463
|
export const ELEVEN_SFX_MIN_SECONDS = 1;
|
|
1435
1464
|
export const ELEVEN_SFX_MAX_SECONDS = 22;
|
|
1436
1465
|
export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
|
|
1437
|
-
export const ELEVEN_V3_MAX_CHARACTERS = 5000;
|
|
1438
|
-
export const SUNO_MIN_SECONDS = 10;
|
|
1439
|
-
export const SUNO_MAX_SECONDS = 360;
|
|
1440
1466
|
/**
|
|
1441
1467
|
* Byte-for-byte the desktop's `clampAudioDuration` in slate/src/shared/pricing.ts,
|
|
1442
1468
|
* INCLUDING the non-finite arm — that one matters: `Math.max(min, NaN)` is NaN,
|
|
@@ -1468,25 +1494,10 @@ export function audioCostKey(input) {
|
|
|
1468
1494
|
const secs = clampAudioSeconds(input.durationSeconds ?? SEED_AUDIO_DEFAULT_SECONDS, SEED_AUDIO_MIN_SECONDS, SEED_AUDIO_MAX_SECONDS, SEED_AUDIO_DEFAULT_SECONDS);
|
|
1469
1495
|
return `seed-audio-${secs}s`;
|
|
1470
1496
|
}
|
|
1471
|
-
if (input.model === 'eleven-v3') {
|
|
1472
|
-
// 100-character buckets, always rounded UP — never under-bill a read.
|
|
1473
|
-
// Mirrors ttsBuckets() in slate/src/shared/pricing.ts, non-finite arm included.
|
|
1474
|
-
const chars = input.characters ?? 0;
|
|
1475
|
-
const buckets = Number.isFinite(chars)
|
|
1476
|
-
? Math.min(ELEVEN_V3_MAX_CHARACTERS / 100, Math.max(1, Math.ceil(chars / 100)))
|
|
1477
|
-
: 1;
|
|
1478
|
-
return `eleven-v3-tts-${buckets}00c`;
|
|
1479
|
-
}
|
|
1480
1497
|
if (input.model === 'eleven-sfx') {
|
|
1481
1498
|
const secs = clampAudioSeconds(input.durationSeconds ?? ELEVEN_SFX_DEFAULT_SECONDS, ELEVEN_SFX_MIN_SECONDS, ELEVEN_SFX_MAX_SECONDS, ELEVEN_SFX_DEFAULT_SECONDS);
|
|
1482
1499
|
return `eleven-sfx-${secs}s`;
|
|
1483
1500
|
}
|
|
1484
|
-
if (input.model === 'suno') {
|
|
1485
|
-
// Flat across every Suno model AND every duration up to 360s — measured
|
|
1486
|
-
// against the live sunoapi.org balance 2026-07-31 (12 credits for a
|
|
1487
|
-
// default call and for duration=240 alike). One call returns TWO songs.
|
|
1488
|
-
return 'suno-generate';
|
|
1489
|
-
}
|
|
1490
1501
|
throw new Error(`Unknown audio model: ${input.model}`);
|
|
1491
1502
|
}
|
|
1492
1503
|
/**
|
|
@@ -1533,6 +1544,13 @@ function resolveVideoModel(raw) {
|
|
|
1533
1544
|
'kling-v3.0-omni-pro': 'kling-v3.0-omni',
|
|
1534
1545
|
'seedance-2.0': 'seedance-2',
|
|
1535
1546
|
'seedance-2-0': 'seedance-2',
|
|
1547
|
+
// ⚠️ The 2.5 spellings must resolve to 2.5, and the BARE `seedance` must keep
|
|
1548
|
+
// resolving to 2.0 — 2.0 is the default video model and holds 1080p/4K, which
|
|
1549
|
+
// 2.5 does not have at all.
|
|
1550
|
+
'seedance-2.5': 'seedance-2.5',
|
|
1551
|
+
'seedance-25': 'seedance-2.5',
|
|
1552
|
+
'seedance-2-5': 'seedance-2.5',
|
|
1553
|
+
'seedance2.5': 'seedance-2.5',
|
|
1536
1554
|
seedance: 'seedance-2',
|
|
1537
1555
|
'veo-3.1': 'veo-3.1-fast',
|
|
1538
1556
|
'veo-3': 'veo-3.1-fast',
|
|
@@ -1555,6 +1573,11 @@ function promptingSkillFor(model) {
|
|
|
1555
1573
|
return 'slates-prompting-kling-v3';
|
|
1556
1574
|
if (model.startsWith('veo'))
|
|
1557
1575
|
return 'slates-prompting-veo-3';
|
|
1576
|
+
// 2.5 BEFORE the generic seedance test — "seedance-2.5" also starts with
|
|
1577
|
+
// "seedance", and falling through hands 2.0's guide to a model with different
|
|
1578
|
+
// limits, a different resolution ladder and an extra task type.
|
|
1579
|
+
if (model.startsWith('seedance-2.5'))
|
|
1580
|
+
return 'slates-prompting-seedance-2-5';
|
|
1558
1581
|
if (model.startsWith('seedance'))
|
|
1559
1582
|
return 'slates-prompting-seedance';
|
|
1560
1583
|
if (model.startsWith('omni-flash'))
|
|
@@ -1563,23 +1586,30 @@ function promptingSkillFor(model) {
|
|
|
1563
1586
|
}
|
|
1564
1587
|
export const generateVideo = {
|
|
1565
1588
|
id: 'slates_generate_video',
|
|
1566
|
-
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
|
|
1589
|
+
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
|
|
1567
1590
|
input: z.object({
|
|
1568
1591
|
prompt: z.string().min(1).max(4000),
|
|
1569
|
-
model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
|
|
1592
|
+
model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | seedance-2.5 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (4-30s takes, 30 image refs, audio-only refs — but 480p/720p ONLY, so stay on seedance-2 whenever resolution matters), Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
|
|
1570
1593
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
|
|
1571
1594
|
aspectRatio: z.enum(['1:1', '16:9', '9:16', '4:3', '3:4', '21:9', '9:21', '4:5', '5:4', '2:3', '3:2']).optional().describe('Veo locks to 16:9 — passing anything else will be ignored or fail. Kling/Seedance support all.'),
|
|
1572
|
-
duration: z.number().int().min(3).max(
|
|
1573
|
-
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive). Veo: 720p/1080p same price, 4K more (8s only).'),
|
|
1595
|
+
duration: z.number().int().min(3).max(30).optional().describe('Seconds. Kling: 5-15. Veo: 4, 6, or 8 only (4K only at 8s). Seedance 2: 4-15. Seedance 2.5: 4-30 (the only model that reaches 30). Omni Flash: 3-10. Default 5 if omitted but always be explicit — cost scales linearly, and a 30s seedance-2.5 take is several hundred credits.'),
|
|
1596
|
+
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance 2: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive, and Pro-only). Seedance 2.5: 480p/720p ONLY (default 720p) — it has no 1080p and no 4K on any provider, so asking for one is rejected, not downgraded. Veo: 720p/1080p same price, 4K more (8s only).'),
|
|
1574
1597
|
firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video: asset UUID or badge code ("IMG-A8") — codes resolve against the project at call time, so a code the user just spoke is always safe to pass.'),
|
|
1575
1598
|
lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code). Veo and Seedance only. Pairs with firstFrameAssetId for guided transitions.'),
|
|
1576
|
-
ingredientAssetIds: z.array(z.string()).max(
|
|
1599
|
+
ingredientAssetIds: z.array(z.string()).max(30).optional().describe('Visual reference / ingredient assets (UUIDs or badge codes) for Kling Omni, Seedance, or Omni Flash. Up to 30 (Seedance 2.5), 9 (Seedance 2), 4 (Kling), or 7 (Omni Flash, combined across all ref params). More is not better: 2-4 strong references beat both extremes, and past 4 reference PEOPLE output stability drops on Seedance regardless of the cap.'),
|
|
1577
1600
|
characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets (UUIDs or badge codes) — keeps a character consistent across the shot.'),
|
|
1578
1601
|
environmentAssetIds: z.array(z.string()).optional().describe('Environment reference assets (UUIDs or badge codes) — keeps a location/setting consistent across the shot.'),
|
|
1579
1602
|
styleAssetIds: z.array(z.string()).optional().describe('Style reference assets (UUIDs or badge codes) — locks the visual style of the shot.'),
|
|
1580
|
-
videoReferenceAssetId: z.string().optional().describe('
|
|
1581
|
-
videoReferenceSeconds: z.number().optional().describe('
|
|
1582
|
-
audioReferenceAssetId: z.string().optional().describe('
|
|
1603
|
+
videoReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into videoReferenceAssetIds; prefer that for anything new. A single VIDEO asset (UUID or badge code) used as a reference. Kept working forever: installed CLI and MCP builds send this shape.'),
|
|
1604
|
+
videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach. Required with videoReferenceAssetId: that clip\'s duration in seconds.'),
|
|
1605
|
+
audioReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into audioReferenceAssetIds; prefer that. A single AUDIO asset (UUID or badge code) used as a reference.'),
|
|
1606
|
+
// ── Multimodal references, the plural surface ──
|
|
1607
|
+
// The capacity sentences are DERIVED from MODEL_FACTS (see
|
|
1608
|
+
// multimodalRefSummary) rather than hand-typed, so a cap change in one
|
|
1609
|
+
// place cannot leave a stale number in a description an LLM reads.
|
|
1610
|
+
videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS (UUIDs or badge codes) read alongside the images and audio in the same generation — own-footage restyle, MOTION TRANSFER ("the character from image 1 performs the motion from video 1"), or dialogue conditioning. Cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; ignored elsewhere. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to combined input+output seconds (the vref key) — pass videoReferenceSecondsEach so the quote is right. If any clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Over the cap is REFUSED, never trimmed: a dropped clip would already have been priced in.`),
|
|
1611
|
+
videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — the bill is Σceil(each) + output seconds. The server re-derives this by probing every uploaded clip, so an understated value just gets corrected upward.'),
|
|
1612
|
+
audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO clips (UUIDs or badge codes) read alongside the images and video — e.g. lip-sync a character to a line ("the character in image 1 speaks the dialogue from audio 1"). Cited as "audio 1", "audio 2"… in the order given. No billing surcharge (Seedance audio is included). ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
|
|
1583
1613
|
sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
|
|
1584
1614
|
audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
|
|
1585
1615
|
generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
|
|
@@ -1648,7 +1678,10 @@ export const generateVideo = {
|
|
|
1648
1678
|
message: `Omni Flash supports 3-10 seconds (you passed ${input.duration}s). Pick a duration in that range.`,
|
|
1649
1679
|
});
|
|
1650
1680
|
}
|
|
1651
|
-
if (input.lastFrameAssetId ||
|
|
1681
|
+
if (input.lastFrameAssetId ||
|
|
1682
|
+
input.videoReferenceAssetId || input.audioReferenceAssetId ||
|
|
1683
|
+
(input.videoReferenceAssetIds?.length ?? 0) > 0 ||
|
|
1684
|
+
(input.audioReferenceAssetIds?.length ?? 0) > 0) {
|
|
1652
1685
|
return ok({
|
|
1653
1686
|
requires_clarification: true,
|
|
1654
1687
|
missing: [],
|
|
@@ -1709,6 +1742,10 @@ export const generateVideo = {
|
|
|
1709
1742
|
refInputs.push({ ref: input.videoReferenceAssetId, role: 'video reference' });
|
|
1710
1743
|
if (input.audioReferenceAssetId)
|
|
1711
1744
|
refInputs.push({ ref: input.audioReferenceAssetId, role: 'audio reference' });
|
|
1745
|
+
for (const r of input.videoReferenceAssetIds ?? [])
|
|
1746
|
+
refInputs.push({ ref: r, role: 'video reference' });
|
|
1747
|
+
for (const r of input.audioReferenceAssetIds ?? [])
|
|
1748
|
+
refInputs.push({ ref: r, role: 'audio reference' });
|
|
1712
1749
|
const resolvedRefs = await resolveAssetRefs(ctx, input.projectId, refInputs.map((r) => r.ref));
|
|
1713
1750
|
const rid = (v) => v ? (resolvedRefs.get(v)?.id ?? v) : v;
|
|
1714
1751
|
const rids = (a) => a?.map((v) => resolvedRefs.get(v)?.id ?? v);
|
|
@@ -1716,6 +1753,8 @@ export const generateVideo = {
|
|
|
1716
1753
|
input.lastFrameAssetId = rid(input.lastFrameAssetId);
|
|
1717
1754
|
input.videoReferenceAssetId = rid(input.videoReferenceAssetId);
|
|
1718
1755
|
input.audioReferenceAssetId = rid(input.audioReferenceAssetId);
|
|
1756
|
+
input.videoReferenceAssetIds = rids(input.videoReferenceAssetIds);
|
|
1757
|
+
input.audioReferenceAssetIds = rids(input.audioReferenceAssetIds);
|
|
1719
1758
|
input.ingredientAssetIds = rids(input.ingredientAssetIds);
|
|
1720
1759
|
input.characterAssetIds = rids(input.characterAssetIds);
|
|
1721
1760
|
input.environmentAssetIds = rids(input.environmentAssetIds);
|
|
@@ -1724,13 +1763,59 @@ export const generateVideo = {
|
|
|
1724
1763
|
// A video reference bills combined input+output seconds (the vref key) —
|
|
1725
1764
|
// the quote needs the clip's length. The server probes the uploaded ref
|
|
1726
1765
|
// and corrects the key anyway, so this only gates quote accuracy.
|
|
1727
|
-
|
|
1766
|
+
// 🚨 Seedance 2.5 PROMPT-INTENT PRE-FLIGHT. With references attached, the
|
|
1767
|
+
// provider sorts a request into one of five task types from the roles PLUS
|
|
1768
|
+
// the prompt's intent, then fails a mismatch ASYNCHRONOUSLY — after the task
|
|
1769
|
+
// queues and credits are reserved. Catch it before the spend, NAME the words,
|
|
1770
|
+
// and never rewrite the prompt (the prompt-transparency invariant).
|
|
1771
|
+
if (input.model === 'seedance-2.5') {
|
|
1772
|
+
// 🚨 EVERY reference shape counts, singular AND plural. The classifier
|
|
1773
|
+
// engages on the presence of reference ROLES, and the plural arrays are
|
|
1774
|
+
// the surface new callers use — listing only the deprecated singular ones
|
|
1775
|
+
// meant the modern call path got no warning at all, which is precisely
|
|
1776
|
+
// backwards.
|
|
1777
|
+
const hasRefs = !!input.firstFrameAssetId || !!input.lastFrameAssetId ||
|
|
1778
|
+
(input.ingredientAssetIds?.length ?? 0) > 0 ||
|
|
1779
|
+
(input.characterAssetIds?.length ?? 0) > 0 ||
|
|
1780
|
+
(input.environmentAssetIds?.length ?? 0) > 0 ||
|
|
1781
|
+
(input.styleAssetIds?.length ?? 0) > 0 ||
|
|
1782
|
+
!!input.videoReferenceAssetId || !!input.audioReferenceAssetId ||
|
|
1783
|
+
(input.videoReferenceAssetIds?.length ?? 0) > 0 ||
|
|
1784
|
+
(input.audioReferenceAssetIds?.length ?? 0) > 0;
|
|
1785
|
+
// Trigger words are DERIVED from the model-facts SSOT, not retyped here —
|
|
1786
|
+
// a local literal had already lost 'continue the story'.
|
|
1787
|
+
const hits = hasRefs ? seedanceTaskIntentWords(input.prompt) : [];
|
|
1788
|
+
if (hits.length > 0 && !input.confirm) {
|
|
1789
|
+
return ok({
|
|
1790
|
+
requires_clarification: true,
|
|
1791
|
+
missing: ['prompt'],
|
|
1792
|
+
message: `Seedance 2.5 reads ${hits.map((w) => `"${w}"`).join(', ')} in this prompt as an EDIT or EXTEND instruction and may run it as a video edit, ` +
|
|
1793
|
+
`which fails after the job has queued. If you mean to edit an existing clip, call slates_edit_video with model "seedance-2.5-edit". ` +
|
|
1794
|
+
`If you mean a fresh shot, describe the finished frame instead of an instruction to change one ("the workshop bench, clear and uncluttered" rather than "remove the tripod"). ` +
|
|
1795
|
+
`Pass confirm=true to send it as written.`,
|
|
1796
|
+
});
|
|
1797
|
+
}
|
|
1798
|
+
}
|
|
1799
|
+
// The quote needs a length for EVERY reference clip, in both shapes. A
|
|
1800
|
+
// missing one doesn't fail the generation (the server probes and corrects
|
|
1801
|
+
// upward) — it silently under-quotes, which is worse than asking.
|
|
1802
|
+
const isSeedance = input.model.startsWith('seedance');
|
|
1803
|
+
if (input.videoReferenceAssetId && isSeedance && !input.videoReferenceSeconds) {
|
|
1728
1804
|
return ok({
|
|
1729
1805
|
requires_clarification: true,
|
|
1730
1806
|
missing: ['videoReferenceSeconds'],
|
|
1731
1807
|
message: 'A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSeconds (the reference clip\'s duration, shown in slates_list_assets) so the pre-flight quote matches the bill.',
|
|
1732
1808
|
});
|
|
1733
1809
|
}
|
|
1810
|
+
const pluralRefCount = input.videoReferenceAssetIds?.length ?? 0;
|
|
1811
|
+
if (pluralRefCount > 0 && isSeedance &&
|
|
1812
|
+
(input.videoReferenceSecondsEach?.length ?? 0) !== pluralRefCount) {
|
|
1813
|
+
return ok({
|
|
1814
|
+
requires_clarification: true,
|
|
1815
|
+
missing: ['videoReferenceSecondsEach'],
|
|
1816
|
+
message: `A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSecondsEach with exactly ${pluralRefCount} duration${pluralRefCount === 1 ? '' : 's'}, in the same order as videoReferenceAssetIds (durations are shown in slates_list_assets), so the pre-flight quote matches the bill.`,
|
|
1817
|
+
});
|
|
1818
|
+
}
|
|
1734
1819
|
const cloud = ctx.cloud();
|
|
1735
1820
|
const registry = await cloud.get('/api/agent/models');
|
|
1736
1821
|
const costKey = videoCostKey({
|
|
@@ -1740,7 +1825,13 @@ export const generateVideo = {
|
|
|
1740
1825
|
sound: input.sound,
|
|
1741
1826
|
seedanceFace: input.seedanceFace,
|
|
1742
1827
|
seedanceRealFace: input.seedanceRealFace,
|
|
1743
|
-
|
|
1828
|
+
// Σ ceil(d - 0.05) over every reference clip, both shapes — the same
|
|
1829
|
+
// expression the desktop's estimateCost and the handler's key builder
|
|
1830
|
+
// use. Quoting only the singular would understate a multi-clip call.
|
|
1831
|
+
videoRefSeconds: (input.videoReferenceAssetId && input.videoReferenceSeconds
|
|
1832
|
+
? Math.ceil(input.videoReferenceSeconds - 0.05)
|
|
1833
|
+
: 0) +
|
|
1834
|
+
(input.videoReferenceSecondsEach ?? []).reduce((n, d) => n + (d > 0 ? Math.ceil(d - 0.05) : 0), 0),
|
|
1744
1835
|
});
|
|
1745
1836
|
// Hard consent gate, checked before any spend: the real-face route is
|
|
1746
1837
|
// consent-attested by design (the desktop enforces it too).
|
|
@@ -1834,8 +1925,13 @@ export const generateVideo = {
|
|
|
1834
1925
|
characterAssetIds: input.characterAssetIds ?? [],
|
|
1835
1926
|
environmentAssetIds: input.environmentAssetIds ?? [],
|
|
1836
1927
|
styleAssetIds: input.styleAssetIds ?? [],
|
|
1928
|
+
// Both shapes on the wire. The route merges and dedupes them, so an old
|
|
1929
|
+
// client sending only the singular and a new one sending only the plural
|
|
1930
|
+
// reach the identical handler params.
|
|
1837
1931
|
videoReferenceAssetId: input.videoReferenceAssetId,
|
|
1838
1932
|
audioReferenceAssetId: input.audioReferenceAssetId,
|
|
1933
|
+
videoReferenceAssetIds: input.videoReferenceAssetIds,
|
|
1934
|
+
audioReferenceAssetIds: input.audioReferenceAssetIds,
|
|
1839
1935
|
sound: input.sound,
|
|
1840
1936
|
audioLanguage: input.audioLanguage,
|
|
1841
1937
|
generateMusic: input.generateMusic,
|
|
@@ -1884,30 +1980,28 @@ export const generateVideo = {
|
|
|
1884
1980
|
// ── Generate audio ──────────────────────────────────────────────
|
|
1885
1981
|
export const generateAudio = {
|
|
1886
1982
|
id: 'slates_generate_audio',
|
|
1887
|
-
description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track.
|
|
1983
|
+
description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Two surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s) and eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop). Which surface for which job: read the slates-model-selection skill. ' +
|
|
1888
1984
|
'🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
|
|
1889
|
-
'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs
|
|
1985
|
+
'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
|
|
1890
1986
|
'projectId is REQUIRED (no headless path). Cost > 17 credits returns requires_confirm — pass confirm=true after explicit user OK. No skill files installed? Call slates_get_prompting_guide first.',
|
|
1891
1987
|
input: z.object({
|
|
1892
1988
|
projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
|
|
1893
1989
|
model: z
|
|
1894
1990
|
.enum(AUDIO_MODELS)
|
|
1895
|
-
.describe('Audio surface. seed-audio = scene/ambience/beds (default choice), eleven-
|
|
1991
|
+
.describe('Audio surface. seed-audio = scene/ambience/beds/dialogue (default choice), eleven-sfx = one precise effect or a seamless loop. Routing doctrine: slates-model-selection skill.'),
|
|
1896
1992
|
prompt: z
|
|
1897
1993
|
.string()
|
|
1898
1994
|
.min(1)
|
|
1899
1995
|
.max(5000)
|
|
1900
|
-
.describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-
|
|
1996
|
+
.describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars.'),
|
|
1901
1997
|
durationSeconds: z
|
|
1902
1998
|
.number()
|
|
1903
1999
|
.optional()
|
|
1904
|
-
.describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.
|
|
2000
|
+
.describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.'),
|
|
1905
2001
|
voice: z
|
|
1906
2002
|
.string()
|
|
1907
2003
|
.optional()
|
|
1908
|
-
.describe('seed-audio
|
|
1909
|
-
stability: z.number().min(0).max(1).optional().describe('eleven-v3 only. 0-1, default 0.5. Lower = more expressive and more variable take-to-take; higher = flatter and repeatable. Raise it for long narration.'),
|
|
1910
|
-
languageCode: z.string().optional().describe('eleven-v3 only — ISO 639-1 code to force a language when the text is ambiguous or code-switched.'),
|
|
2004
|
+
.describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
|
|
1911
2005
|
speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
|
|
1912
2006
|
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
|
|
1913
2007
|
pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
|
|
@@ -1923,36 +2017,22 @@ export const generateAudio = {
|
|
|
1923
2017
|
.string()
|
|
1924
2018
|
.optional()
|
|
1925
2019
|
.describe('seed-audio only — ONE image asset to score what is in frame. MUTUALLY EXCLUSIVE with audioReferenceAssetIds.'),
|
|
1926
|
-
|
|
1927
|
-
customMode: z
|
|
1928
|
-
.boolean()
|
|
1929
|
-
.optional()
|
|
1930
|
-
.describe('suno only. false (default) = prompt is a ≤500-char DESCRIPTION and lyrics get written for you. true = style + title required, and prompt becomes the EXACT LYRICS, sung as written. Putting a description in the prompt while customMode=true wastes a full generation.'),
|
|
1931
|
-
instrumental: z.boolean().optional().describe('suno only — score with no vocals. Usually right for a film bed: an unasked-for vocal fights dialogue.'),
|
|
1932
|
-
style: z.string().optional().describe('suno only — genre + era + instrumentation + tempo ("90s trip-hop, dusty breakbeat, Rhodes, 85 bpm"). Required in customMode. Steer here, not by piling adjectives into the prompt.'),
|
|
1933
|
-
title: z.string().optional().describe('suno only — track title. Required in customMode.'),
|
|
1934
|
-
negativeTags: z.string().optional().describe('suno only — comma-separated things to keep out ("brass, EDM drop, male vocal").'),
|
|
1935
|
-
vocalGender: z.enum(['m', 'f']).optional().describe('suno only — the WIRE values m/f, not "male"/"female".'),
|
|
1936
|
-
styleWeight: z.number().min(0).max(1).optional().describe('suno only — 0-1, how hard the track hugs `style`. Higher = more genre-obedient and more generic; lower = more room to surprise you.'),
|
|
1937
|
-
weirdnessConstraint: z.number().min(0).max(1).optional().describe('suno only — 0-1 experimentation dial. Low is safe and on-brief; high wanders. Leave unset for a film bed.'),
|
|
1938
|
-
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE + ' Recommended for suno (2-3 min renders).'),
|
|
2020
|
+
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
1939
2021
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate after explicit user OK.'),
|
|
1940
2022
|
}),
|
|
1941
2023
|
run: async (input, ctx) => {
|
|
1942
2024
|
// ── Per-surface clarification + constraint gates ──
|
|
1943
2025
|
const cfgDefaults = {
|
|
1944
|
-
'seed-audio':
|
|
1945
|
-
'eleven-sfx':
|
|
1946
|
-
'eleven-v3': undefined,
|
|
1947
|
-
suno: undefined,
|
|
2026
|
+
'seed-audio': SEED_AUDIO_DEFAULT_SECONDS,
|
|
2027
|
+
'eleven-sfx': ELEVEN_SFX_DEFAULT_SECONDS,
|
|
1948
2028
|
};
|
|
1949
2029
|
const seconds = input.durationSeconds ?? cfgDefaults[input.model];
|
|
1950
2030
|
if (input.model === 'seed-audio') {
|
|
1951
|
-
if (seconds
|
|
2031
|
+
if (seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
|
|
1952
2032
|
return ok({
|
|
1953
2033
|
requires_clarification: true,
|
|
1954
2034
|
missing: ['durationSeconds'],
|
|
1955
|
-
message: `Seed Audio needs a durationSeconds of
|
|
2035
|
+
message: `Seed Audio needs a durationSeconds of ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS}. It has NO duration parameter — the number is written into the prompt AND is what the user is billed, so it must be a deliberate choice. Ask the user how long the bed should be (a few seconds longer than the clip it sits under, so the edit has handles).`,
|
|
1956
2036
|
});
|
|
1957
2037
|
}
|
|
1958
2038
|
if ((input.audioReferenceAssetIds?.length ?? 0) > 0 && input.imageReferenceAssetId) {
|
|
@@ -1960,29 +2040,17 @@ export const generateAudio = {
|
|
|
1960
2040
|
}
|
|
1961
2041
|
}
|
|
1962
2042
|
if (input.model === 'eleven-sfx') {
|
|
1963
|
-
if (seconds
|
|
2043
|
+
if (seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
|
|
1964
2044
|
return ok({
|
|
1965
2045
|
requires_clarification: true,
|
|
1966
2046
|
missing: ['durationSeconds'],
|
|
1967
|
-
message: `Sound Effects needs a durationSeconds of
|
|
2047
|
+
message: `Sound Effects needs a durationSeconds of ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. It is billed per second and is never left for the model to pick (that would make the charge non-deterministic). Roughly: 0.5-1s for an impact, 2-4s for a whoosh, 8-22s for a loopable bed.`,
|
|
1968
2048
|
});
|
|
1969
2049
|
}
|
|
1970
2050
|
if (input.prompt.length > 450) {
|
|
1971
2051
|
throw new Error(`Sound Effects accepts up to 450 characters — this prompt is ${input.prompt.length}.`);
|
|
1972
2052
|
}
|
|
1973
2053
|
}
|
|
1974
|
-
if (input.model === 'eleven-v3' && input.prompt.length > ELEVEN_V3_MAX_CHARACTERS) {
|
|
1975
|
-
throw new Error(`Eleven v3 accepts up to ${ELEVEN_V3_MAX_CHARACTERS} characters — this script is ${input.prompt.length}.`);
|
|
1976
|
-
}
|
|
1977
|
-
if (input.model === 'suno' && input.customMode === true) {
|
|
1978
|
-
if (!input.style || !input.title) {
|
|
1979
|
-
return ok({
|
|
1980
|
-
requires_clarification: true,
|
|
1981
|
-
missing: [...(input.style ? [] : ['style']), ...(input.title ? [] : ['title'])],
|
|
1982
|
-
message: 'Suno custom mode requires style and title. Remember that in custom mode the prompt field is the EXACT LYRICS (unless instrumental=true, where it is ignored) — if you meant to describe a mood, use customMode=false instead.',
|
|
1983
|
-
});
|
|
1984
|
-
}
|
|
1985
|
-
}
|
|
1986
2054
|
await ctx.desktop().requireCapability('audio-generation', 'audio generation');
|
|
1987
2055
|
// ── Resolve asset refs at CALL time (UUIDs or badge codes) ──
|
|
1988
2056
|
const refInputs = [];
|
|
@@ -2002,7 +2070,6 @@ export const generateAudio = {
|
|
|
2002
2070
|
const costKey = audioCostKey({
|
|
2003
2071
|
model: input.model,
|
|
2004
2072
|
durationSeconds: seconds,
|
|
2005
|
-
characters: input.prompt.length,
|
|
2006
2073
|
});
|
|
2007
2074
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
2008
2075
|
if (!entry) {
|
|
@@ -2020,7 +2087,6 @@ export const generateAudio = {
|
|
|
2020
2087
|
(input.model === 'seed-audio'
|
|
2021
2088
|
? `You are billed for the ${seconds}s you requested regardless of the returned length. `
|
|
2022
2089
|
: '') +
|
|
2023
|
-
(input.model === 'suno' ? 'This returns TWO songs for that one price. ' : '') +
|
|
2024
2090
|
'Confirm with the user, then call again with confirm: true.',
|
|
2025
2091
|
});
|
|
2026
2092
|
}
|
|
@@ -2036,8 +2102,6 @@ export const generateAudio = {
|
|
|
2036
2102
|
prompt: input.prompt,
|
|
2037
2103
|
durationSeconds: seconds,
|
|
2038
2104
|
voice: input.voice,
|
|
2039
|
-
stability: input.stability,
|
|
2040
|
-
languageCode: input.languageCode,
|
|
2041
2105
|
speed: input.speed,
|
|
2042
2106
|
volume: input.volume,
|
|
2043
2107
|
pitch: input.pitch,
|
|
@@ -2046,15 +2110,6 @@ export const generateAudio = {
|
|
|
2046
2110
|
promptInfluence: input.promptInfluence,
|
|
2047
2111
|
audioReferenceAssetIds: (input.audioReferenceAssetIds ?? []).map((r) => rid(r)),
|
|
2048
2112
|
imageReferenceAssetId: rid(input.imageReferenceAssetId),
|
|
2049
|
-
sunoModel: input.sunoModel,
|
|
2050
|
-
customMode: input.customMode,
|
|
2051
|
-
instrumental: input.instrumental,
|
|
2052
|
-
style: input.style,
|
|
2053
|
-
title: input.title,
|
|
2054
|
-
negativeTags: input.negativeTags,
|
|
2055
|
-
vocalGender: input.vocalGender,
|
|
2056
|
-
styleWeight: input.styleWeight,
|
|
2057
|
-
weirdnessConstraint: input.weirdnessConstraint,
|
|
2058
2113
|
background: input.background,
|
|
2059
2114
|
});
|
|
2060
2115
|
if (!result.success)
|
|
@@ -2062,10 +2117,8 @@ export const generateAudio = {
|
|
|
2062
2117
|
if (result.background) {
|
|
2063
2118
|
return backgroundSubmitted(`${input.model} audio generation`, [result.generationId].filter(Boolean), { model: input.model, variant: costKey, projectId: input.projectId, cost_credits: totalCents }, refEcho);
|
|
2064
2119
|
}
|
|
2065
|
-
const siblings = result.siblingAssets ?? [];
|
|
2066
2120
|
return {
|
|
2067
2121
|
text: `Generated ${input.model} audio into project ${input.projectId} for ${fmtCredits(totalCents)}` +
|
|
2068
|
-
(siblings.length > 0 ? ` (${siblings.length + 1} tracks — Suno returns two variations)` : '') +
|
|
2069
2122
|
`. Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
|
|
2070
2123
|
(refEcho ? ` ${refEcho}` : ''),
|
|
2071
2124
|
data: {
|
|
@@ -2076,7 +2129,6 @@ export const generateAudio = {
|
|
|
2076
2129
|
cost_cents: totalCents,
|
|
2077
2130
|
cost_credits: totalCents,
|
|
2078
2131
|
asset: result.asset,
|
|
2079
|
-
...(siblings.length > 0 ? { siblingAssets: siblings } : {}),
|
|
2080
2132
|
generationId: result.generationId,
|
|
2081
2133
|
},
|
|
2082
2134
|
};
|
|
@@ -2085,27 +2137,19 @@ export const generateAudio = {
|
|
|
2085
2137
|
// ── Generate lip-sync ───────────────────────────────────────────
|
|
2086
2138
|
export const generateLipSync = {
|
|
2087
2139
|
id: 'slates_generate_lip_sync',
|
|
2088
|
-
description: 'Lip-sync a still image (avatar) or a video clip to audio.
|
|
2140
|
+
description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip (~$0.11 / 5s); sourceType=image animates a still avatar (avatar-standard ~$0.42 / 5s; avatar-pro ~$0.86 / 5s). Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
|
|
2089
2141
|
input: z.object({
|
|
2090
2142
|
projectId: z.string().uuid().describe('Slates project the source asset lives in. The new lip-synced video lands here.'),
|
|
2091
2143
|
sourceAssetId: z.string().uuid().describe('Asset id of the still image (avatar flow) or video clip (lip-sync flow). Must already exist in the project — use slates_upload_reference_image or slates_generate_image / slates_generate_video first if needed.'),
|
|
2092
2144
|
sourceType: z.enum(['image', 'video']).describe('"image" = animate a still portrait (avatar). "video" = re-sync an existing talking-head clip. Determines pricing — be deliberate.'),
|
|
2093
|
-
audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText
|
|
2145
|
+
audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText. "upload" = use the file at audioFilePath (absolute path on the user\'s machine).'),
|
|
2094
2146
|
ttsText: z.string().min(1).max(2000).optional().describe('Required when audioMethod=tts. The exact words the avatar/clip will speak.'),
|
|
2095
|
-
ttsVoice: z.string().optional().describe('
|
|
2096
|
-
ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('
|
|
2097
|
-
ttsSpeed: z.number().min(0.5).max(2).optional().describe('
|
|
2147
|
+
ttsVoice: z.string().optional().describe('Voice id (e.g. "oversea_male1"). See slates-prompting-lip-sync skill for the voice catalog.'),
|
|
2148
|
+
ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('TTS language. Default EN.'),
|
|
2149
|
+
ttsSpeed: z.number().min(0.5).max(2).optional().describe('TTS speech rate. Default 1.0. Range 0.5-2.0.'),
|
|
2098
2150
|
audioFilePath: z.string().optional().describe('Required when audioMethod=upload. Absolute path to the audio file on the user\'s machine (mp3, wav, m4a).'),
|
|
2099
|
-
avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('
|
|
2100
|
-
klingProvider: z.enum(['fal', 'kling']).optional().describe('
|
|
2101
|
-
engine: z.enum(['kling', 'seedance-2']).optional().describe('Default kling (cheap utility). seedance-2 = premium single-pass: natural speech generated in the video, voice cloned from a video source, audio included. Credits only.'),
|
|
2102
|
-
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
|
|
2103
|
-
aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
|
|
2104
|
-
seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the source (default TRUE for lip-sync; the faceless route would reject it). Bills the -face key.'),
|
|
2105
|
-
seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the source shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
|
|
2106
|
-
realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
|
|
2107
|
-
sourceSeconds: z.number().optional().describe('Seedance engine + sourceType=video: the source clip\'s duration in seconds (from the asset listing). Feeds the vref cost key (input+output billing).'),
|
|
2108
|
-
audioSeconds: z.number().optional().describe('Seedance engine + audioMethod=upload: the audio file\'s duration in seconds — sets the output length (4-15s).'),
|
|
2151
|
+
avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('Image-source only. avatar-standard (~14 credits/5s) for general use. avatar-pro (~29 credits/5s) for sharper face fidelity.'),
|
|
2152
|
+
klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. Leave unset: all agent generations bill Slates credits (BYOK is retired).'),
|
|
2109
2153
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
2110
2154
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required for avatar-pro.'),
|
|
2111
2155
|
}),
|
|
@@ -2124,41 +2168,8 @@ export const generateLipSync = {
|
|
|
2124
2168
|
message: 'audioMethod=upload requires audioFilePath. Pass an absolute path to the audio file on the user\'s machine.',
|
|
2125
2169
|
});
|
|
2126
2170
|
}
|
|
2127
|
-
const isSeedance = input.engine === 'seedance-2';
|
|
2128
2171
|
let costKey;
|
|
2129
|
-
|
|
2130
|
-
if (isSeedance) {
|
|
2131
|
-
// Consent gate before any spend, mirroring slates_generate_video.
|
|
2132
|
-
if (input.seedanceRealFace && !input.realFaceConsent) {
|
|
2133
|
-
return ok({
|
|
2134
|
-
requires_clarification: true,
|
|
2135
|
-
missing: ['realFaceConsent'],
|
|
2136
|
-
message: 'Real-person lip-sync needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
|
|
2137
|
-
});
|
|
2138
|
-
}
|
|
2139
|
-
if (input.sourceType === 'video' && !input.sourceSeconds) {
|
|
2140
|
-
return ok({
|
|
2141
|
-
requires_clarification: true,
|
|
2142
|
-
missing: ['sourceSeconds'],
|
|
2143
|
-
message: 'Seedance lip-sync on a video source bills combined input+output seconds. Pass sourceSeconds (the clip\'s duration from slates_list_assets, must be 2-15s).',
|
|
2144
|
-
});
|
|
2145
|
-
}
|
|
2146
|
-
const clamp = (n) => Math.min(15, Math.max(4, Math.ceil(n)));
|
|
2147
|
-
seedanceDuration = clamp(input.sourceType === 'video' && input.sourceSeconds
|
|
2148
|
-
? input.sourceSeconds
|
|
2149
|
-
: input.audioSeconds ?? (input.ttsText ? input.ttsText.length / 13 : 5));
|
|
2150
|
-
costKey = videoCostKey({
|
|
2151
|
-
model: 'seedance-2',
|
|
2152
|
-
duration: seedanceDuration,
|
|
2153
|
-
videoResolution: input.videoResolution ?? '1080p',
|
|
2154
|
-
// Lip-sync sources are faces by definition — face route unless
|
|
2155
|
-
// explicitly disabled or escalated to realface.
|
|
2156
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
2157
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
2158
|
-
videoRefSeconds: input.sourceType === 'video' ? input.sourceSeconds ?? 0 : 0,
|
|
2159
|
-
});
|
|
2160
|
-
}
|
|
2161
|
-
else if (input.sourceType === 'video') {
|
|
2172
|
+
if (input.sourceType === 'video') {
|
|
2162
2173
|
costKey = 'kling-lip-sync-video-5s';
|
|
2163
2174
|
}
|
|
2164
2175
|
else {
|
|
@@ -2187,7 +2198,7 @@ export const generateLipSync = {
|
|
|
2187
2198
|
estimated_cents: totalCents,
|
|
2188
2199
|
estimated_credits: totalCents,
|
|
2189
2200
|
source_ref: sourceRef,
|
|
2190
|
-
message: `Cost: ${fmtCredits(totalCents)} for
|
|
2201
|
+
message: `Cost: ${fmtCredits(totalCents)} for 5s lip-sync (${costKey}). ` +
|
|
2191
2202
|
`Source: ${sourceRef}. ${audioPreview}. ` +
|
|
2192
2203
|
`Re-call with confirm=true after the user explicitly OKs the spend. ` +
|
|
2193
2204
|
`When discussing with the user, refer to the source by its code (matches the gallery badge).`,
|
|
@@ -2210,28 +2221,12 @@ export const generateLipSync = {
|
|
|
2210
2221
|
klingProvider: input.klingProvider,
|
|
2211
2222
|
estimatedCost: totalCents,
|
|
2212
2223
|
background: input.background,
|
|
2213
|
-
// Seedance engine passthrough — the desktop delegates to the seedance
|
|
2214
|
-
// ref-to-video path (vref billing, face cascade, consent gate). The
|
|
2215
|
-
// durations ride along so the desktop bills exactly what was quoted.
|
|
2216
|
-
...(isSeedance
|
|
2217
|
-
? {
|
|
2218
|
-
lipSyncEngine: 'seedance-2',
|
|
2219
|
-
duration: seedanceDuration,
|
|
2220
|
-
videoResolution: input.videoResolution,
|
|
2221
|
-
aspectRatio: input.aspectRatio,
|
|
2222
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
2223
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
2224
|
-
realFaceConsent: input.realFaceConsent === true,
|
|
2225
|
-
sourceDurationSeconds: input.sourceSeconds,
|
|
2226
|
-
audioDurationSeconds: input.audioSeconds,
|
|
2227
|
-
}
|
|
2228
|
-
: {}),
|
|
2229
2224
|
});
|
|
2230
2225
|
if (!result.success)
|
|
2231
2226
|
throw new Error(result.error ?? 'Lip-sync generation failed');
|
|
2232
2227
|
if (result.background) {
|
|
2233
2228
|
const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
|
|
2234
|
-
return backgroundSubmitted(
|
|
2229
|
+
return backgroundSubmitted(`5s lip-sync (${costKey})`, ids, {
|
|
2235
2230
|
variant: costKey,
|
|
2236
2231
|
projectId: input.projectId,
|
|
2237
2232
|
sourceAssetId: input.sourceAssetId,
|
|
@@ -2240,7 +2235,7 @@ export const generateLipSync = {
|
|
|
2240
2235
|
});
|
|
2241
2236
|
}
|
|
2242
2237
|
return {
|
|
2243
|
-
text: `Generated
|
|
2238
|
+
text: `Generated 5s lip-sync (${costKey}) into project ${input.projectId} ` +
|
|
2244
2239
|
`for ${fmtCredits(totalCents)}. ` +
|
|
2245
2240
|
(input.audioMethod === 'tts'
|
|
2246
2241
|
? `Spoken: "${(input.ttsText ?? '').slice(0, 60)}${(input.ttsText ?? '').length > 60 ? '...' : ''}"`
|
|
@@ -2261,58 +2256,21 @@ export const generateLipSync = {
|
|
|
2261
2256
|
// ── Generate motion transfer ────────────────────────────────────
|
|
2262
2257
|
export const generateMotionTransfer = {
|
|
2263
2258
|
id: 'slates_generate_motion_transfer',
|
|
2264
|
-
description: 'Transfer the motion from a reference video onto a target image character.
|
|
2259
|
+
description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std ($0.95 / 5s) or kling-mc-pro ($1.26 / 5s), structured skeleton/depth retargeting, always 5s. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. Both tiers hit the >$0.50 confirm gate.',
|
|
2265
2260
|
input: z.object({
|
|
2266
2261
|
projectId: z.string().uuid().describe('Slates project. Both source and target assets must live here.'),
|
|
2267
|
-
sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project.
|
|
2262
|
+
sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s.'),
|
|
2268
2263
|
targetImageAssetId: z.string().uuid().describe('Asset id of the target image (the character that will perform the motion). Must already exist in the project.'),
|
|
2269
|
-
motionModel: z.enum(['kling-mc-std', 'kling-mc-pro'
|
|
2270
|
-
characterOrientation: z.enum(['video', 'image']).optional().describe('
|
|
2271
|
-
prompt: z.string().optional().describe('
|
|
2272
|
-
klingProvider: z.enum(['fal', 'kling']).optional().describe('
|
|
2273
|
-
duration: z.number().int().min(4).max(15).optional().describe('Seedance engine only — output duration in seconds (4-15). Defaults to the driving clip\'s length.'),
|
|
2274
|
-
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
|
|
2275
|
-
aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
|
|
2276
|
-
seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the clip/image (default TRUE for motion transfer). Bills the -face key.'),
|
|
2277
|
-
seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the driving clip/subject shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
|
|
2278
|
-
realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
|
|
2279
|
-
sourceVideoSeconds: z.number().optional().describe('Seedance engine: the driving clip\'s duration in seconds (from the asset listing, 2-15s). Feeds the vref cost key (input+output billing).'),
|
|
2264
|
+
motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std (~32 credits) general motion; kling-mc-pro (~42 credits) cleaner anatomy — default.'),
|
|
2265
|
+
characterOrientation: z.enum(['video', 'image']).optional().describe('"video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
|
|
2266
|
+
prompt: z.string().optional().describe('Optional refinement. Read slates-prompting-motion-transfer.'),
|
|
2267
|
+
klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. "fal" (default) uses Slates credits.'),
|
|
2280
2268
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
2281
2269
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required — both tiers exceed.'),
|
|
2282
2270
|
}),
|
|
2283
2271
|
async run(input, ctx) {
|
|
2284
2272
|
const motionModel = input.motionModel ?? 'kling-mc-pro';
|
|
2285
|
-
const
|
|
2286
|
-
let costKey;
|
|
2287
|
-
let seedanceDuration = 0;
|
|
2288
|
-
if (isSeedance) {
|
|
2289
|
-
if (input.seedanceRealFace && !input.realFaceConsent) {
|
|
2290
|
-
return ok({
|
|
2291
|
-
requires_clarification: true,
|
|
2292
|
-
missing: ['realFaceConsent'],
|
|
2293
|
-
message: 'Real-person motion transfer needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
|
|
2294
|
-
});
|
|
2295
|
-
}
|
|
2296
|
-
if (!input.sourceVideoSeconds) {
|
|
2297
|
-
return ok({
|
|
2298
|
-
requires_clarification: true,
|
|
2299
|
-
missing: ['sourceVideoSeconds'],
|
|
2300
|
-
message: 'Seedance motion transfer bills combined input+output seconds. Pass sourceVideoSeconds (the driving clip\'s duration from slates_list_assets, must be 2-15s).',
|
|
2301
|
-
});
|
|
2302
|
-
}
|
|
2303
|
-
seedanceDuration = input.duration ?? Math.min(15, Math.max(4, Math.ceil(input.sourceVideoSeconds)));
|
|
2304
|
-
costKey = videoCostKey({
|
|
2305
|
-
model: 'seedance-2',
|
|
2306
|
-
duration: seedanceDuration,
|
|
2307
|
-
videoResolution: input.videoResolution ?? '1080p',
|
|
2308
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
2309
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
2310
|
-
videoRefSeconds: input.sourceVideoSeconds,
|
|
2311
|
-
});
|
|
2312
|
-
}
|
|
2313
|
-
else {
|
|
2314
|
-
costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
|
|
2315
|
-
}
|
|
2273
|
+
const costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
|
|
2316
2274
|
const cloud = ctx.cloud();
|
|
2317
2275
|
const registry = await cloud.get('/api/agent/models');
|
|
2318
2276
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -2336,9 +2294,9 @@ export const generateMotionTransfer = {
|
|
|
2336
2294
|
estimated_credits: totalCents,
|
|
2337
2295
|
source_ref: source,
|
|
2338
2296
|
target_ref: target,
|
|
2339
|
-
message: `Cost: ${fmtCredits(totalCents)} for
|
|
2297
|
+
message: `Cost: ${fmtCredits(totalCents)} for 5s ${motionModel} (${costKey}). ` +
|
|
2340
2298
|
`Transferring motion from ${source} onto ${target}. ` +
|
|
2341
|
-
`Re-call with confirm=true after the user explicitly OKs the spend
|
|
2299
|
+
`Re-call with confirm=true after the user explicitly OKs the spend, or pick kling-mc-std to save ~10 credits. ` +
|
|
2342
2300
|
`When discussing with the user, refer to the assets by those codes — they'll match the gallery badges.`,
|
|
2343
2301
|
});
|
|
2344
2302
|
}
|
|
@@ -2356,27 +2314,12 @@ export const generateMotionTransfer = {
|
|
|
2356
2314
|
klingProvider: input.klingProvider,
|
|
2357
2315
|
estimatedCost: totalCents,
|
|
2358
2316
|
background: input.background,
|
|
2359
|
-
// Seedance engine passthrough — the desktop delegates to the seedance
|
|
2360
|
-
// ref-to-video path (vref billing, face cascade, consent gate).
|
|
2361
|
-
...(isSeedance
|
|
2362
|
-
? {
|
|
2363
|
-
duration: seedanceDuration,
|
|
2364
|
-
videoResolution: input.videoResolution,
|
|
2365
|
-
aspectRatio: input.aspectRatio,
|
|
2366
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
2367
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
2368
|
-
realFaceConsent: input.realFaceConsent === true,
|
|
2369
|
-
// Ride the caller-supplied clip duration through — the asset row's
|
|
2370
|
-
// duration can be null for imported clips.
|
|
2371
|
-
sourceVideoDurationSeconds: input.sourceVideoSeconds,
|
|
2372
|
-
}
|
|
2373
|
-
: {}),
|
|
2374
2317
|
});
|
|
2375
2318
|
if (!result.success)
|
|
2376
2319
|
throw new Error(result.error ?? 'Motion transfer generation failed');
|
|
2377
2320
|
if (result.background) {
|
|
2378
2321
|
const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
|
|
2379
|
-
return backgroundSubmitted(
|
|
2322
|
+
return backgroundSubmitted(`5s motion transfer (${motionModel})`, ids, {
|
|
2380
2323
|
variant: costKey,
|
|
2381
2324
|
motionModel,
|
|
2382
2325
|
projectId: input.projectId,
|
|
@@ -2387,7 +2330,7 @@ export const generateMotionTransfer = {
|
|
|
2387
2330
|
});
|
|
2388
2331
|
}
|
|
2389
2332
|
return {
|
|
2390
|
-
text: `Generated
|
|
2333
|
+
text: `Generated 5s motion transfer (${motionModel}) into project ${input.projectId} ` +
|
|
2391
2334
|
`for ${fmtCredits(totalCents)}.` +
|
|
2392
2335
|
(input.prompt ? ` Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` : ''),
|
|
2393
2336
|
data: {
|
|
@@ -2407,15 +2350,17 @@ export const generateMotionTransfer = {
|
|
|
2407
2350
|
// ── Edit video (Kling O3 video-to-video) ────────────────────────
|
|
2408
2351
|
export const editVideo = {
|
|
2409
2352
|
id: 'slates_edit_video',
|
|
2410
|
-
description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements)
|
|
2353
|
+
description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat), or seedance-2.5-edit (4–30s clips — the ONLY engine that takes a clip over 15s; 480p/720p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds — always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
|
|
2411
2354
|
input: z.object({
|
|
2412
2355
|
projectId: z.string().uuid().describe('Project the source clip lives in.'),
|
|
2413
2356
|
sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
|
|
2414
2357
|
prompt: z.string().min(1).max(2500).describe('The change, not the whole scene — e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
|
|
2415
|
-
model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs.'),
|
|
2358
|
+
model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit', 'seedance-2.5-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs. seedance-2.5-edit (480p/720p, 4–30s) is the ONLY engine that accepts a clip longer than 15s; it also takes NO character/style refs on this op.'),
|
|
2416
2359
|
characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2417
2360
|
styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2418
|
-
keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit output
|
|
2361
|
+
keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
|
|
2362
|
+
videoResolution: z.enum(['480p', '720p']).optional().describe('seedance-2.5-edit ONLY (default 720p). Ignored by the Kling and Omni Flash engines, whose output follows the source clip.'),
|
|
2363
|
+
seedanceFace: z.boolean().optional().describe('seedance-2.5-edit ONLY: set true when a CHARACTER FACE is visible in the clip. Faceless edits run on BytePlus; faces are blocked there and must route to the relaxed provider, which costs ~35% more. A face edit submitted without this flag is rejected by the provider, not silently downgraded.'),
|
|
2419
2364
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
2420
2365
|
confirm: z.boolean().optional().describe('Set true to bypass the cost confirm gate after the user OKs the spend.'),
|
|
2421
2366
|
}),
|
|
@@ -2427,10 +2372,13 @@ export const editVideo = {
|
|
|
2427
2372
|
}
|
|
2428
2373
|
const model = input.model ?? 'kling-v3.0-omni-edit';
|
|
2429
2374
|
const isOmniFlashEdit = model === 'omni-flash-edit';
|
|
2430
|
-
|
|
2431
|
-
|
|
2432
|
-
|
|
2433
|
-
|
|
2375
|
+
const isSeedanceEdit = model === 'seedance-2.5-edit';
|
|
2376
|
+
// Kling edit: 3–15s source clips; Omni Flash edit: 3–10s; Seedance 2.5
|
|
2377
|
+
// edit: 4–30s — the only engine that takes a clip over 15s.
|
|
2378
|
+
const minClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MIN_SECONDS : 3;
|
|
2379
|
+
const maxClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MAX_SECONDS : isOmniFlashEdit ? 10 : 15;
|
|
2380
|
+
if ((isOmniFlashEdit || isSeedanceEdit) && ((input.characterAssetIds?.length ?? 0) > 0 || (input.styleAssetIds?.length ?? 0) > 0)) {
|
|
2381
|
+
throw new Error(`${model} takes the prompt and the source clip only on this op — no character/style reference images. Drop the refs, or switch to kling-v3.0-omni-edit which supports elements.`);
|
|
2434
2382
|
}
|
|
2435
2383
|
// Resolve refs (UUIDs or badge codes) against the project AT CALL TIME.
|
|
2436
2384
|
const refInputs = [
|
|
@@ -2459,12 +2407,24 @@ export const editVideo = {
|
|
|
2459
2407
|
if (!Number.isFinite(clipSeconds) || clipSeconds <= 0) {
|
|
2460
2408
|
throw new Error('Source clip has no recorded duration — cannot quote the edit. Re-import the clip or pick another.');
|
|
2461
2409
|
}
|
|
2462
|
-
if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds <
|
|
2463
|
-
throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts
|
|
2464
|
-
`Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip
|
|
2465
|
-
|
|
2466
|
-
|
|
2467
|
-
|
|
2410
|
+
if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds < minClipSeconds - 0.05) {
|
|
2411
|
+
throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts ${minClipSeconds}–${maxClipSeconds}s. ` +
|
|
2412
|
+
`Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip` +
|
|
2413
|
+
(maxClipSeconds < SEEDANCE_25_EDIT_MAX_SECONDS ? ', or switch to seedance-2.5-edit which accepts up to 30s' : '') +
|
|
2414
|
+
'.');
|
|
2415
|
+
}
|
|
2416
|
+
const billedSeconds = Math.min(maxClipSeconds, Math.max(minClipSeconds, Math.ceil(clipSeconds - 0.05)));
|
|
2417
|
+
// Three engines, three key shapes — only Seedance's carries a resolution and
|
|
2418
|
+
// a face route, because only its price moves with them.
|
|
2419
|
+
const costKey = isSeedanceEdit
|
|
2420
|
+
? seedanceEditCostKey({
|
|
2421
|
+
duration: billedSeconds,
|
|
2422
|
+
videoResolution: input.videoResolution ?? '720p',
|
|
2423
|
+
seedanceFace: input.seedanceFace === true,
|
|
2424
|
+
})
|
|
2425
|
+
: isOmniFlashEdit
|
|
2426
|
+
? omniFlashEditCostKey(billedSeconds)
|
|
2427
|
+
: klingEditCostKey(model, billedSeconds);
|
|
2468
2428
|
const cloud = ctx.cloud();
|
|
2469
2429
|
const registry = await cloud.get('/api/agent/models');
|
|
2470
2430
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -2503,6 +2463,8 @@ export const editVideo = {
|
|
|
2503
2463
|
characterAssetIds,
|
|
2504
2464
|
styleAssetIds,
|
|
2505
2465
|
keepAudio: input.keepAudio !== false,
|
|
2466
|
+
videoResolution: input.videoResolution,
|
|
2467
|
+
seedanceFace: input.seedanceFace,
|
|
2506
2468
|
background: input.background,
|
|
2507
2469
|
});
|
|
2508
2470
|
if (!result.success)
|
|
@@ -3105,6 +3067,53 @@ export const updateFrame = {
|
|
|
3105
3067
|
}));
|
|
3106
3068
|
},
|
|
3107
3069
|
};
|
|
3070
|
+
/**
|
|
3071
|
+
* Batch form of `slates_update_frame`.
|
|
3072
|
+
*
|
|
3073
|
+
* Writing a scene's worth of frames — motion prompts, shot labels — is ONE
|
|
3074
|
+
* logical edit. Doing it as N `slates_update_frame` calls costs N LLM
|
|
3075
|
+
* round-trips and makes the desktop refetch the whole storyboard N times for a
|
|
3076
|
+
* single intent. This is also where the storyboard's deleted "generate motion
|
|
3077
|
+
* prompts" button's capability went: the agent can be told "redo scene 3,
|
|
3078
|
+
* handheld", which that fixed-shape button never could.
|
|
3079
|
+
*
|
|
3080
|
+
* Hits `POST /agent/frames/batch-update`, which validates every id BEFORE the
|
|
3081
|
+
* first write and emits one broadcast for the batch.
|
|
3082
|
+
*/
|
|
3083
|
+
export const batchUpdateFrames = {
|
|
3084
|
+
id: 'slates_batch_update_frames',
|
|
3085
|
+
description: 'Update MANY frames in one call — motion prompts, shot labels, notes, asset binding, scene/position, frameType. Prefer this over repeated slates_update_frame when writing a scene or a whole storyboard: it is one round-trip and one UI refresh. Every id is validated before anything is written, so the batch never lands half-applied.',
|
|
3086
|
+
input: z.object({
|
|
3087
|
+
updates: z
|
|
3088
|
+
.array(z.object({
|
|
3089
|
+
frameId: z.string().uuid(),
|
|
3090
|
+
shotLabel: z.string().optional(),
|
|
3091
|
+
notes: z.string().optional(),
|
|
3092
|
+
assetId: z.string().uuid().nullable().optional(),
|
|
3093
|
+
sceneId: z.string().uuid().nullable().optional(),
|
|
3094
|
+
position: z.number().int().min(0).optional(),
|
|
3095
|
+
frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
|
|
3096
|
+
motionPrompt: z.string().nullable().optional(),
|
|
3097
|
+
}))
|
|
3098
|
+
.min(1),
|
|
3099
|
+
}),
|
|
3100
|
+
async run(input, ctx) {
|
|
3101
|
+
return ok(await ctx.desktop().post('/agent/frames/batch-update', {
|
|
3102
|
+
updates: input.updates.map((u) => ({
|
|
3103
|
+
id: u.frameId,
|
|
3104
|
+
data: {
|
|
3105
|
+
shotLabel: u.shotLabel,
|
|
3106
|
+
notes: u.notes,
|
|
3107
|
+
assetId: u.assetId,
|
|
3108
|
+
sceneId: u.sceneId,
|
|
3109
|
+
position: u.position,
|
|
3110
|
+
frameType: u.frameType,
|
|
3111
|
+
motionPrompt: u.motionPrompt,
|
|
3112
|
+
},
|
|
3113
|
+
})),
|
|
3114
|
+
}));
|
|
3115
|
+
},
|
|
3116
|
+
};
|
|
3108
3117
|
export const deleteFrame = {
|
|
3109
3118
|
id: 'slates_delete_frame',
|
|
3110
3119
|
description: 'Delete a frame from its scene (the referenced asset is untouched).',
|
|
@@ -3152,13 +3161,30 @@ function resolveGuideTopic(topic) {
|
|
|
3152
3161
|
// Audio — seed-audio BEFORE the seedance check: "seed-audio" also starts
|
|
3153
3162
|
// with "seed", and falling through would hand the video guide to the audio
|
|
3154
3163
|
// model (the exact class of aliasing bug this comment block warns about).
|
|
3155
|
-
|
|
3164
|
+
// Speech, dialogue and scratch VO all live on Seed Audio now — the TTS
|
|
3165
|
+
// surface is gone, so "tts"/"voiceover" must NOT land on the ElevenLabs
|
|
3166
|
+
// guide, which is SFX-only.
|
|
3167
|
+
if (t.startsWith('seed-audio') ||
|
|
3168
|
+
t === 'seed audio' ||
|
|
3169
|
+
t === 'audio' ||
|
|
3170
|
+
t === 'tts' ||
|
|
3171
|
+
t === 'voiceover' ||
|
|
3172
|
+
t === 'dialogue') {
|
|
3156
3173
|
return 'slates-prompting-seed-audio';
|
|
3157
|
-
|
|
3174
|
+
}
|
|
3175
|
+
if (t.startsWith('eleven') || t.startsWith('elevenlabs') || t === 'sfx' || t === 'sound-effects' || t === 'sound effects') {
|
|
3158
3176
|
return 'slates-prompting-elevenlabs';
|
|
3159
3177
|
}
|
|
3160
|
-
|
|
3161
|
-
|
|
3178
|
+
// ⚠️ 2.5 BEFORE the generic `seedance` prefix — the same ordering trap as
|
|
3179
|
+
// seed-audio-before-seedance above. Falling through silently hands the 2.0
|
|
3180
|
+
// guide to 2.5, whose limits, resolutions and task types are all different.
|
|
3181
|
+
if (t.startsWith('seedance-2.5') ||
|
|
3182
|
+
t.startsWith('seedance-25') ||
|
|
3183
|
+
t.startsWith('seedance-2-5') ||
|
|
3184
|
+
t.startsWith('seedance2.5') ||
|
|
3185
|
+
t === 'seedance 2.5') {
|
|
3186
|
+
return 'slates-prompting-seedance-2-5';
|
|
3187
|
+
}
|
|
3162
3188
|
if (t.startsWith('seedance'))
|
|
3163
3189
|
return 'slates-prompting-seedance';
|
|
3164
3190
|
if (t.startsWith('avatar-') || t.includes('lip-sync'))
|
|
@@ -3185,7 +3211,7 @@ export const getPromptingGuide = {
|
|
|
3185
3211
|
topic: z
|
|
3186
3212
|
.string()
|
|
3187
3213
|
.min(1)
|
|
3188
|
-
.describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-
|
|
3214
|
+
.describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
|
|
3189
3215
|
}),
|
|
3190
3216
|
async run(input) {
|
|
3191
3217
|
const resolved = resolveGuideTopic(input.topic);
|
|
@@ -3274,6 +3300,7 @@ export const ALL_OPERATIONS = [
|
|
|
3274
3300
|
deleteScene,
|
|
3275
3301
|
reorderScenes,
|
|
3276
3302
|
updateFrame,
|
|
3303
|
+
batchUpdateFrames,
|
|
3277
3304
|
deleteFrame,
|
|
3278
3305
|
getPromptingGuide,
|
|
3279
3306
|
];
|