@slatesvideo/shared 0.5.5 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -2
- package/dist/index.js +6 -3
- package/dist/operations/index.d.ts +106 -19
- package/dist/operations/index.js +624 -172
- package/dist/prompts/character-sheet.d.ts +2 -1
- package/dist/prompts/character-sheet.js +82 -13
- package/dist/prompts/model-facts.d.ts +43 -1
- package/dist/prompts/model-facts.js +104 -2
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +208 -5
- package/dist/prompts/reference-composer.d.ts +21 -3
- package/dist/prompts/reference-composer.js +80 -10
- package/dist/prompts/reference-rules.d.ts +19 -2
- package/dist/prompts/reference-rules.js +18 -1
- package/dist/skills/content.js +8 -5
- package/exports/slates-prompt-builder/generated/SKILL.md +2 -2
- package/exports/slates-prompt-builder/generated/reference-character.md +9 -4
- package/exports/slates-prompt-builder/generated/reference-seedance.md +12 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +11 -11
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +1 -1
- package/skills/slates-character-identity.md +9 -4
- package/skills/slates-model-selection.md +39 -11
- package/skills/slates-prompting-elevenlabs.md +69 -0
- package/skills/slates-prompting-lip-sync.md +12 -14
- package/skills/slates-prompting-motion-transfer.md +18 -14
- package/skills/slates-prompting-seed-audio.md +110 -0
- package/skills/slates-prompting-seedance-2-5.md +215 -0
- package/skills/slates-prompting-seedance.md +17 -1
package/dist/operations/index.js
CHANGED
|
@@ -13,6 +13,9 @@ import { z } from 'zod';
|
|
|
13
13
|
import { SlatesCloudClient } from '../clients/cloud.js';
|
|
14
14
|
import { SlatesDesktopClient } from '../clients/desktop.js';
|
|
15
15
|
import { SKILLS } from '../skills/content.js';
|
|
16
|
+
// Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
|
|
17
|
+
// "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
|
|
18
|
+
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } from '../prompts/model-facts.js';
|
|
16
19
|
export function defaultContext() {
|
|
17
20
|
return {
|
|
18
21
|
cloud: () => new SlatesCloudClient(),
|
|
@@ -120,8 +123,11 @@ export const listAvailableModels = {
|
|
|
120
123
|
const table = models.map((m) => `${m.model} ${creditCost(m)}`).join('\n');
|
|
121
124
|
return {
|
|
122
125
|
text: `${models.length} COST keys (credits per generation)${input.filter ? ` matching "${input.filter}"` : ''}. ` +
|
|
123
|
-
`NOTE: these are billing keys for cost lookup ONLY — the \`model\` param on
|
|
124
|
-
|
|
126
|
+
`NOTE: these are billing keys for cost lookup ONLY — the \`model\` param on the generate ops takes a BASE id. ` +
|
|
127
|
+
// Derived from the SSOT arrays, not restated: a hand-written list here
|
|
128
|
+
// drifts the moment a model lands (it already omitted omni-flash).
|
|
129
|
+
`slates_generate_video: ${VIDEO_MODELS.join(' | ')} (duration/videoResolution as separate params). ` +
|
|
130
|
+
`slates_generate_audio: ${AUDIO_MODELS.join(' | ')} (durationSeconds as a separate param):\n` +
|
|
125
131
|
table,
|
|
126
132
|
data: { count: models.length },
|
|
127
133
|
};
|
|
@@ -133,7 +139,7 @@ export const estimateGenerationCost = {
|
|
|
133
139
|
input: z.object({
|
|
134
140
|
model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
|
|
135
141
|
quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
|
|
136
|
-
duration: z.number().int().min(
|
|
142
|
+
duration: z.number().int().min(1).max(360).optional().describe('Seconds. Video 3-15 (cost scales linearly; required with a video base id). Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22 — required with either audio base id.'),
|
|
137
143
|
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Video only. Seedance defaults to 1080p.'),
|
|
138
144
|
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
|
|
139
145
|
quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
|
|
@@ -152,6 +158,42 @@ export const estimateGenerationCost = {
|
|
|
152
158
|
if (img)
|
|
153
159
|
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
|
|
154
160
|
}
|
|
161
|
+
// 2a) audio base id → seconds. Both surfaces bill per second, so a
|
|
162
|
+
// duration is always required. Runs BEFORE the video resolver: it is
|
|
163
|
+
// forgiving by design and "seed-audio" would otherwise be mistaken for
|
|
164
|
+
// a seedance spelling.
|
|
165
|
+
if (!key && AUDIO_MODELS.includes(input.model)) {
|
|
166
|
+
const m = input.model;
|
|
167
|
+
if (!input.duration) {
|
|
168
|
+
return ok({
|
|
169
|
+
requires_clarification: true,
|
|
170
|
+
missing: ['duration'],
|
|
171
|
+
message: m === 'seed-audio'
|
|
172
|
+
? 'Seed Audio cost scales with the REQUESTED duration — and that duration is what the user is billed regardless of what comes back (it has no duration parameter; the number is written into the prompt). Pass duration in seconds (3-120).'
|
|
173
|
+
: 'Sound Effects bills per second — pass duration in seconds (1-22).',
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
// REFUSE an out-of-range duration rather than quoting the clamped price.
|
|
177
|
+
// audioCostKey clamps (it has to — it mirrors the desktop, which clamps),
|
|
178
|
+
// so without this an agent asking for 2s of Seed Audio would be handed a
|
|
179
|
+
// real 3s price and no hint that 2s is not a thing it can order. The
|
|
180
|
+
// generate op gates the same way; the two must agree or the quote is a
|
|
181
|
+
// promise the generation refuses to keep.
|
|
182
|
+
const audioBounds = {
|
|
183
|
+
'seed-audio': { min: SEED_AUDIO_MIN_SECONDS, max: SEED_AUDIO_MAX_SECONDS },
|
|
184
|
+
'eleven-sfx': { min: ELEVEN_SFX_MIN_SECONDS, max: ELEVEN_SFX_MAX_SECONDS },
|
|
185
|
+
};
|
|
186
|
+
const bounds = audioBounds[m];
|
|
187
|
+
if (input.duration < bounds.min || input.duration > bounds.max) {
|
|
188
|
+
return ok({
|
|
189
|
+
requires_clarification: true,
|
|
190
|
+
missing: ['duration'],
|
|
191
|
+
message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
|
|
192
|
+
'Re-ask with a duration in range.',
|
|
193
|
+
});
|
|
194
|
+
}
|
|
195
|
+
key = audioCostKey({ model: m, durationSeconds: input.duration });
|
|
196
|
+
}
|
|
155
197
|
// 2b) Kling O3 edit base id + duration (ceiled source-clip length)
|
|
156
198
|
if (!key && (input.model === 'kling-v3.0-omni-edit' || input.model === 'kling-v3.0-omni-pro-edit')) {
|
|
157
199
|
if (!input.duration) {
|
|
@@ -181,7 +223,12 @@ export const estimateGenerationCost = {
|
|
|
181
223
|
duration,
|
|
182
224
|
videoResolution: input.videoResolution ??
|
|
183
225
|
resolved.videoResolution ??
|
|
184
|
-
|
|
226
|
+
// Seedance quotes are resolution-scaled, so a missing resolution has
|
|
227
|
+
// to fall back to the model's OWN default — 2.5 has no 1080p at all,
|
|
228
|
+
// and a blanket '1080p' here quoted a key that does not exist.
|
|
229
|
+
(resolved.model === 'seedance-2.5' ? '720p'
|
|
230
|
+
: resolved.model.startsWith('seedance') ? '1080p'
|
|
231
|
+
: undefined),
|
|
185
232
|
sound: input.sound ?? resolved.sound,
|
|
186
233
|
seedanceFace: input.seedanceFace ?? resolved.seedanceFace,
|
|
187
234
|
seedanceRealFace: input.seedanceRealFace,
|
|
@@ -190,7 +237,7 @@ export const estimateGenerationCost = {
|
|
|
190
237
|
}
|
|
191
238
|
const perCredits = key != null ? byKey.get(key) : undefined;
|
|
192
239
|
if (key == null || perCredits == null) {
|
|
193
|
-
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
|
|
240
|
+
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
|
|
194
241
|
}
|
|
195
242
|
const qty = input.quantity ?? 1;
|
|
196
243
|
const totalCredits = perCredits * qty;
|
|
@@ -439,16 +486,16 @@ export const getAssetVideoFrames = {
|
|
|
439
486
|
};
|
|
440
487
|
export const uploadReferenceImage = {
|
|
441
488
|
id: 'slates_upload_reference_image',
|
|
442
|
-
description: 'Add a reference image
|
|
489
|
+
description: 'Add a reference image, video clip, or audio file to a Slates project from disk. Pass either filePath (absolute path to a local file) or dataUrl (base64 data: URL) — exactly one. Set type:"video" to bring in a clip (the user\'s own footage to edit/relocate/trim) or type:"audio" for music/VO/SFX they already have; both are probed on ingest, so duration (and for video, dimensions) are available immediately, and audio gets its waveform. Default type is "image". dataUrl is image-only.',
|
|
443
490
|
input: z
|
|
444
491
|
.object({
|
|
445
492
|
projectId: z.string().uuid(),
|
|
446
493
|
filePath: z.string().optional(),
|
|
447
494
|
dataUrl: z.string().optional(),
|
|
448
495
|
type: z
|
|
449
|
-
.enum(['image', 'video'])
|
|
496
|
+
.enum(['image', 'video', 'audio'])
|
|
450
497
|
.optional()
|
|
451
|
-
.describe('Asset kind for a filePath import — "image" (default) or "
|
|
498
|
+
.describe('Asset kind for a filePath import — "image" (default), "video", or "audio". A dataUrl is always an image.'),
|
|
452
499
|
})
|
|
453
500
|
.refine((d) => !!d.filePath !== !!d.dataUrl, {
|
|
454
501
|
message: 'Pass exactly one of filePath or dataUrl',
|
|
@@ -463,8 +510,8 @@ export const uploadReferenceImage = {
|
|
|
463
510
|
});
|
|
464
511
|
return ok(r);
|
|
465
512
|
}
|
|
466
|
-
if (input.type === 'video') {
|
|
467
|
-
throw new Error(
|
|
513
|
+
if (input.type === 'video' || input.type === 'audio') {
|
|
514
|
+
throw new Error(`dataUrl uploads are image-only — pass a filePath to import ${input.type === 'audio' ? 'an audio file' : 'a video clip'}.`);
|
|
468
515
|
}
|
|
469
516
|
const r = await desktop.post('/agent/assets/upload-base64', {
|
|
470
517
|
projectId: input.projectId,
|
|
@@ -506,6 +553,82 @@ export const moveAssetsToFolder = {
|
|
|
506
553
|
return ok(await ctx.desktop().post('/agent/folders/move-assets', input));
|
|
507
554
|
},
|
|
508
555
|
};
|
|
556
|
+
export const moveAssetsToProject = {
|
|
557
|
+
id: 'slates_move_assets_to_project',
|
|
558
|
+
description: 'Move assets (images, videos, audio) out of one project into another. The media files move on disk into the destination project folder — this is a real re-home, not a copy. Moved assets leave whatever gallery folder they were in and are issued fresh badge codes in the destination.',
|
|
559
|
+
input: z.object({
|
|
560
|
+
sourceProjectId: z.string().uuid(),
|
|
561
|
+
// Badge codes ("IMG-A8") resolve against sourceProjectId at call time.
|
|
562
|
+
assetIds: z.array(z.string().min(1)).min(1),
|
|
563
|
+
targetProjectId: z.string().uuid(),
|
|
564
|
+
}),
|
|
565
|
+
async run(input, ctx) {
|
|
566
|
+
if (input.sourceProjectId === input.targetProjectId) {
|
|
567
|
+
throw new Error('sourceProjectId and targetProjectId are the same — nothing to move.');
|
|
568
|
+
}
|
|
569
|
+
const resolved = await resolveAssetRefs(ctx, input.sourceProjectId, input.assetIds);
|
|
570
|
+
const assetIds = input.assetIds.map((ref) => resolved.get(ref)?.id ?? ref);
|
|
571
|
+
const result = await ctx.desktop().post('/agent/assets/move-to-project', {
|
|
572
|
+
assetIds,
|
|
573
|
+
targetProjectId: input.targetProjectId,
|
|
574
|
+
// Sent so the route can ENFORCE it. resolveAssetRefs only validates
|
|
575
|
+
// badge codes against the source project — a raw UUID passes straight
|
|
576
|
+
// through, so without this a caller could move an asset out of a
|
|
577
|
+
// project it never named.
|
|
578
|
+
sourceProjectId: input.sourceProjectId,
|
|
579
|
+
});
|
|
580
|
+
// Timeline clips key media by PATH, so the move repointed edits in OTHER
|
|
581
|
+
// projects at files that now live inside the destination — and deleting the
|
|
582
|
+
// destination deletes those files. Say it in the TEXT, not just the data:
|
|
583
|
+
// an agent summarising this result must be able to pass the warning on.
|
|
584
|
+
const refs = result.externalClipRefs ?? [];
|
|
585
|
+
if (refs.length > 0) {
|
|
586
|
+
const clips = refs.reduce((n, r) => n + r.clipCount, 0);
|
|
587
|
+
const where = refs.map((r) => `"${r.projectName}"`).join(', ');
|
|
588
|
+
return ok(result, `${JSON.stringify(result)}\n\n⚠️ ${clips} timeline clip(s) in ${where} now reference the moved ` +
|
|
589
|
+
`file(s) inside the destination project. Deleting the destination project would break those ` +
|
|
590
|
+
`edits. Tell the user — this is the one cross-project reference Slates allows.`);
|
|
591
|
+
}
|
|
592
|
+
return ok(result);
|
|
593
|
+
},
|
|
594
|
+
};
|
|
595
|
+
export const copyAssetsToProject = {
|
|
596
|
+
id: 'slates_copy_assets_to_project',
|
|
597
|
+
description: 'Copy assets (images, videos, audio) from one project into another. The originals stay exactly where they are — new files, new rows, new badge codes in the destination. Use this instead of slates_move_assets_to_project when the asset is already in use where it lives: an image that is a character/environment/style identity or sits in a storyboard frame CANNOT be moved out (that would leave the other project pointing at a file it no longer owns), but it can always be copied. Lineage is not copied; the copy starts clean.',
|
|
598
|
+
input: z.object({
|
|
599
|
+
sourceProjectId: z.string().uuid(),
|
|
600
|
+
// Badge codes ("IMG-A8") resolve against sourceProjectId at call time.
|
|
601
|
+
assetIds: z.array(z.string().min(1)).min(1),
|
|
602
|
+
targetProjectId: z.string().uuid(),
|
|
603
|
+
}),
|
|
604
|
+
async run(input, ctx) {
|
|
605
|
+
if (input.sourceProjectId === input.targetProjectId) {
|
|
606
|
+
throw new Error('sourceProjectId and targetProjectId are the same — nothing to copy.');
|
|
607
|
+
}
|
|
608
|
+
const resolved = await resolveAssetRefs(ctx, input.sourceProjectId, input.assetIds);
|
|
609
|
+
const assetIds = input.assetIds.map((ref) => resolved.get(ref)?.id ?? ref);
|
|
610
|
+
return ok(await ctx.desktop().post('/agent/assets/copy-to-project', {
|
|
611
|
+
assetIds,
|
|
612
|
+
targetProjectId: input.targetProjectId,
|
|
613
|
+
// Enforced route-side for the same reason as the move op: resolveAssetRefs
|
|
614
|
+
// only validates badge codes, so a raw UUID would otherwise let a caller
|
|
615
|
+
// copy out of a project it never named.
|
|
616
|
+
sourceProjectId: input.sourceProjectId,
|
|
617
|
+
}));
|
|
618
|
+
},
|
|
619
|
+
};
|
|
620
|
+
export const moveEntityToProject = {
|
|
621
|
+
id: 'slates_move_entity_to_project',
|
|
622
|
+
description: "Move a character, environment, or style into another project, taking every image it references with it. This is the fix when slates_move_assets_to_project refuses an asset because an entity still uses it: the identity image can't leave on its own, but the whole entity can. Refuses (rather than cascading) if one of its images is ALSO used by something else — copy the images instead in that case. Storyboard frames are not movable this way; moving a frame's image out would empty the shot.",
|
|
623
|
+
input: z.object({
|
|
624
|
+
kind: z.enum(['character', 'environment', 'style']),
|
|
625
|
+
entityId: z.string().uuid(),
|
|
626
|
+
targetProjectId: z.string().uuid(),
|
|
627
|
+
}),
|
|
628
|
+
async run(input, ctx) {
|
|
629
|
+
return ok(await ctx.desktop().post('/agent/entities/move-to-project', input));
|
|
630
|
+
},
|
|
631
|
+
};
|
|
509
632
|
// ── Characters ──────────────────────────────────────────────────
|
|
510
633
|
export const listCharacters = {
|
|
511
634
|
id: 'slates_list_characters',
|
|
@@ -1197,6 +1320,11 @@ export const VIDEO_MODELS = [
|
|
|
1197
1320
|
'veo-3.1-fast',
|
|
1198
1321
|
'veo-3.1-standard',
|
|
1199
1322
|
'seedance-2',
|
|
1323
|
+
// Seedance 2.5 is a SECOND SEAT, not a replacement: 30s takes, 30 image
|
|
1324
|
+
// references, audio-only references — and 480p/720p ONLY. 2.0 keeps the
|
|
1325
|
+
// ladder to native 4K and stays the default. Its EDIT row is not here; edit
|
|
1326
|
+
// models live on slates_edit_video, same as the Kling and Omni Flash ones.
|
|
1327
|
+
'seedance-2.5',
|
|
1200
1328
|
'omni-flash',
|
|
1201
1329
|
];
|
|
1202
1330
|
// Model → registry cost-key. Each provider's keys ship with their own
|
|
@@ -1220,16 +1348,24 @@ const KLING_TIER_MAP = {
|
|
|
1220
1348
|
// asserts this builder byte-matches the desktop's klingCreditKey/seedanceCreditKey.
|
|
1221
1349
|
export function videoCostKey(input) {
|
|
1222
1350
|
if (input.model.startsWith('seedance')) {
|
|
1223
|
-
// Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (
|
|
1224
|
-
// res × duration). AI-face route bills the `-face-` key (~45% over
|
|
1225
|
-
// consented real-person route bills the premium `-realface-` key
|
|
1226
|
-
// endpoint). A reference video flips to `-vref-{res}-{T}s`,
|
|
1227
|
-
|
|
1351
|
+
// Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
|
|
1352
|
+
// × vref × res × duration). AI-face route bills the `-face-` key (~45% over
|
|
1353
|
+
// faceless); consented real-person route bills the premium `-realface-` key
|
|
1354
|
+
// (fal partner endpoint). A reference video flips to `-vref-{res}-{T}s`,
|
|
1355
|
+
// T = in + out.
|
|
1356
|
+
//
|
|
1357
|
+
// ⚠️ EVERY BOUND HERE IS VERSION-SCOPED. 2.5 is 480p/720p only, runs to 30s,
|
|
1358
|
+
// and takes references to 30s combined — so its vref total reaches 60, DOUBLE
|
|
1359
|
+
// 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would quote a real key at a
|
|
1360
|
+
// fraction of the real bill.
|
|
1361
|
+
const v25 = input.model.startsWith('seedance-2.5');
|
|
1362
|
+
const res = input.videoResolution ?? (v25 ? '720p' : '1080p');
|
|
1228
1363
|
const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
|
|
1229
1364
|
const vrefSecs = input.videoRefSeconds ?? 0;
|
|
1230
1365
|
if (vrefSecs > 0) {
|
|
1231
1366
|
// ceil(x - 0.05) matches the server's probe rounding — quote = bill.
|
|
1232
|
-
const
|
|
1367
|
+
const maxTotal = v25 ? SEEDANCE_25_VREF_MAX_TOTAL : SEEDANCE_20_VREF_MAX_TOTAL;
|
|
1368
|
+
const total = Math.min(maxTotal, Math.max(6, Math.ceil(vrefSecs - 0.05) + input.duration));
|
|
1233
1369
|
return `${input.model}${face}-vref-${res}-${total}s`;
|
|
1234
1370
|
}
|
|
1235
1371
|
return `${input.model}${face}-${res}-${input.duration}s`;
|
|
@@ -1279,6 +1415,91 @@ export function klingEditCostKey(model, duration) {
|
|
|
1279
1415
|
export function omniFlashEditCostKey(duration) {
|
|
1280
1416
|
return `omni-flash-edit-${duration}s`;
|
|
1281
1417
|
}
|
|
1418
|
+
/** Max billed (input + output) seconds on a Seedance video-reference gen.
|
|
1419
|
+
* 2.0: refs 2–15s + output 4–15s. 2.5: refs to 30s + output to 30s. */
|
|
1420
|
+
const SEEDANCE_20_VREF_MAX_TOTAL = 30;
|
|
1421
|
+
const SEEDANCE_25_VREF_MAX_TOTAL = 60;
|
|
1422
|
+
/** Seedance 2.5 source-clip bounds for the edit task type. */
|
|
1423
|
+
export const SEEDANCE_25_EDIT_MIN_SECONDS = 4;
|
|
1424
|
+
export const SEEDANCE_25_EDIT_MAX_SECONDS = 30;
|
|
1425
|
+
// Seedance 2.5 video-edit cost key — mirrors seedanceCreditKey() in
|
|
1426
|
+
// slate/src/shared/pricing.ts for the edit row (must byte-match; checked by the
|
|
1427
|
+
// slates-api pricing-consistency script). Unlike the Kling and Omni Flash edit
|
|
1428
|
+
// keys this one carries a RESOLUTION and a FACE ROUTE, because Seedance's edit
|
|
1429
|
+
// price moves with both. Duration is the CEILED source-clip length: the edit
|
|
1430
|
+
// task type forces `duration: -1` on the wire, so the source clip is the only
|
|
1431
|
+
// honest quote.
|
|
1432
|
+
export function seedanceEditCostKey(input) {
|
|
1433
|
+
const res = input.videoResolution ?? '720p';
|
|
1434
|
+
const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
|
|
1435
|
+
return `seedance-2.5-edit${face}-${res}-${input.duration}s`;
|
|
1436
|
+
}
|
|
1437
|
+
// ── Audio ───────────────────────────────────────────────────────
|
|
1438
|
+
// Exported: the exact `model` ids slates_generate_audio accepts — consumed
|
|
1439
|
+
// by the desktop Studio Agent system prompt (SSOT; never restate these ids
|
|
1440
|
+
// in prose that can drift). Mirrors VIDEO_MODELS for the third media type.
|
|
1441
|
+
export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx'];
|
|
1442
|
+
/**
|
|
1443
|
+
* Per-surface bounds and defaults.
|
|
1444
|
+
*
|
|
1445
|
+
* 🚨 THESE FOUR NUMBERS PER SURFACE LIVE IN THREE REPOS. A change is a
|
|
1446
|
+
* three-site edit, every time:
|
|
1447
|
+
* 1. HERE (`audioCostKey`, the agent's pre-flight quote)
|
|
1448
|
+
* 2. `slate/src/shared/pricing.ts` → MODEL_REGISTRY `audio.durationSeconds`
|
|
1449
|
+
* (min/max/default), read by `clampAudioDuration` + `audioCreditKey`
|
|
1450
|
+
* 3. `slates-api/src/lib/audio-keys.ts` → the server's fail-closed bounds
|
|
1451
|
+
* `slates-api/scripts/pricing-consistency-check.mjs` §4 asserts 1 and 2 agree
|
|
1452
|
+
* at EVERY value including out-of-range ones; the gate check covers 3.
|
|
1453
|
+
*
|
|
1454
|
+
* The MINs used to be missing here, and the clamp floor was a hardcoded 1. That
|
|
1455
|
+
* made `slates_estimate_generation_cost({model:'seed-audio', duration:2})`
|
|
1456
|
+
* quote a real `seed-audio-2s` price for a generation the desktop would bill as
|
|
1457
|
+
* 3s and the proxy would REJECT outright. Same for an omitted duration, which
|
|
1458
|
+
* quoted `seed-audio-1s` against the desktop's `seed-audio-15s`.
|
|
1459
|
+
*/
|
|
1460
|
+
export const SEED_AUDIO_MIN_SECONDS = 3;
|
|
1461
|
+
export const SEED_AUDIO_MAX_SECONDS = 120;
|
|
1462
|
+
export const SEED_AUDIO_DEFAULT_SECONDS = 15;
|
|
1463
|
+
export const ELEVEN_SFX_MIN_SECONDS = 1;
|
|
1464
|
+
export const ELEVEN_SFX_MAX_SECONDS = 22;
|
|
1465
|
+
export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
|
|
1466
|
+
/**
|
|
1467
|
+
* Byte-for-byte the desktop's `clampAudioDuration` in slate/src/shared/pricing.ts,
|
|
1468
|
+
* INCLUDING the non-finite arm — that one matters: `Math.max(min, NaN)` is NaN,
|
|
1469
|
+
* so without it a NaN duration produces the key `seed-audio-NaNs` here while the
|
|
1470
|
+
* desktop quotes the default. Divergence at a value neither side can bill is
|
|
1471
|
+
* still divergence; the checker sweeps for it.
|
|
1472
|
+
*/
|
|
1473
|
+
function clampAudioSeconds(value, min, max, fallback) {
|
|
1474
|
+
if (!Number.isFinite(value))
|
|
1475
|
+
return fallback;
|
|
1476
|
+
return Math.min(max, Math.max(min, Math.ceil(value)));
|
|
1477
|
+
}
|
|
1478
|
+
function clampInt(value, min, max) {
|
|
1479
|
+
return Math.min(max, Math.max(min, Math.ceil(value)));
|
|
1480
|
+
}
|
|
1481
|
+
// Exported for scripts/pricing-consistency-check.mjs (slates-api repo), which
|
|
1482
|
+
// asserts this builder byte-matches the desktop's audioCreditKey().
|
|
1483
|
+
//
|
|
1484
|
+
// Seed Audio has NO duration parameter — length is driven by the prompt text.
|
|
1485
|
+
// We bill the REQUESTED duration (shape B, locked 2026-07-31): the desktop
|
|
1486
|
+
// injects "... N seconds" into the prompt and bills seed-audio-{N}s, so
|
|
1487
|
+
// display == billing with no amendment to the pricing law. The server probes
|
|
1488
|
+
// the returned audio.duration afterwards and logs SEED AUDIO BILLING DRIFT.
|
|
1489
|
+
export function audioCostKey(input) {
|
|
1490
|
+
if (input.model === 'seed-audio') {
|
|
1491
|
+
// `?? default` before the clamp, not `?? 0` — the desktop resolves a missing
|
|
1492
|
+
// duration to the registry DEFAULT, and a quote that doesn't match what the
|
|
1493
|
+
// desktop would bill is the whole bug class this mirrors away.
|
|
1494
|
+
const secs = clampAudioSeconds(input.durationSeconds ?? SEED_AUDIO_DEFAULT_SECONDS, SEED_AUDIO_MIN_SECONDS, SEED_AUDIO_MAX_SECONDS, SEED_AUDIO_DEFAULT_SECONDS);
|
|
1495
|
+
return `seed-audio-${secs}s`;
|
|
1496
|
+
}
|
|
1497
|
+
if (input.model === 'eleven-sfx') {
|
|
1498
|
+
const secs = clampAudioSeconds(input.durationSeconds ?? ELEVEN_SFX_DEFAULT_SECONDS, ELEVEN_SFX_MIN_SECONDS, ELEVEN_SFX_MAX_SECONDS, ELEVEN_SFX_DEFAULT_SECONDS);
|
|
1499
|
+
return `eleven-sfx-${secs}s`;
|
|
1500
|
+
}
|
|
1501
|
+
throw new Error(`Unknown audio model: ${input.model}`);
|
|
1502
|
+
}
|
|
1282
1503
|
/**
|
|
1283
1504
|
* Forgiving model-id resolver. Agents routinely paste registry COST keys
|
|
1284
1505
|
* ("kling-v3-standard-8s", "seedance-2-1080p-8s") into the `model` param —
|
|
@@ -1323,6 +1544,13 @@ function resolveVideoModel(raw) {
|
|
|
1323
1544
|
'kling-v3.0-omni-pro': 'kling-v3.0-omni',
|
|
1324
1545
|
'seedance-2.0': 'seedance-2',
|
|
1325
1546
|
'seedance-2-0': 'seedance-2',
|
|
1547
|
+
// ⚠️ The 2.5 spellings must resolve to 2.5, and the BARE `seedance` must keep
|
|
1548
|
+
// resolving to 2.0 — 2.0 is the default video model and holds 1080p/4K, which
|
|
1549
|
+
// 2.5 does not have at all.
|
|
1550
|
+
'seedance-2.5': 'seedance-2.5',
|
|
1551
|
+
'seedance-25': 'seedance-2.5',
|
|
1552
|
+
'seedance-2-5': 'seedance-2.5',
|
|
1553
|
+
'seedance2.5': 'seedance-2.5',
|
|
1326
1554
|
seedance: 'seedance-2',
|
|
1327
1555
|
'veo-3.1': 'veo-3.1-fast',
|
|
1328
1556
|
'veo-3': 'veo-3.1-fast',
|
|
@@ -1345,6 +1573,11 @@ function promptingSkillFor(model) {
|
|
|
1345
1573
|
return 'slates-prompting-kling-v3';
|
|
1346
1574
|
if (model.startsWith('veo'))
|
|
1347
1575
|
return 'slates-prompting-veo-3';
|
|
1576
|
+
// 2.5 BEFORE the generic seedance test — "seedance-2.5" also starts with
|
|
1577
|
+
// "seedance", and falling through hands 2.0's guide to a model with different
|
|
1578
|
+
// limits, a different resolution ladder and an extra task type.
|
|
1579
|
+
if (model.startsWith('seedance-2.5'))
|
|
1580
|
+
return 'slates-prompting-seedance-2-5';
|
|
1348
1581
|
if (model.startsWith('seedance'))
|
|
1349
1582
|
return 'slates-prompting-seedance';
|
|
1350
1583
|
if (model.startsWith('omni-flash'))
|
|
@@ -1353,23 +1586,30 @@ function promptingSkillFor(model) {
|
|
|
1353
1586
|
}
|
|
1354
1587
|
export const generateVideo = {
|
|
1355
1588
|
id: 'slates_generate_video',
|
|
1356
|
-
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
|
|
1589
|
+
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
|
|
1357
1590
|
input: z.object({
|
|
1358
1591
|
prompt: z.string().min(1).max(4000),
|
|
1359
|
-
model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
|
|
1592
|
+
model: z.string().describe('One of: kling-v3.0-std | kling-v3.0-pro | kling-v3.0-omni | seedance-2 | seedance-2.5 | veo-3.1-fast | veo-3.1-standard | omni-flash. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (4-30s takes, 30 image refs, audio-only refs — but 480p/720p ONLY, so stay on seedance-2 whenever resolution matters), Veo = native-synced-audio niche only (16:9, 4/6/8s) — never the default, omni-flash = cheap 720p tier with audio included (3-10s, 16:9/9:16; t2v, single-start-frame i2v, or up to 7 reference images; no last frame / video / audio refs). All are VIDEO-only. For per-call cost, call slates_estimate_generation_cost — never quote prices from memory.'),
|
|
1360
1593
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
|
|
1361
1594
|
aspectRatio: z.enum(['1:1', '16:9', '9:16', '4:3', '3:4', '21:9', '9:21', '4:5', '5:4', '2:3', '3:2']).optional().describe('Veo locks to 16:9 — passing anything else will be ignored or fail. Kling/Seedance support all.'),
|
|
1362
|
-
duration: z.number().int().min(3).max(
|
|
1363
|
-
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive). Veo: 720p/1080p same price, 4K more (8s only).'),
|
|
1595
|
+
duration: z.number().int().min(3).max(30).optional().describe('Seconds. Kling: 5-15. Veo: 4, 6, or 8 only (4K only at 8s). Seedance 2: 4-15. Seedance 2.5: 4-30 (the only model that reaches 30). Omni Flash: 3-10. Default 5 if omitted but always be explicit — cost scales linearly, and a 30s seedance-2.5 take is several hundred credits.'),
|
|
1596
|
+
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Veo + Seedance. Seedance 2: 480p/720p/1080p/4K (default 1080p; 4K is native, the most expensive, and Pro-only). Seedance 2.5: 480p/720p ONLY (default 720p) — it has no 1080p and no 4K on any provider, so asking for one is rejected, not downgraded. Veo: 720p/1080p same price, 4K more (8s only).'),
|
|
1364
1597
|
firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video: asset UUID or badge code ("IMG-A8") — codes resolve against the project at call time, so a code the user just spoke is always safe to pass.'),
|
|
1365
1598
|
lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code). Veo and Seedance only. Pairs with firstFrameAssetId for guided transitions.'),
|
|
1366
|
-
ingredientAssetIds: z.array(z.string()).max(
|
|
1599
|
+
ingredientAssetIds: z.array(z.string()).max(30).optional().describe('Visual reference / ingredient assets (UUIDs or badge codes) for Kling Omni, Seedance, or Omni Flash. Up to 30 (Seedance 2.5), 9 (Seedance 2), 4 (Kling), or 7 (Omni Flash, combined across all ref params). More is not better: 2-4 strong references beat both extremes, and past 4 reference PEOPLE output stability drops on Seedance regardless of the cap.'),
|
|
1367
1600
|
characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets (UUIDs or badge codes) — keeps a character consistent across the shot.'),
|
|
1368
1601
|
environmentAssetIds: z.array(z.string()).optional().describe('Environment reference assets (UUIDs or badge codes) — keeps a location/setting consistent across the shot.'),
|
|
1369
1602
|
styleAssetIds: z.array(z.string()).optional().describe('Style reference assets (UUIDs or badge codes) — locks the visual style of the shot.'),
|
|
1370
|
-
videoReferenceAssetId: z.string().optional().describe('
|
|
1371
|
-
videoReferenceSeconds: z.number().optional().describe('
|
|
1372
|
-
audioReferenceAssetId: z.string().optional().describe('
|
|
1603
|
+
videoReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into videoReferenceAssetIds; prefer that for anything new. A single VIDEO asset (UUID or badge code) used as a reference. Kept working forever: installed CLI and MCP builds send this shape.'),
|
|
1604
|
+
videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach. Required with videoReferenceAssetId: that clip\'s duration in seconds.'),
|
|
1605
|
+
audioReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into audioReferenceAssetIds; prefer that. A single AUDIO asset (UUID or badge code) used as a reference.'),
|
|
1606
|
+
// ── Multimodal references, the plural surface ──
|
|
1607
|
+
// The capacity sentences are DERIVED from MODEL_FACTS (see
|
|
1608
|
+
// multimodalRefSummary) rather than hand-typed, so a cap change in one
|
|
1609
|
+
// place cannot leave a stale number in a description an LLM reads.
|
|
1610
|
+
videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS (UUIDs or badge codes) read alongside the images and audio in the same generation — own-footage restyle, MOTION TRANSFER ("the character from image 1 performs the motion from video 1"), or dialogue conditioning. Cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; ignored elsewhere. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to combined input+output seconds (the vref key) — pass videoReferenceSecondsEach so the quote is right. If any clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Over the cap is REFUSED, never trimmed: a dropped clip would already have been priced in.`),
|
|
1611
|
+
videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — the bill is Σceil(each) + output seconds. The server re-derives this by probing every uploaded clip, so an understated value just gets corrected upward.'),
|
|
1612
|
+
audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO clips (UUIDs or badge codes) read alongside the images and video — e.g. lip-sync a character to a line ("the character in image 1 speaks the dialogue from audio 1"). Cited as "audio 1", "audio 2"… in the order given. No billing surcharge (Seedance audio is included). ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
|
|
1373
1613
|
sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
|
|
1374
1614
|
audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
|
|
1375
1615
|
generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
|
|
@@ -1438,7 +1678,10 @@ export const generateVideo = {
|
|
|
1438
1678
|
message: `Omni Flash supports 3-10 seconds (you passed ${input.duration}s). Pick a duration in that range.`,
|
|
1439
1679
|
});
|
|
1440
1680
|
}
|
|
1441
|
-
if (input.lastFrameAssetId ||
|
|
1681
|
+
if (input.lastFrameAssetId ||
|
|
1682
|
+
input.videoReferenceAssetId || input.audioReferenceAssetId ||
|
|
1683
|
+
(input.videoReferenceAssetIds?.length ?? 0) > 0 ||
|
|
1684
|
+
(input.audioReferenceAssetIds?.length ?? 0) > 0) {
|
|
1442
1685
|
return ok({
|
|
1443
1686
|
requires_clarification: true,
|
|
1444
1687
|
missing: [],
|
|
@@ -1499,6 +1742,10 @@ export const generateVideo = {
|
|
|
1499
1742
|
refInputs.push({ ref: input.videoReferenceAssetId, role: 'video reference' });
|
|
1500
1743
|
if (input.audioReferenceAssetId)
|
|
1501
1744
|
refInputs.push({ ref: input.audioReferenceAssetId, role: 'audio reference' });
|
|
1745
|
+
for (const r of input.videoReferenceAssetIds ?? [])
|
|
1746
|
+
refInputs.push({ ref: r, role: 'video reference' });
|
|
1747
|
+
for (const r of input.audioReferenceAssetIds ?? [])
|
|
1748
|
+
refInputs.push({ ref: r, role: 'audio reference' });
|
|
1502
1749
|
const resolvedRefs = await resolveAssetRefs(ctx, input.projectId, refInputs.map((r) => r.ref));
|
|
1503
1750
|
const rid = (v) => v ? (resolvedRefs.get(v)?.id ?? v) : v;
|
|
1504
1751
|
const rids = (a) => a?.map((v) => resolvedRefs.get(v)?.id ?? v);
|
|
@@ -1506,6 +1753,8 @@ export const generateVideo = {
|
|
|
1506
1753
|
input.lastFrameAssetId = rid(input.lastFrameAssetId);
|
|
1507
1754
|
input.videoReferenceAssetId = rid(input.videoReferenceAssetId);
|
|
1508
1755
|
input.audioReferenceAssetId = rid(input.audioReferenceAssetId);
|
|
1756
|
+
input.videoReferenceAssetIds = rids(input.videoReferenceAssetIds);
|
|
1757
|
+
input.audioReferenceAssetIds = rids(input.audioReferenceAssetIds);
|
|
1509
1758
|
input.ingredientAssetIds = rids(input.ingredientAssetIds);
|
|
1510
1759
|
input.characterAssetIds = rids(input.characterAssetIds);
|
|
1511
1760
|
input.environmentAssetIds = rids(input.environmentAssetIds);
|
|
@@ -1514,13 +1763,59 @@ export const generateVideo = {
|
|
|
1514
1763
|
// A video reference bills combined input+output seconds (the vref key) —
|
|
1515
1764
|
// the quote needs the clip's length. The server probes the uploaded ref
|
|
1516
1765
|
// and corrects the key anyway, so this only gates quote accuracy.
|
|
1517
|
-
|
|
1766
|
+
// 🚨 Seedance 2.5 PROMPT-INTENT PRE-FLIGHT. With references attached, the
|
|
1767
|
+
// provider sorts a request into one of five task types from the roles PLUS
|
|
1768
|
+
// the prompt's intent, then fails a mismatch ASYNCHRONOUSLY — after the task
|
|
1769
|
+
// queues and credits are reserved. Catch it before the spend, NAME the words,
|
|
1770
|
+
// and never rewrite the prompt (the prompt-transparency invariant).
|
|
1771
|
+
if (input.model === 'seedance-2.5') {
|
|
1772
|
+
// 🚨 EVERY reference shape counts, singular AND plural. The classifier
|
|
1773
|
+
// engages on the presence of reference ROLES, and the plural arrays are
|
|
1774
|
+
// the surface new callers use — listing only the deprecated singular ones
|
|
1775
|
+
// meant the modern call path got no warning at all, which is precisely
|
|
1776
|
+
// backwards.
|
|
1777
|
+
const hasRefs = !!input.firstFrameAssetId || !!input.lastFrameAssetId ||
|
|
1778
|
+
(input.ingredientAssetIds?.length ?? 0) > 0 ||
|
|
1779
|
+
(input.characterAssetIds?.length ?? 0) > 0 ||
|
|
1780
|
+
(input.environmentAssetIds?.length ?? 0) > 0 ||
|
|
1781
|
+
(input.styleAssetIds?.length ?? 0) > 0 ||
|
|
1782
|
+
!!input.videoReferenceAssetId || !!input.audioReferenceAssetId ||
|
|
1783
|
+
(input.videoReferenceAssetIds?.length ?? 0) > 0 ||
|
|
1784
|
+
(input.audioReferenceAssetIds?.length ?? 0) > 0;
|
|
1785
|
+
// Trigger words are DERIVED from the model-facts SSOT, not retyped here —
|
|
1786
|
+
// a local literal had already lost 'continue the story'.
|
|
1787
|
+
const hits = hasRefs ? seedanceTaskIntentWords(input.prompt) : [];
|
|
1788
|
+
if (hits.length > 0 && !input.confirm) {
|
|
1789
|
+
return ok({
|
|
1790
|
+
requires_clarification: true,
|
|
1791
|
+
missing: ['prompt'],
|
|
1792
|
+
message: `Seedance 2.5 reads ${hits.map((w) => `"${w}"`).join(', ')} in this prompt as an EDIT or EXTEND instruction and may run it as a video edit, ` +
|
|
1793
|
+
`which fails after the job has queued. If you mean to edit an existing clip, call slates_edit_video with model "seedance-2.5-edit". ` +
|
|
1794
|
+
`If you mean a fresh shot, describe the finished frame instead of an instruction to change one ("the workshop bench, clear and uncluttered" rather than "remove the tripod"). ` +
|
|
1795
|
+
`Pass confirm=true to send it as written.`,
|
|
1796
|
+
});
|
|
1797
|
+
}
|
|
1798
|
+
}
|
|
1799
|
+
// The quote needs a length for EVERY reference clip, in both shapes. A
|
|
1800
|
+
// missing one doesn't fail the generation (the server probes and corrects
|
|
1801
|
+
// upward) — it silently under-quotes, which is worse than asking.
|
|
1802
|
+
const isSeedance = input.model.startsWith('seedance');
|
|
1803
|
+
if (input.videoReferenceAssetId && isSeedance && !input.videoReferenceSeconds) {
|
|
1518
1804
|
return ok({
|
|
1519
1805
|
requires_clarification: true,
|
|
1520
1806
|
missing: ['videoReferenceSeconds'],
|
|
1521
1807
|
message: 'A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSeconds (the reference clip\'s duration, shown in slates_list_assets) so the pre-flight quote matches the bill.',
|
|
1522
1808
|
});
|
|
1523
1809
|
}
|
|
1810
|
+
const pluralRefCount = input.videoReferenceAssetIds?.length ?? 0;
|
|
1811
|
+
if (pluralRefCount > 0 && isSeedance &&
|
|
1812
|
+
(input.videoReferenceSecondsEach?.length ?? 0) !== pluralRefCount) {
|
|
1813
|
+
return ok({
|
|
1814
|
+
requires_clarification: true,
|
|
1815
|
+
missing: ['videoReferenceSecondsEach'],
|
|
1816
|
+
message: `A Seedance video reference bills on combined input+output seconds. Pass videoReferenceSecondsEach with exactly ${pluralRefCount} duration${pluralRefCount === 1 ? '' : 's'}, in the same order as videoReferenceAssetIds (durations are shown in slates_list_assets), so the pre-flight quote matches the bill.`,
|
|
1817
|
+
});
|
|
1818
|
+
}
|
|
1524
1819
|
const cloud = ctx.cloud();
|
|
1525
1820
|
const registry = await cloud.get('/api/agent/models');
|
|
1526
1821
|
const costKey = videoCostKey({
|
|
@@ -1530,7 +1825,13 @@ export const generateVideo = {
|
|
|
1530
1825
|
sound: input.sound,
|
|
1531
1826
|
seedanceFace: input.seedanceFace,
|
|
1532
1827
|
seedanceRealFace: input.seedanceRealFace,
|
|
1533
|
-
|
|
1828
|
+
// Σ ceil(d - 0.05) over every reference clip, both shapes — the same
|
|
1829
|
+
// expression the desktop's estimateCost and the handler's key builder
|
|
1830
|
+
// use. Quoting only the singular would understate a multi-clip call.
|
|
1831
|
+
videoRefSeconds: (input.videoReferenceAssetId && input.videoReferenceSeconds
|
|
1832
|
+
? Math.ceil(input.videoReferenceSeconds - 0.05)
|
|
1833
|
+
: 0) +
|
|
1834
|
+
(input.videoReferenceSecondsEach ?? []).reduce((n, d) => n + (d > 0 ? Math.ceil(d - 0.05) : 0), 0),
|
|
1534
1835
|
});
|
|
1535
1836
|
// Hard consent gate, checked before any spend: the real-face route is
|
|
1536
1837
|
// consent-attested by design (the desktop enforces it too).
|
|
@@ -1624,8 +1925,13 @@ export const generateVideo = {
|
|
|
1624
1925
|
characterAssetIds: input.characterAssetIds ?? [],
|
|
1625
1926
|
environmentAssetIds: input.environmentAssetIds ?? [],
|
|
1626
1927
|
styleAssetIds: input.styleAssetIds ?? [],
|
|
1928
|
+
// Both shapes on the wire. The route merges and dedupes them, so an old
|
|
1929
|
+
// client sending only the singular and a new one sending only the plural
|
|
1930
|
+
// reach the identical handler params.
|
|
1627
1931
|
videoReferenceAssetId: input.videoReferenceAssetId,
|
|
1628
1932
|
audioReferenceAssetId: input.audioReferenceAssetId,
|
|
1933
|
+
videoReferenceAssetIds: input.videoReferenceAssetIds,
|
|
1934
|
+
audioReferenceAssetIds: input.audioReferenceAssetIds,
|
|
1629
1935
|
sound: input.sound,
|
|
1630
1936
|
audioLanguage: input.audioLanguage,
|
|
1631
1937
|
generateMusic: input.generateMusic,
|
|
@@ -1671,30 +1977,179 @@ export const generateVideo = {
|
|
|
1671
1977
|
};
|
|
1672
1978
|
},
|
|
1673
1979
|
};
|
|
1980
|
+
// ── Generate audio ──────────────────────────────────────────────
|
|
1981
|
+
export const generateAudio = {
|
|
1982
|
+
id: 'slates_generate_audio',
|
|
1983
|
+
description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Two surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s) and eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop). Which surface for which job: read the slates-model-selection skill. ' +
|
|
1984
|
+
'🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
|
|
1985
|
+
'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
|
|
1986
|
+
'projectId is REQUIRED (no headless path). Cost > 17 credits returns requires_confirm — pass confirm=true after explicit user OK. No skill files installed? Call slates_get_prompting_guide first.',
|
|
1987
|
+
input: z.object({
|
|
1988
|
+
projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
|
|
1989
|
+
model: z
|
|
1990
|
+
.enum(AUDIO_MODELS)
|
|
1991
|
+
.describe('Audio surface. seed-audio = scene/ambience/beds/dialogue (default choice), eleven-sfx = one precise effect or a seamless loop. Routing doctrine: slates-model-selection skill.'),
|
|
1992
|
+
prompt: z
|
|
1993
|
+
.string()
|
|
1994
|
+
.min(1)
|
|
1995
|
+
.max(5000)
|
|
1996
|
+
.describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars.'),
|
|
1997
|
+
durationSeconds: z
|
|
1998
|
+
.number()
|
|
1999
|
+
.optional()
|
|
2000
|
+
.describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.'),
|
|
2001
|
+
voice: z
|
|
2002
|
+
.string()
|
|
2003
|
+
.optional()
|
|
2004
|
+
.describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
|
|
2005
|
+
speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
|
|
2006
|
+
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
|
|
2007
|
+
pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
|
|
2008
|
+
multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
|
|
2009
|
+
loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
|
|
2010
|
+
promptInfluence: z.number().min(0).max(1).optional().describe('eleven-sfx only — 0-1, default 0.3. Higher hugs your wording with less variation between takes.'),
|
|
2011
|
+
audioReferenceAssetIds: z
|
|
2012
|
+
.array(z.string())
|
|
2013
|
+
.max(3)
|
|
2014
|
+
.optional()
|
|
2015
|
+
.describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId — the API rejects both.'),
|
|
2016
|
+
imageReferenceAssetId: z
|
|
2017
|
+
.string()
|
|
2018
|
+
.optional()
|
|
2019
|
+
.describe('seed-audio only — ONE image asset to score what is in frame. MUTUALLY EXCLUSIVE with audioReferenceAssetIds.'),
|
|
2020
|
+
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
2021
|
+
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate after explicit user OK.'),
|
|
2022
|
+
}),
|
|
2023
|
+
run: async (input, ctx) => {
|
|
2024
|
+
// ── Per-surface clarification + constraint gates ──
|
|
2025
|
+
const cfgDefaults = {
|
|
2026
|
+
'seed-audio': SEED_AUDIO_DEFAULT_SECONDS,
|
|
2027
|
+
'eleven-sfx': ELEVEN_SFX_DEFAULT_SECONDS,
|
|
2028
|
+
};
|
|
2029
|
+
const seconds = input.durationSeconds ?? cfgDefaults[input.model];
|
|
2030
|
+
if (input.model === 'seed-audio') {
|
|
2031
|
+
if (seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
|
|
2032
|
+
return ok({
|
|
2033
|
+
requires_clarification: true,
|
|
2034
|
+
missing: ['durationSeconds'],
|
|
2035
|
+
message: `Seed Audio needs a durationSeconds of ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS}. It has NO duration parameter — the number is written into the prompt AND is what the user is billed, so it must be a deliberate choice. Ask the user how long the bed should be (a few seconds longer than the clip it sits under, so the edit has handles).`,
|
|
2036
|
+
});
|
|
2037
|
+
}
|
|
2038
|
+
if ((input.audioReferenceAssetIds?.length ?? 0) > 0 && input.imageReferenceAssetId) {
|
|
2039
|
+
throw new Error('Seed Audio takes audio references OR one image reference, never both — the provider rejects the combination.');
|
|
2040
|
+
}
|
|
2041
|
+
}
|
|
2042
|
+
if (input.model === 'eleven-sfx') {
|
|
2043
|
+
if (seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
|
|
2044
|
+
return ok({
|
|
2045
|
+
requires_clarification: true,
|
|
2046
|
+
missing: ['durationSeconds'],
|
|
2047
|
+
message: `Sound Effects needs a durationSeconds of ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. It is billed per second and is never left for the model to pick (that would make the charge non-deterministic). Roughly: 0.5-1s for an impact, 2-4s for a whoosh, 8-22s for a loopable bed.`,
|
|
2048
|
+
});
|
|
2049
|
+
}
|
|
2050
|
+
if (input.prompt.length > 450) {
|
|
2051
|
+
throw new Error(`Sound Effects accepts up to 450 characters — this prompt is ${input.prompt.length}.`);
|
|
2052
|
+
}
|
|
2053
|
+
}
|
|
2054
|
+
await ctx.desktop().requireCapability('audio-generation', 'audio generation');
|
|
2055
|
+
// ── Resolve asset refs at CALL time (UUIDs or badge codes) ──
|
|
2056
|
+
const refInputs = [];
|
|
2057
|
+
for (const ref of input.audioReferenceAssetIds ?? [])
|
|
2058
|
+
refInputs.push({ ref, role: 'audio reference' });
|
|
2059
|
+
if (input.imageReferenceAssetId)
|
|
2060
|
+
refInputs.push({ ref: input.imageReferenceAssetId, role: 'image reference' });
|
|
2061
|
+
const resolvedRefs = refInputs.length > 0
|
|
2062
|
+
? await resolveAssetRefs(ctx, input.projectId, refInputs.map((r) => r.ref))
|
|
2063
|
+
: new Map();
|
|
2064
|
+
const rid = (v) => (v ? (resolvedRefs.get(v)?.id ?? v) : v);
|
|
2065
|
+
const refEcho = refInputs.length > 0 ? describeResolvedRefs(refInputs, resolvedRefs) : '';
|
|
2066
|
+
// ── Cost — from the CLOUD registry, keyed by the SAME builder the desktop
|
|
2067
|
+
// and the proxy use. Never quote a price from memory. ──
|
|
2068
|
+
const cloud = ctx.cloud();
|
|
2069
|
+
const registry = await cloud.get('/api/agent/models');
|
|
2070
|
+
const costKey = audioCostKey({
|
|
2071
|
+
model: input.model,
|
|
2072
|
+
durationSeconds: seconds,
|
|
2073
|
+
});
|
|
2074
|
+
const entry = registry.models.find((m) => m.model === costKey);
|
|
2075
|
+
if (!entry) {
|
|
2076
|
+
throw new Error(`Audio variant not in registry: ${costKey}. Available audio models: ${AUDIO_MODELS.join(' | ')}.`);
|
|
2077
|
+
}
|
|
2078
|
+
const totalCents = creditCost(entry);
|
|
2079
|
+
if (!input.confirm && totalCents > CONFIRM_CREDITS) {
|
|
2080
|
+
return ok({
|
|
2081
|
+
requires_confirm: true,
|
|
2082
|
+
model: input.model,
|
|
2083
|
+
variant: costKey,
|
|
2084
|
+
cost_credits: totalCents,
|
|
2085
|
+
cost_display: fmtCredits(totalCents),
|
|
2086
|
+
message: `${input.model} audio will cost ${fmtCredits(totalCents)}. ` +
|
|
2087
|
+
(input.model === 'seed-audio'
|
|
2088
|
+
? `You are billed for the ${seconds}s you requested regardless of the returned length. `
|
|
2089
|
+
: '') +
|
|
2090
|
+
'Confirm with the user, then call again with confirm: true.',
|
|
2091
|
+
});
|
|
2092
|
+
}
|
|
2093
|
+
// ── Submit through the DESKTOP so the progress card, the project save,
|
|
2094
|
+
// and recovery all behave exactly like a UI-triggered run. ──
|
|
2095
|
+
const desktop = ctx.desktop();
|
|
2096
|
+
if (input.background) {
|
|
2097
|
+
await desktop.requireCapability('background-generation', 'background generation');
|
|
2098
|
+
}
|
|
2099
|
+
const result = await desktop.post('/agent/generation/audio', {
|
|
2100
|
+
projectId: input.projectId,
|
|
2101
|
+
model: input.model,
|
|
2102
|
+
prompt: input.prompt,
|
|
2103
|
+
durationSeconds: seconds,
|
|
2104
|
+
voice: input.voice,
|
|
2105
|
+
speed: input.speed,
|
|
2106
|
+
volume: input.volume,
|
|
2107
|
+
pitch: input.pitch,
|
|
2108
|
+
multilingual: input.multilingual,
|
|
2109
|
+
loop: input.loop,
|
|
2110
|
+
promptInfluence: input.promptInfluence,
|
|
2111
|
+
audioReferenceAssetIds: (input.audioReferenceAssetIds ?? []).map((r) => rid(r)),
|
|
2112
|
+
imageReferenceAssetId: rid(input.imageReferenceAssetId),
|
|
2113
|
+
background: input.background,
|
|
2114
|
+
});
|
|
2115
|
+
if (!result.success)
|
|
2116
|
+
throw new Error(result.error ?? 'Audio generation failed');
|
|
2117
|
+
if (result.background) {
|
|
2118
|
+
return backgroundSubmitted(`${input.model} audio generation`, [result.generationId].filter(Boolean), { model: input.model, variant: costKey, projectId: input.projectId, cost_credits: totalCents }, refEcho);
|
|
2119
|
+
}
|
|
2120
|
+
return {
|
|
2121
|
+
text: `Generated ${input.model} audio into project ${input.projectId} for ${fmtCredits(totalCents)}` +
|
|
2122
|
+
`. Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
|
|
2123
|
+
(refEcho ? ` ${refEcho}` : ''),
|
|
2124
|
+
data: {
|
|
2125
|
+
model: input.model,
|
|
2126
|
+
variant: costKey,
|
|
2127
|
+
projectId: input.projectId,
|
|
2128
|
+
durationSeconds: seconds,
|
|
2129
|
+
cost_cents: totalCents,
|
|
2130
|
+
cost_credits: totalCents,
|
|
2131
|
+
asset: result.asset,
|
|
2132
|
+
generationId: result.generationId,
|
|
2133
|
+
},
|
|
2134
|
+
};
|
|
2135
|
+
},
|
|
2136
|
+
};
|
|
1674
2137
|
// ── Generate lip-sync ───────────────────────────────────────────
|
|
1675
2138
|
export const generateLipSync = {
|
|
1676
2139
|
id: 'slates_generate_lip_sync',
|
|
1677
|
-
description: 'Lip-sync a still image (avatar) or a video clip to audio.
|
|
2140
|
+
description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip (~$0.11 / 5s); sourceType=image animates a still avatar (avatar-standard ~$0.42 / 5s; avatar-pro ~$0.86 / 5s). Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
|
|
1678
2141
|
input: z.object({
|
|
1679
2142
|
projectId: z.string().uuid().describe('Slates project the source asset lives in. The new lip-synced video lands here.'),
|
|
1680
2143
|
sourceAssetId: z.string().uuid().describe('Asset id of the still image (avatar flow) or video clip (lip-sync flow). Must already exist in the project — use slates_upload_reference_image or slates_generate_image / slates_generate_video first if needed.'),
|
|
1681
2144
|
sourceType: z.enum(['image', 'video']).describe('"image" = animate a still portrait (avatar). "video" = re-sync an existing talking-head clip. Determines pricing — be deliberate.'),
|
|
1682
|
-
audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText
|
|
2145
|
+
audioMethod: z.enum(['tts', 'upload']).describe('"tts" = generate speech from ttsText. "upload" = use the file at audioFilePath (absolute path on the user\'s machine).'),
|
|
1683
2146
|
ttsText: z.string().min(1).max(2000).optional().describe('Required when audioMethod=tts. The exact words the avatar/clip will speak.'),
|
|
1684
|
-
ttsVoice: z.string().optional().describe('
|
|
1685
|
-
ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('
|
|
1686
|
-
ttsSpeed: z.number().min(0.5).max(2).optional().describe('
|
|
2147
|
+
ttsVoice: z.string().optional().describe('Voice id (e.g. "oversea_male1"). See slates-prompting-lip-sync skill for the voice catalog.'),
|
|
2148
|
+
ttsLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('TTS language. Default EN.'),
|
|
2149
|
+
ttsSpeed: z.number().min(0.5).max(2).optional().describe('TTS speech rate. Default 1.0. Range 0.5-2.0.'),
|
|
1687
2150
|
audioFilePath: z.string().optional().describe('Required when audioMethod=upload. Absolute path to the audio file on the user\'s machine (mp3, wav, m4a).'),
|
|
1688
|
-
avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('
|
|
1689
|
-
klingProvider: z.enum(['fal', 'kling']).optional().describe('
|
|
1690
|
-
engine: z.enum(['kling', 'seedance-2']).optional().describe('Default kling (cheap utility). seedance-2 = premium single-pass: natural speech generated in the video, voice cloned from a video source, audio included. Credits only.'),
|
|
1691
|
-
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
|
|
1692
|
-
aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
|
|
1693
|
-
seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the source (default TRUE for lip-sync; the faceless route would reject it). Bills the -face key.'),
|
|
1694
|
-
seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the source shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
|
|
1695
|
-
realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
|
|
1696
|
-
sourceSeconds: z.number().optional().describe('Seedance engine + sourceType=video: the source clip\'s duration in seconds (from the asset listing). Feeds the vref cost key (input+output billing).'),
|
|
1697
|
-
audioSeconds: z.number().optional().describe('Seedance engine + audioMethod=upload: the audio file\'s duration in seconds — sets the output length (4-15s).'),
|
|
2151
|
+
avatarModel: z.enum(['avatar-standard', 'avatar-pro']).optional().describe('Image-source only. avatar-standard (~14 credits/5s) for general use. avatar-pro (~29 credits/5s) for sharper face fidelity.'),
|
|
2152
|
+
klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. Leave unset: all agent generations bill Slates credits (BYOK is retired).'),
|
|
1698
2153
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
1699
2154
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required for avatar-pro.'),
|
|
1700
2155
|
}),
|
|
@@ -1713,41 +2168,8 @@ export const generateLipSync = {
|
|
|
1713
2168
|
message: 'audioMethod=upload requires audioFilePath. Pass an absolute path to the audio file on the user\'s machine.',
|
|
1714
2169
|
});
|
|
1715
2170
|
}
|
|
1716
|
-
const isSeedance = input.engine === 'seedance-2';
|
|
1717
2171
|
let costKey;
|
|
1718
|
-
|
|
1719
|
-
if (isSeedance) {
|
|
1720
|
-
// Consent gate before any spend, mirroring slates_generate_video.
|
|
1721
|
-
if (input.seedanceRealFace && !input.realFaceConsent) {
|
|
1722
|
-
return ok({
|
|
1723
|
-
requires_clarification: true,
|
|
1724
|
-
missing: ['realFaceConsent'],
|
|
1725
|
-
message: 'Real-person lip-sync needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
|
|
1726
|
-
});
|
|
1727
|
-
}
|
|
1728
|
-
if (input.sourceType === 'video' && !input.sourceSeconds) {
|
|
1729
|
-
return ok({
|
|
1730
|
-
requires_clarification: true,
|
|
1731
|
-
missing: ['sourceSeconds'],
|
|
1732
|
-
message: 'Seedance lip-sync on a video source bills combined input+output seconds. Pass sourceSeconds (the clip\'s duration from slates_list_assets, must be 2-15s).',
|
|
1733
|
-
});
|
|
1734
|
-
}
|
|
1735
|
-
const clamp = (n) => Math.min(15, Math.max(4, Math.ceil(n)));
|
|
1736
|
-
seedanceDuration = clamp(input.sourceType === 'video' && input.sourceSeconds
|
|
1737
|
-
? input.sourceSeconds
|
|
1738
|
-
: input.audioSeconds ?? (input.ttsText ? input.ttsText.length / 13 : 5));
|
|
1739
|
-
costKey = videoCostKey({
|
|
1740
|
-
model: 'seedance-2',
|
|
1741
|
-
duration: seedanceDuration,
|
|
1742
|
-
videoResolution: input.videoResolution ?? '1080p',
|
|
1743
|
-
// Lip-sync sources are faces by definition — face route unless
|
|
1744
|
-
// explicitly disabled or escalated to realface.
|
|
1745
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
1746
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
1747
|
-
videoRefSeconds: input.sourceType === 'video' ? input.sourceSeconds ?? 0 : 0,
|
|
1748
|
-
});
|
|
1749
|
-
}
|
|
1750
|
-
else if (input.sourceType === 'video') {
|
|
2172
|
+
if (input.sourceType === 'video') {
|
|
1751
2173
|
costKey = 'kling-lip-sync-video-5s';
|
|
1752
2174
|
}
|
|
1753
2175
|
else {
|
|
@@ -1776,7 +2198,7 @@ export const generateLipSync = {
|
|
|
1776
2198
|
estimated_cents: totalCents,
|
|
1777
2199
|
estimated_credits: totalCents,
|
|
1778
2200
|
source_ref: sourceRef,
|
|
1779
|
-
message: `Cost: ${fmtCredits(totalCents)} for
|
|
2201
|
+
message: `Cost: ${fmtCredits(totalCents)} for 5s lip-sync (${costKey}). ` +
|
|
1780
2202
|
`Source: ${sourceRef}. ${audioPreview}. ` +
|
|
1781
2203
|
`Re-call with confirm=true after the user explicitly OKs the spend. ` +
|
|
1782
2204
|
`When discussing with the user, refer to the source by its code (matches the gallery badge).`,
|
|
@@ -1799,28 +2221,12 @@ export const generateLipSync = {
|
|
|
1799
2221
|
klingProvider: input.klingProvider,
|
|
1800
2222
|
estimatedCost: totalCents,
|
|
1801
2223
|
background: input.background,
|
|
1802
|
-
// Seedance engine passthrough — the desktop delegates to the seedance
|
|
1803
|
-
// ref-to-video path (vref billing, face cascade, consent gate). The
|
|
1804
|
-
// durations ride along so the desktop bills exactly what was quoted.
|
|
1805
|
-
...(isSeedance
|
|
1806
|
-
? {
|
|
1807
|
-
lipSyncEngine: 'seedance-2',
|
|
1808
|
-
duration: seedanceDuration,
|
|
1809
|
-
videoResolution: input.videoResolution,
|
|
1810
|
-
aspectRatio: input.aspectRatio,
|
|
1811
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
1812
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
1813
|
-
realFaceConsent: input.realFaceConsent === true,
|
|
1814
|
-
sourceDurationSeconds: input.sourceSeconds,
|
|
1815
|
-
audioDurationSeconds: input.audioSeconds,
|
|
1816
|
-
}
|
|
1817
|
-
: {}),
|
|
1818
2224
|
});
|
|
1819
2225
|
if (!result.success)
|
|
1820
2226
|
throw new Error(result.error ?? 'Lip-sync generation failed');
|
|
1821
2227
|
if (result.background) {
|
|
1822
2228
|
const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
|
|
1823
|
-
return backgroundSubmitted(
|
|
2229
|
+
return backgroundSubmitted(`5s lip-sync (${costKey})`, ids, {
|
|
1824
2230
|
variant: costKey,
|
|
1825
2231
|
projectId: input.projectId,
|
|
1826
2232
|
sourceAssetId: input.sourceAssetId,
|
|
@@ -1829,7 +2235,7 @@ export const generateLipSync = {
|
|
|
1829
2235
|
});
|
|
1830
2236
|
}
|
|
1831
2237
|
return {
|
|
1832
|
-
text: `Generated
|
|
2238
|
+
text: `Generated 5s lip-sync (${costKey}) into project ${input.projectId} ` +
|
|
1833
2239
|
`for ${fmtCredits(totalCents)}. ` +
|
|
1834
2240
|
(input.audioMethod === 'tts'
|
|
1835
2241
|
? `Spoken: "${(input.ttsText ?? '').slice(0, 60)}${(input.ttsText ?? '').length > 60 ? '...' : ''}"`
|
|
@@ -1850,58 +2256,21 @@ export const generateLipSync = {
|
|
|
1850
2256
|
// ── Generate motion transfer ────────────────────────────────────
|
|
1851
2257
|
export const generateMotionTransfer = {
|
|
1852
2258
|
id: 'slates_generate_motion_transfer',
|
|
1853
|
-
description: 'Transfer the motion from a reference video onto a target image character.
|
|
2259
|
+
description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std ($0.95 / 5s) or kling-mc-pro ($1.26 / 5s), structured skeleton/depth retargeting, always 5s. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. Both tiers hit the >$0.50 confirm gate.',
|
|
1854
2260
|
input: z.object({
|
|
1855
2261
|
projectId: z.string().uuid().describe('Slates project. Both source and target assets must live here.'),
|
|
1856
|
-
sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project.
|
|
2262
|
+
sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s.'),
|
|
1857
2263
|
targetImageAssetId: z.string().uuid().describe('Asset id of the target image (the character that will perform the motion). Must already exist in the project.'),
|
|
1858
|
-
motionModel: z.enum(['kling-mc-std', 'kling-mc-pro'
|
|
1859
|
-
characterOrientation: z.enum(['video', 'image']).optional().describe('
|
|
1860
|
-
prompt: z.string().optional().describe('
|
|
1861
|
-
klingProvider: z.enum(['fal', 'kling']).optional().describe('
|
|
1862
|
-
duration: z.number().int().min(4).max(15).optional().describe('Seedance engine only — output duration in seconds (4-15). Defaults to the driving clip\'s length.'),
|
|
1863
|
-
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Seedance engine only. Default 1080p.'),
|
|
1864
|
-
aspectRatio: z.string().optional().describe('Seedance engine only. Default 16:9.'),
|
|
1865
|
-
seedanceFace: z.boolean().optional().describe('Seedance engine only — a character\'s face is in the clip/image (default TRUE for motion transfer). Bills the -face key.'),
|
|
1866
|
-
seedanceRealFace: z.boolean().optional().describe('Seedance engine only — the driving clip/subject shows a REAL person. Premium -realface key; REQUIRES realFaceConsent=true.'),
|
|
1867
|
-
realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace — set true only after the user explicitly confirms they hold rights/consent to the likeness.'),
|
|
1868
|
-
sourceVideoSeconds: z.number().optional().describe('Seedance engine: the driving clip\'s duration in seconds (from the asset listing, 2-15s). Feeds the vref cost key (input+output billing).'),
|
|
2264
|
+
motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std (~32 credits) general motion; kling-mc-pro (~42 credits) cleaner anatomy — default.'),
|
|
2265
|
+
characterOrientation: z.enum(['video', 'image']).optional().describe('"video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
|
|
2266
|
+
prompt: z.string().optional().describe('Optional refinement. Read slates-prompting-motion-transfer.'),
|
|
2267
|
+
klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. "fal" (default) uses Slates credits.'),
|
|
1869
2268
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
1870
2269
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate. Required — both tiers exceed.'),
|
|
1871
2270
|
}),
|
|
1872
2271
|
async run(input, ctx) {
|
|
1873
2272
|
const motionModel = input.motionModel ?? 'kling-mc-pro';
|
|
1874
|
-
const
|
|
1875
|
-
let costKey;
|
|
1876
|
-
let seedanceDuration = 0;
|
|
1877
|
-
if (isSeedance) {
|
|
1878
|
-
if (input.seedanceRealFace && !input.realFaceConsent) {
|
|
1879
|
-
return ok({
|
|
1880
|
-
requires_clarification: true,
|
|
1881
|
-
missing: ['realFaceConsent'],
|
|
1882
|
-
message: 'Real-person motion transfer needs consent: confirm with the user that they hold the rights/consent to this likeness, then retry with realFaceConsent=true.',
|
|
1883
|
-
});
|
|
1884
|
-
}
|
|
1885
|
-
if (!input.sourceVideoSeconds) {
|
|
1886
|
-
return ok({
|
|
1887
|
-
requires_clarification: true,
|
|
1888
|
-
missing: ['sourceVideoSeconds'],
|
|
1889
|
-
message: 'Seedance motion transfer bills combined input+output seconds. Pass sourceVideoSeconds (the driving clip\'s duration from slates_list_assets, must be 2-15s).',
|
|
1890
|
-
});
|
|
1891
|
-
}
|
|
1892
|
-
seedanceDuration = input.duration ?? Math.min(15, Math.max(4, Math.ceil(input.sourceVideoSeconds)));
|
|
1893
|
-
costKey = videoCostKey({
|
|
1894
|
-
model: 'seedance-2',
|
|
1895
|
-
duration: seedanceDuration,
|
|
1896
|
-
videoResolution: input.videoResolution ?? '1080p',
|
|
1897
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
1898
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
1899
|
-
videoRefSeconds: input.sourceVideoSeconds,
|
|
1900
|
-
});
|
|
1901
|
-
}
|
|
1902
|
-
else {
|
|
1903
|
-
costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
|
|
1904
|
-
}
|
|
2273
|
+
const costKey = motionModel === 'kling-mc-std' ? 'kling-mc-std-5s' : 'kling-mc-pro-5s';
|
|
1905
2274
|
const cloud = ctx.cloud();
|
|
1906
2275
|
const registry = await cloud.get('/api/agent/models');
|
|
1907
2276
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -1925,9 +2294,9 @@ export const generateMotionTransfer = {
|
|
|
1925
2294
|
estimated_credits: totalCents,
|
|
1926
2295
|
source_ref: source,
|
|
1927
2296
|
target_ref: target,
|
|
1928
|
-
message: `Cost: ${fmtCredits(totalCents)} for
|
|
2297
|
+
message: `Cost: ${fmtCredits(totalCents)} for 5s ${motionModel} (${costKey}). ` +
|
|
1929
2298
|
`Transferring motion from ${source} onto ${target}. ` +
|
|
1930
|
-
`Re-call with confirm=true after the user explicitly OKs the spend
|
|
2299
|
+
`Re-call with confirm=true after the user explicitly OKs the spend, or pick kling-mc-std to save ~10 credits. ` +
|
|
1931
2300
|
`When discussing with the user, refer to the assets by those codes — they'll match the gallery badges.`,
|
|
1932
2301
|
});
|
|
1933
2302
|
}
|
|
@@ -1945,27 +2314,12 @@ export const generateMotionTransfer = {
|
|
|
1945
2314
|
klingProvider: input.klingProvider,
|
|
1946
2315
|
estimatedCost: totalCents,
|
|
1947
2316
|
background: input.background,
|
|
1948
|
-
// Seedance engine passthrough — the desktop delegates to the seedance
|
|
1949
|
-
// ref-to-video path (vref billing, face cascade, consent gate).
|
|
1950
|
-
...(isSeedance
|
|
1951
|
-
? {
|
|
1952
|
-
duration: seedanceDuration,
|
|
1953
|
-
videoResolution: input.videoResolution,
|
|
1954
|
-
aspectRatio: input.aspectRatio,
|
|
1955
|
-
seedanceFace: input.seedanceFace !== false && input.seedanceRealFace !== true,
|
|
1956
|
-
seedanceRealFace: input.seedanceRealFace === true,
|
|
1957
|
-
realFaceConsent: input.realFaceConsent === true,
|
|
1958
|
-
// Ride the caller-supplied clip duration through — the asset row's
|
|
1959
|
-
// duration can be null for imported clips.
|
|
1960
|
-
sourceVideoDurationSeconds: input.sourceVideoSeconds,
|
|
1961
|
-
}
|
|
1962
|
-
: {}),
|
|
1963
2317
|
});
|
|
1964
2318
|
if (!result.success)
|
|
1965
2319
|
throw new Error(result.error ?? 'Motion transfer generation failed');
|
|
1966
2320
|
if (result.background) {
|
|
1967
2321
|
const ids = result.generationIds ?? (result.generationId ? [result.generationId] : []);
|
|
1968
|
-
return backgroundSubmitted(
|
|
2322
|
+
return backgroundSubmitted(`5s motion transfer (${motionModel})`, ids, {
|
|
1969
2323
|
variant: costKey,
|
|
1970
2324
|
motionModel,
|
|
1971
2325
|
projectId: input.projectId,
|
|
@@ -1976,7 +2330,7 @@ export const generateMotionTransfer = {
|
|
|
1976
2330
|
});
|
|
1977
2331
|
}
|
|
1978
2332
|
return {
|
|
1979
|
-
text: `Generated
|
|
2333
|
+
text: `Generated 5s motion transfer (${motionModel}) into project ${input.projectId} ` +
|
|
1980
2334
|
`for ${fmtCredits(totalCents)}.` +
|
|
1981
2335
|
(input.prompt ? ` Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` : ''),
|
|
1982
2336
|
data: {
|
|
@@ -1996,15 +2350,17 @@ export const generateMotionTransfer = {
|
|
|
1996
2350
|
// ── Edit video (Kling O3 video-to-video) ────────────────────────
|
|
1997
2351
|
export const editVideo = {
|
|
1998
2352
|
id: 'slates_edit_video',
|
|
1999
|
-
description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements)
|
|
2353
|
+
description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat), or seedance-2.5-edit (4–30s clips — the ONLY engine that takes a clip over 15s; 480p/720p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds — always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
|
|
2000
2354
|
input: z.object({
|
|
2001
2355
|
projectId: z.string().uuid().describe('Project the source clip lives in.'),
|
|
2002
2356
|
sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
|
|
2003
2357
|
prompt: z.string().min(1).max(2500).describe('The change, not the whole scene — e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
|
|
2004
|
-
model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs.'),
|
|
2358
|
+
model: z.enum(['kling-v3.0-omni-edit', 'kling-v3.0-omni-pro-edit', 'omni-flash-edit', 'seedance-2.5-edit']).optional().describe('Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s, 720p, 3–10s) for prompt-only edits — it takes NO character/style refs. seedance-2.5-edit (480p/720p, 4–30s) is the ONLY engine that accepts a clip longer than 15s; it also takes NO character/style refs on this op.'),
|
|
2005
2359
|
characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2006
2360
|
styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2007
|
-
keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit output
|
|
2361
|
+
keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
|
|
2362
|
+
videoResolution: z.enum(['480p', '720p']).optional().describe('seedance-2.5-edit ONLY (default 720p). Ignored by the Kling and Omni Flash engines, whose output follows the source clip.'),
|
|
2363
|
+
seedanceFace: z.boolean().optional().describe('seedance-2.5-edit ONLY: set true when a CHARACTER FACE is visible in the clip. Faceless edits run on BytePlus; faces are blocked there and must route to the relaxed provider, which costs ~35% more. A face edit submitted without this flag is rejected by the provider, not silently downgraded.'),
|
|
2008
2364
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
2009
2365
|
confirm: z.boolean().optional().describe('Set true to bypass the cost confirm gate after the user OKs the spend.'),
|
|
2010
2366
|
}),
|
|
@@ -2016,10 +2372,13 @@ export const editVideo = {
|
|
|
2016
2372
|
}
|
|
2017
2373
|
const model = input.model ?? 'kling-v3.0-omni-edit';
|
|
2018
2374
|
const isOmniFlashEdit = model === 'omni-flash-edit';
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
|
|
2375
|
+
const isSeedanceEdit = model === 'seedance-2.5-edit';
|
|
2376
|
+
// Kling edit: 3–15s source clips; Omni Flash edit: 3–10s; Seedance 2.5
|
|
2377
|
+
// edit: 4–30s — the only engine that takes a clip over 15s.
|
|
2378
|
+
const minClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MIN_SECONDS : 3;
|
|
2379
|
+
const maxClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MAX_SECONDS : isOmniFlashEdit ? 10 : 15;
|
|
2380
|
+
if ((isOmniFlashEdit || isSeedanceEdit) && ((input.characterAssetIds?.length ?? 0) > 0 || (input.styleAssetIds?.length ?? 0) > 0)) {
|
|
2381
|
+
throw new Error(`${model} takes the prompt and the source clip only on this op — no character/style reference images. Drop the refs, or switch to kling-v3.0-omni-edit which supports elements.`);
|
|
2023
2382
|
}
|
|
2024
2383
|
// Resolve refs (UUIDs or badge codes) against the project AT CALL TIME.
|
|
2025
2384
|
const refInputs = [
|
|
@@ -2048,12 +2407,24 @@ export const editVideo = {
|
|
|
2048
2407
|
if (!Number.isFinite(clipSeconds) || clipSeconds <= 0) {
|
|
2049
2408
|
throw new Error('Source clip has no recorded duration — cannot quote the edit. Re-import the clip or pick another.');
|
|
2050
2409
|
}
|
|
2051
|
-
if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds <
|
|
2052
|
-
throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts
|
|
2053
|
-
`Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip
|
|
2054
|
-
|
|
2055
|
-
|
|
2056
|
-
|
|
2410
|
+
if (clipSeconds > maxClipSeconds + 0.05 || clipSeconds < minClipSeconds - 0.05) {
|
|
2411
|
+
throw new Error(`Source clip is ${clipSeconds.toFixed(1)}s — ${model} accepts ${minClipSeconds}–${maxClipSeconds}s. ` +
|
|
2412
|
+
`Trim it first with slates_trim_video (e.g. inSec 0, outSec ${maxClipSeconds}), then edit the trimmed clip` +
|
|
2413
|
+
(maxClipSeconds < SEEDANCE_25_EDIT_MAX_SECONDS ? ', or switch to seedance-2.5-edit which accepts up to 30s' : '') +
|
|
2414
|
+
'.');
|
|
2415
|
+
}
|
|
2416
|
+
const billedSeconds = Math.min(maxClipSeconds, Math.max(minClipSeconds, Math.ceil(clipSeconds - 0.05)));
|
|
2417
|
+
// Three engines, three key shapes — only Seedance's carries a resolution and
|
|
2418
|
+
// a face route, because only its price moves with them.
|
|
2419
|
+
const costKey = isSeedanceEdit
|
|
2420
|
+
? seedanceEditCostKey({
|
|
2421
|
+
duration: billedSeconds,
|
|
2422
|
+
videoResolution: input.videoResolution ?? '720p',
|
|
2423
|
+
seedanceFace: input.seedanceFace === true,
|
|
2424
|
+
})
|
|
2425
|
+
: isOmniFlashEdit
|
|
2426
|
+
? omniFlashEditCostKey(billedSeconds)
|
|
2427
|
+
: klingEditCostKey(model, billedSeconds);
|
|
2057
2428
|
const cloud = ctx.cloud();
|
|
2058
2429
|
const registry = await cloud.get('/api/agent/models');
|
|
2059
2430
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -2092,6 +2463,8 @@ export const editVideo = {
|
|
|
2092
2463
|
characterAssetIds,
|
|
2093
2464
|
styleAssetIds,
|
|
2094
2465
|
keepAudio: input.keepAudio !== false,
|
|
2466
|
+
videoResolution: input.videoResolution,
|
|
2467
|
+
seedanceFace: input.seedanceFace,
|
|
2095
2468
|
background: input.background,
|
|
2096
2469
|
});
|
|
2097
2470
|
if (!result.success)
|
|
@@ -2694,6 +3067,53 @@ export const updateFrame = {
|
|
|
2694
3067
|
}));
|
|
2695
3068
|
},
|
|
2696
3069
|
};
|
|
3070
|
+
/**
|
|
3071
|
+
* Batch form of `slates_update_frame`.
|
|
3072
|
+
*
|
|
3073
|
+
* Writing a scene's worth of frames — motion prompts, shot labels — is ONE
|
|
3074
|
+
* logical edit. Doing it as N `slates_update_frame` calls costs N LLM
|
|
3075
|
+
* round-trips and makes the desktop refetch the whole storyboard N times for a
|
|
3076
|
+
* single intent. This is also where the storyboard's deleted "generate motion
|
|
3077
|
+
* prompts" button's capability went: the agent can be told "redo scene 3,
|
|
3078
|
+
* handheld", which that fixed-shape button never could.
|
|
3079
|
+
*
|
|
3080
|
+
* Hits `POST /agent/frames/batch-update`, which validates every id BEFORE the
|
|
3081
|
+
* first write and emits one broadcast for the batch.
|
|
3082
|
+
*/
|
|
3083
|
+
export const batchUpdateFrames = {
|
|
3084
|
+
id: 'slates_batch_update_frames',
|
|
3085
|
+
description: 'Update MANY frames in one call — motion prompts, shot labels, notes, asset binding, scene/position, frameType. Prefer this over repeated slates_update_frame when writing a scene or a whole storyboard: it is one round-trip and one UI refresh. Every id is validated before anything is written, so the batch never lands half-applied.',
|
|
3086
|
+
input: z.object({
|
|
3087
|
+
updates: z
|
|
3088
|
+
.array(z.object({
|
|
3089
|
+
frameId: z.string().uuid(),
|
|
3090
|
+
shotLabel: z.string().optional(),
|
|
3091
|
+
notes: z.string().optional(),
|
|
3092
|
+
assetId: z.string().uuid().nullable().optional(),
|
|
3093
|
+
sceneId: z.string().uuid().nullable().optional(),
|
|
3094
|
+
position: z.number().int().min(0).optional(),
|
|
3095
|
+
frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
|
|
3096
|
+
motionPrompt: z.string().nullable().optional(),
|
|
3097
|
+
}))
|
|
3098
|
+
.min(1),
|
|
3099
|
+
}),
|
|
3100
|
+
async run(input, ctx) {
|
|
3101
|
+
return ok(await ctx.desktop().post('/agent/frames/batch-update', {
|
|
3102
|
+
updates: input.updates.map((u) => ({
|
|
3103
|
+
id: u.frameId,
|
|
3104
|
+
data: {
|
|
3105
|
+
shotLabel: u.shotLabel,
|
|
3106
|
+
notes: u.notes,
|
|
3107
|
+
assetId: u.assetId,
|
|
3108
|
+
sceneId: u.sceneId,
|
|
3109
|
+
position: u.position,
|
|
3110
|
+
frameType: u.frameType,
|
|
3111
|
+
motionPrompt: u.motionPrompt,
|
|
3112
|
+
},
|
|
3113
|
+
})),
|
|
3114
|
+
}));
|
|
3115
|
+
},
|
|
3116
|
+
};
|
|
2697
3117
|
export const deleteFrame = {
|
|
2698
3118
|
id: 'slates_delete_frame',
|
|
2699
3119
|
description: 'Delete a frame from its scene (the referenced asset is untouched).',
|
|
@@ -2738,6 +3158,33 @@ function resolveGuideTopic(topic) {
|
|
|
2738
3158
|
return 'slates-prompting-kling-v3';
|
|
2739
3159
|
if (t.startsWith('kling-v3'))
|
|
2740
3160
|
return 'slates-prompting-kling-v3';
|
|
3161
|
+
// Audio — seed-audio BEFORE the seedance check: "seed-audio" also starts
|
|
3162
|
+
// with "seed", and falling through would hand the video guide to the audio
|
|
3163
|
+
// model (the exact class of aliasing bug this comment block warns about).
|
|
3164
|
+
// Speech, dialogue and scratch VO all live on Seed Audio now — the TTS
|
|
3165
|
+
// surface is gone, so "tts"/"voiceover" must NOT land on the ElevenLabs
|
|
3166
|
+
// guide, which is SFX-only.
|
|
3167
|
+
if (t.startsWith('seed-audio') ||
|
|
3168
|
+
t === 'seed audio' ||
|
|
3169
|
+
t === 'audio' ||
|
|
3170
|
+
t === 'tts' ||
|
|
3171
|
+
t === 'voiceover' ||
|
|
3172
|
+
t === 'dialogue') {
|
|
3173
|
+
return 'slates-prompting-seed-audio';
|
|
3174
|
+
}
|
|
3175
|
+
if (t.startsWith('eleven') || t.startsWith('elevenlabs') || t === 'sfx' || t === 'sound-effects' || t === 'sound effects') {
|
|
3176
|
+
return 'slates-prompting-elevenlabs';
|
|
3177
|
+
}
|
|
3178
|
+
// ⚠️ 2.5 BEFORE the generic `seedance` prefix — the same ordering trap as
|
|
3179
|
+
// seed-audio-before-seedance above. Falling through silently hands the 2.0
|
|
3180
|
+
// guide to 2.5, whose limits, resolutions and task types are all different.
|
|
3181
|
+
if (t.startsWith('seedance-2.5') ||
|
|
3182
|
+
t.startsWith('seedance-25') ||
|
|
3183
|
+
t.startsWith('seedance-2-5') ||
|
|
3184
|
+
t.startsWith('seedance2.5') ||
|
|
3185
|
+
t === 'seedance 2.5') {
|
|
3186
|
+
return 'slates-prompting-seedance-2-5';
|
|
3187
|
+
}
|
|
2741
3188
|
if (t.startsWith('seedance'))
|
|
2742
3189
|
return 'slates-prompting-seedance';
|
|
2743
3190
|
if (t.startsWith('avatar-') || t.includes('lip-sync'))
|
|
@@ -2764,7 +3211,7 @@ export const getPromptingGuide = {
|
|
|
2764
3211
|
topic: z
|
|
2765
3212
|
.string()
|
|
2766
3213
|
.min(1)
|
|
2767
|
-
.describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
|
|
3214
|
+
.describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
|
|
2768
3215
|
}),
|
|
2769
3216
|
async run(input) {
|
|
2770
3217
|
const resolved = resolveGuideTopic(input.topic);
|
|
@@ -2796,6 +3243,9 @@ export const ALL_OPERATIONS = [
|
|
|
2796
3243
|
listFolders,
|
|
2797
3244
|
createFolder,
|
|
2798
3245
|
moveAssetsToFolder,
|
|
3246
|
+
moveAssetsToProject,
|
|
3247
|
+
copyAssetsToProject,
|
|
3248
|
+
moveEntityToProject,
|
|
2799
3249
|
listCharacters,
|
|
2800
3250
|
createCharacter,
|
|
2801
3251
|
setCharacterIdentity,
|
|
@@ -2810,6 +3260,7 @@ export const ALL_OPERATIONS = [
|
|
|
2810
3260
|
addFrame,
|
|
2811
3261
|
generateImage,
|
|
2812
3262
|
generateVideo,
|
|
3263
|
+
generateAudio,
|
|
2813
3264
|
generateLipSync,
|
|
2814
3265
|
generateMotionTransfer,
|
|
2815
3266
|
editVideo,
|
|
@@ -2849,6 +3300,7 @@ export const ALL_OPERATIONS = [
|
|
|
2849
3300
|
deleteScene,
|
|
2850
3301
|
reorderScenes,
|
|
2851
3302
|
updateFrame,
|
|
3303
|
+
batchUpdateFrames,
|
|
2852
3304
|
deleteFrame,
|
|
2853
3305
|
getPromptingGuide,
|
|
2854
3306
|
];
|