@slatesvideo/shared 0.5.5 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +2 -2
- package/dist/operations/index.d.ts +81 -1
- package/dist/operations/index.js +435 -10
- package/dist/prompts/character-sheet.d.ts +2 -1
- package/dist/prompts/character-sheet.js +82 -13
- package/dist/prompts/model-facts.d.ts +1 -1
- package/dist/prompts/model-facts.js +32 -0
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +194 -0
- package/dist/prompts/reference-rules.d.ts +19 -2
- package/dist/prompts/reference-rules.js +18 -1
- package/dist/skills/content.js +5 -2
- package/exports/slates-prompt-builder/generated/reference-character.md +9 -4
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +6 -6
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +1 -1
- package/skills/slates-character-identity.md +9 -4
- package/skills/slates-model-selection.md +26 -0
- package/skills/slates-prompting-elevenlabs.md +131 -0
- package/skills/slates-prompting-seed-audio.md +110 -0
- package/skills/slates-prompting-suno.md +110 -0
package/dist/operations/index.js
CHANGED
|
@@ -120,8 +120,11 @@ export const listAvailableModels = {
|
|
|
120
120
|
const table = models.map((m) => `${m.model} ${creditCost(m)}`).join('\n');
|
|
121
121
|
return {
|
|
122
122
|
text: `${models.length} COST keys (credits per generation)${input.filter ? ` matching "${input.filter}"` : ''}. ` +
|
|
123
|
-
`NOTE: these are billing keys for cost lookup ONLY — the \`model\` param on
|
|
124
|
-
|
|
123
|
+
`NOTE: these are billing keys for cost lookup ONLY — the \`model\` param on the generate ops takes a BASE id. ` +
|
|
124
|
+
// Derived from the SSOT arrays, not restated: a hand-written list here
|
|
125
|
+
// drifts the moment a model lands (it already omitted omni-flash).
|
|
126
|
+
`slates_generate_video: ${VIDEO_MODELS.join(' | ')} (duration/videoResolution as separate params). ` +
|
|
127
|
+
`slates_generate_audio: ${AUDIO_MODELS.join(' | ')} (durationSeconds as a separate param):\n` +
|
|
125
128
|
table,
|
|
126
129
|
data: { count: models.length },
|
|
127
130
|
};
|
|
@@ -133,7 +136,8 @@ export const estimateGenerationCost = {
|
|
|
133
136
|
input: z.object({
|
|
134
137
|
model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
|
|
135
138
|
quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
|
|
136
|
-
duration: z.number().int().min(
|
|
139
|
+
duration: z.number().int().min(1).max(360).optional().describe('Seconds. Video 3-15 (cost scales linearly; required with a video base id). Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22. Ignored by eleven-v3 (per-character) and suno (flat).'),
|
|
140
|
+
characters: z.number().int().min(1).max(5000).optional().describe('eleven-v3 only — script length in characters, billed in 100-char buckets rounded up.'),
|
|
137
141
|
videoResolution: z.enum(['480p', '720p', '1080p', '4k']).optional().describe('Video only. Seedance defaults to 1080p.'),
|
|
138
142
|
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
|
|
139
143
|
quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
|
|
@@ -152,6 +156,52 @@ export const estimateGenerationCost = {
|
|
|
152
156
|
if (img)
|
|
153
157
|
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
|
|
154
158
|
}
|
|
159
|
+
// 2a) audio base id → seconds (seed-audio, eleven-sfx), characters
|
|
160
|
+
// (eleven-v3), or flat (suno). Runs BEFORE the video resolver: it is
|
|
161
|
+
// forgiving by design and "seed-audio" would otherwise be mistaken for
|
|
162
|
+
// a seedance spelling.
|
|
163
|
+
if (!key && AUDIO_MODELS.includes(input.model)) {
|
|
164
|
+
const m = input.model;
|
|
165
|
+
if ((m === 'seed-audio' || m === 'eleven-sfx') && !input.duration) {
|
|
166
|
+
return ok({
|
|
167
|
+
requires_clarification: true,
|
|
168
|
+
missing: ['duration'],
|
|
169
|
+
message: m === 'seed-audio'
|
|
170
|
+
? 'Seed Audio cost scales with the REQUESTED duration — and that duration is what the user is billed regardless of what comes back (it has no duration parameter; the number is written into the prompt). Pass duration in seconds (3-120).'
|
|
171
|
+
: 'Sound Effects bills per second — pass duration in seconds (1-22).',
|
|
172
|
+
});
|
|
173
|
+
}
|
|
174
|
+
if (m === 'eleven-v3' && !input.characters) {
|
|
175
|
+
return ok({
|
|
176
|
+
requires_clarification: true,
|
|
177
|
+
missing: ['characters'],
|
|
178
|
+
message: 'Eleven v3 bills per 100 characters of script, rounded up — pass characters (the length of the text you intend to speak).',
|
|
179
|
+
});
|
|
180
|
+
}
|
|
181
|
+
// REFUSE an out-of-range duration rather than quoting the clamped price.
|
|
182
|
+
// audioCostKey clamps (it has to — it mirrors the desktop, which clamps),
|
|
183
|
+
// so without this an agent asking for 2s of Seed Audio would be handed a
|
|
184
|
+
// real 3s price and no hint that 2s is not a thing it can order. The
|
|
185
|
+
// generate op gates the same way; the two must agree or the quote is a
|
|
186
|
+
// promise the generation refuses to keep.
|
|
187
|
+
const audioBounds = {
|
|
188
|
+
'seed-audio': { min: SEED_AUDIO_MIN_SECONDS, max: SEED_AUDIO_MAX_SECONDS },
|
|
189
|
+
'eleven-sfx': { min: ELEVEN_SFX_MIN_SECONDS, max: ELEVEN_SFX_MAX_SECONDS },
|
|
190
|
+
suno: { min: SUNO_MIN_SECONDS, max: SUNO_MAX_SECONDS },
|
|
191
|
+
};
|
|
192
|
+
const bounds = audioBounds[m];
|
|
193
|
+
if (bounds && input.duration != null && (input.duration < bounds.min || input.duration > bounds.max)) {
|
|
194
|
+
return ok({
|
|
195
|
+
requires_clarification: true,
|
|
196
|
+
missing: ['duration'],
|
|
197
|
+
message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
|
|
198
|
+
(m === 'suno'
|
|
199
|
+
? 'Suno duration is FREE and flat-priced, so any value in range costs the same.'
|
|
200
|
+
: 'Re-ask with a duration in range.'),
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
key = audioCostKey({ model: m, durationSeconds: input.duration, characters: input.characters });
|
|
204
|
+
}
|
|
155
205
|
// 2b) Kling O3 edit base id + duration (ceiled source-clip length)
|
|
156
206
|
if (!key && (input.model === 'kling-v3.0-omni-edit' || input.model === 'kling-v3.0-omni-pro-edit')) {
|
|
157
207
|
if (!input.duration) {
|
|
@@ -190,7 +240,7 @@ export const estimateGenerationCost = {
|
|
|
190
240
|
}
|
|
191
241
|
const perCredits = key != null ? byKey.get(key) : undefined;
|
|
192
242
|
if (key == null || perCredits == null) {
|
|
193
|
-
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
|
|
243
|
+
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/characters/resolution params, or use slates_list_available_models with a filter.`);
|
|
194
244
|
}
|
|
195
245
|
const qty = input.quantity ?? 1;
|
|
196
246
|
const totalCredits = perCredits * qty;
|
|
@@ -439,16 +489,16 @@ export const getAssetVideoFrames = {
|
|
|
439
489
|
};
|
|
440
490
|
export const uploadReferenceImage = {
|
|
441
491
|
id: 'slates_upload_reference_image',
|
|
442
|
-
description: 'Add a reference image
|
|
492
|
+
description: 'Add a reference image, video clip, or audio file to a Slates project from disk. Pass either filePath (absolute path to a local file) or dataUrl (base64 data: URL) — exactly one. Set type:"video" to bring in a clip (the user\'s own footage to edit/relocate/trim) or type:"audio" for music/VO/SFX they already have; both are probed on ingest, so duration (and for video, dimensions) are available immediately, and audio gets its waveform. Default type is "image". dataUrl is image-only.',
|
|
443
493
|
input: z
|
|
444
494
|
.object({
|
|
445
495
|
projectId: z.string().uuid(),
|
|
446
496
|
filePath: z.string().optional(),
|
|
447
497
|
dataUrl: z.string().optional(),
|
|
448
498
|
type: z
|
|
449
|
-
.enum(['image', 'video'])
|
|
499
|
+
.enum(['image', 'video', 'audio'])
|
|
450
500
|
.optional()
|
|
451
|
-
.describe('Asset kind for a filePath import — "image" (default) or "
|
|
501
|
+
.describe('Asset kind for a filePath import — "image" (default), "video", or "audio". A dataUrl is always an image.'),
|
|
452
502
|
})
|
|
453
503
|
.refine((d) => !!d.filePath !== !!d.dataUrl, {
|
|
454
504
|
message: 'Pass exactly one of filePath or dataUrl',
|
|
@@ -463,8 +513,8 @@ export const uploadReferenceImage = {
|
|
|
463
513
|
});
|
|
464
514
|
return ok(r);
|
|
465
515
|
}
|
|
466
|
-
if (input.type === 'video') {
|
|
467
|
-
throw new Error(
|
|
516
|
+
if (input.type === 'video' || input.type === 'audio') {
|
|
517
|
+
throw new Error(`dataUrl uploads are image-only — pass a filePath to import ${input.type === 'audio' ? 'an audio file' : 'a video clip'}.`);
|
|
468
518
|
}
|
|
469
519
|
const r = await desktop.post('/agent/assets/upload-base64', {
|
|
470
520
|
projectId: input.projectId,
|
|
@@ -506,6 +556,82 @@ export const moveAssetsToFolder = {
|
|
|
506
556
|
return ok(await ctx.desktop().post('/agent/folders/move-assets', input));
|
|
507
557
|
},
|
|
508
558
|
};
|
|
559
|
+
export const moveAssetsToProject = {
|
|
560
|
+
id: 'slates_move_assets_to_project',
|
|
561
|
+
description: 'Move assets (images, videos, audio) out of one project into another. The media files move on disk into the destination project folder — this is a real re-home, not a copy. Moved assets leave whatever gallery folder they were in and are issued fresh badge codes in the destination.',
|
|
562
|
+
input: z.object({
|
|
563
|
+
sourceProjectId: z.string().uuid(),
|
|
564
|
+
// Badge codes ("IMG-A8") resolve against sourceProjectId at call time.
|
|
565
|
+
assetIds: z.array(z.string().min(1)).min(1),
|
|
566
|
+
targetProjectId: z.string().uuid(),
|
|
567
|
+
}),
|
|
568
|
+
async run(input, ctx) {
|
|
569
|
+
if (input.sourceProjectId === input.targetProjectId) {
|
|
570
|
+
throw new Error('sourceProjectId and targetProjectId are the same — nothing to move.');
|
|
571
|
+
}
|
|
572
|
+
const resolved = await resolveAssetRefs(ctx, input.sourceProjectId, input.assetIds);
|
|
573
|
+
const assetIds = input.assetIds.map((ref) => resolved.get(ref)?.id ?? ref);
|
|
574
|
+
const result = await ctx.desktop().post('/agent/assets/move-to-project', {
|
|
575
|
+
assetIds,
|
|
576
|
+
targetProjectId: input.targetProjectId,
|
|
577
|
+
// Sent so the route can ENFORCE it. resolveAssetRefs only validates
|
|
578
|
+
// badge codes against the source project — a raw UUID passes straight
|
|
579
|
+
// through, so without this a caller could move an asset out of a
|
|
580
|
+
// project it never named.
|
|
581
|
+
sourceProjectId: input.sourceProjectId,
|
|
582
|
+
});
|
|
583
|
+
// Timeline clips key media by PATH, so the move repointed edits in OTHER
|
|
584
|
+
// projects at files that now live inside the destination — and deleting the
|
|
585
|
+
// destination deletes those files. Say it in the TEXT, not just the data:
|
|
586
|
+
// an agent summarising this result must be able to pass the warning on.
|
|
587
|
+
const refs = result.externalClipRefs ?? [];
|
|
588
|
+
if (refs.length > 0) {
|
|
589
|
+
const clips = refs.reduce((n, r) => n + r.clipCount, 0);
|
|
590
|
+
const where = refs.map((r) => `"${r.projectName}"`).join(', ');
|
|
591
|
+
return ok(result, `${JSON.stringify(result)}\n\n⚠️ ${clips} timeline clip(s) in ${where} now reference the moved ` +
|
|
592
|
+
`file(s) inside the destination project. Deleting the destination project would break those ` +
|
|
593
|
+
`edits. Tell the user — this is the one cross-project reference Slates allows.`);
|
|
594
|
+
}
|
|
595
|
+
return ok(result);
|
|
596
|
+
},
|
|
597
|
+
};
|
|
598
|
+
export const copyAssetsToProject = {
|
|
599
|
+
id: 'slates_copy_assets_to_project',
|
|
600
|
+
description: 'Copy assets (images, videos, audio) from one project into another. The originals stay exactly where they are — new files, new rows, new badge codes in the destination. Use this instead of slates_move_assets_to_project when the asset is already in use where it lives: an image that is a character/environment/style identity or sits in a storyboard frame CANNOT be moved out (that would leave the other project pointing at a file it no longer owns), but it can always be copied. Lineage is not copied; the copy starts clean.',
|
|
601
|
+
input: z.object({
|
|
602
|
+
sourceProjectId: z.string().uuid(),
|
|
603
|
+
// Badge codes ("IMG-A8") resolve against sourceProjectId at call time.
|
|
604
|
+
assetIds: z.array(z.string().min(1)).min(1),
|
|
605
|
+
targetProjectId: z.string().uuid(),
|
|
606
|
+
}),
|
|
607
|
+
async run(input, ctx) {
|
|
608
|
+
if (input.sourceProjectId === input.targetProjectId) {
|
|
609
|
+
throw new Error('sourceProjectId and targetProjectId are the same — nothing to copy.');
|
|
610
|
+
}
|
|
611
|
+
const resolved = await resolveAssetRefs(ctx, input.sourceProjectId, input.assetIds);
|
|
612
|
+
const assetIds = input.assetIds.map((ref) => resolved.get(ref)?.id ?? ref);
|
|
613
|
+
return ok(await ctx.desktop().post('/agent/assets/copy-to-project', {
|
|
614
|
+
assetIds,
|
|
615
|
+
targetProjectId: input.targetProjectId,
|
|
616
|
+
// Enforced route-side for the same reason as the move op: resolveAssetRefs
|
|
617
|
+
// only validates badge codes, so a raw UUID would otherwise let a caller
|
|
618
|
+
// copy out of a project it never named.
|
|
619
|
+
sourceProjectId: input.sourceProjectId,
|
|
620
|
+
}));
|
|
621
|
+
},
|
|
622
|
+
};
|
|
623
|
+
export const moveEntityToProject = {
|
|
624
|
+
id: 'slates_move_entity_to_project',
|
|
625
|
+
description: "Move a character, environment, or style into another project, taking every image it references with it. This is the fix when slates_move_assets_to_project refuses an asset because an entity still uses it: the identity image can't leave on its own, but the whole entity can. Refuses (rather than cascading) if one of its images is ALSO used by something else — copy the images instead in that case. Storyboard frames are not movable this way; moving a frame's image out would empty the shot.",
|
|
626
|
+
input: z.object({
|
|
627
|
+
kind: z.enum(['character', 'environment', 'style']),
|
|
628
|
+
entityId: z.string().uuid(),
|
|
629
|
+
targetProjectId: z.string().uuid(),
|
|
630
|
+
}),
|
|
631
|
+
async run(input, ctx) {
|
|
632
|
+
return ok(await ctx.desktop().post('/agent/entities/move-to-project', input));
|
|
633
|
+
},
|
|
634
|
+
};
|
|
509
635
|
// ── Characters ──────────────────────────────────────────────────
|
|
510
636
|
export const listCharacters = {
|
|
511
637
|
id: 'slates_list_characters',
|
|
@@ -1279,6 +1405,90 @@ export function klingEditCostKey(model, duration) {
|
|
|
1279
1405
|
export function omniFlashEditCostKey(duration) {
|
|
1280
1406
|
return `omni-flash-edit-${duration}s`;
|
|
1281
1407
|
}
|
|
1408
|
+
// ── Audio ───────────────────────────────────────────────────────
|
|
1409
|
+
// Exported: the exact `model` ids slates_generate_audio accepts — consumed
|
|
1410
|
+
// by the desktop Studio Agent system prompt (SSOT; never restate these ids
|
|
1411
|
+
// in prose that can drift). Mirrors VIDEO_MODELS for the third media type.
|
|
1412
|
+
export const AUDIO_MODELS = ['seed-audio', 'eleven-v3', 'eleven-sfx', 'suno'];
|
|
1413
|
+
/**
|
|
1414
|
+
* Per-surface bounds and defaults.
|
|
1415
|
+
*
|
|
1416
|
+
* 🚨 THESE FOUR NUMBERS PER SURFACE LIVE IN THREE REPOS. A change is a
|
|
1417
|
+
* three-site edit, every time:
|
|
1418
|
+
* 1. HERE (`audioCostKey`, the agent's pre-flight quote)
|
|
1419
|
+
* 2. `slate/src/shared/pricing.ts` → MODEL_REGISTRY `audio.durationSeconds`
|
|
1420
|
+
* (min/max/default), read by `clampAudioDuration` + `audioCreditKey`
|
|
1421
|
+
* 3. `slates-api/src/lib/audio-keys.ts` → the server's fail-closed bounds
|
|
1422
|
+
* `slates-api/scripts/pricing-consistency-check.mjs` §4 asserts 1 and 2 agree
|
|
1423
|
+
* at EVERY value including out-of-range ones; the gate check covers 3.
|
|
1424
|
+
*
|
|
1425
|
+
* The MINs used to be missing here, and the clamp floor was a hardcoded 1. That
|
|
1426
|
+
* made `slates_estimate_generation_cost({model:'seed-audio', duration:2})`
|
|
1427
|
+
* quote a real `seed-audio-2s` price for a generation the desktop would bill as
|
|
1428
|
+
* 3s and the proxy would REJECT outright. Same for an omitted duration, which
|
|
1429
|
+
* quoted `seed-audio-1s` against the desktop's `seed-audio-15s`.
|
|
1430
|
+
*/
|
|
1431
|
+
export const SEED_AUDIO_MIN_SECONDS = 3;
|
|
1432
|
+
export const SEED_AUDIO_MAX_SECONDS = 120;
|
|
1433
|
+
export const SEED_AUDIO_DEFAULT_SECONDS = 15;
|
|
1434
|
+
export const ELEVEN_SFX_MIN_SECONDS = 1;
|
|
1435
|
+
export const ELEVEN_SFX_MAX_SECONDS = 22;
|
|
1436
|
+
export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
|
|
1437
|
+
export const ELEVEN_V3_MAX_CHARACTERS = 5000;
|
|
1438
|
+
export const SUNO_MIN_SECONDS = 10;
|
|
1439
|
+
export const SUNO_MAX_SECONDS = 360;
|
|
1440
|
+
/**
|
|
1441
|
+
* Byte-for-byte the desktop's `clampAudioDuration` in slate/src/shared/pricing.ts,
|
|
1442
|
+
* INCLUDING the non-finite arm — that one matters: `Math.max(min, NaN)` is NaN,
|
|
1443
|
+
* so without it a NaN duration produces the key `seed-audio-NaNs` here while the
|
|
1444
|
+
* desktop quotes the default. Divergence at a value neither side can bill is
|
|
1445
|
+
* still divergence; the checker sweeps for it.
|
|
1446
|
+
*/
|
|
1447
|
+
function clampAudioSeconds(value, min, max, fallback) {
|
|
1448
|
+
if (!Number.isFinite(value))
|
|
1449
|
+
return fallback;
|
|
1450
|
+
return Math.min(max, Math.max(min, Math.ceil(value)));
|
|
1451
|
+
}
|
|
1452
|
+
function clampInt(value, min, max) {
|
|
1453
|
+
return Math.min(max, Math.max(min, Math.ceil(value)));
|
|
1454
|
+
}
|
|
1455
|
+
// Exported for scripts/pricing-consistency-check.mjs (slates-api repo), which
|
|
1456
|
+
// asserts this builder byte-matches the desktop's audioCreditKey().
|
|
1457
|
+
//
|
|
1458
|
+
// Seed Audio has NO duration parameter — length is driven by the prompt text.
|
|
1459
|
+
// We bill the REQUESTED duration (shape B, locked 2026-07-31): the desktop
|
|
1460
|
+
// injects "... N seconds" into the prompt and bills seed-audio-{N}s, so
|
|
1461
|
+
// display == billing with no amendment to the pricing law. The server probes
|
|
1462
|
+
// the returned audio.duration afterwards and logs SEED AUDIO BILLING DRIFT.
|
|
1463
|
+
export function audioCostKey(input) {
|
|
1464
|
+
if (input.model === 'seed-audio') {
|
|
1465
|
+
// `?? default` before the clamp, not `?? 0` — the desktop resolves a missing
|
|
1466
|
+
// duration to the registry DEFAULT, and a quote that doesn't match what the
|
|
1467
|
+
// desktop would bill is the whole bug class this mirrors away.
|
|
1468
|
+
const secs = clampAudioSeconds(input.durationSeconds ?? SEED_AUDIO_DEFAULT_SECONDS, SEED_AUDIO_MIN_SECONDS, SEED_AUDIO_MAX_SECONDS, SEED_AUDIO_DEFAULT_SECONDS);
|
|
1469
|
+
return `seed-audio-${secs}s`;
|
|
1470
|
+
}
|
|
1471
|
+
if (input.model === 'eleven-v3') {
|
|
1472
|
+
// 100-character buckets, always rounded UP — never under-bill a read.
|
|
1473
|
+
// Mirrors ttsBuckets() in slate/src/shared/pricing.ts, non-finite arm included.
|
|
1474
|
+
const chars = input.characters ?? 0;
|
|
1475
|
+
const buckets = Number.isFinite(chars)
|
|
1476
|
+
? Math.min(ELEVEN_V3_MAX_CHARACTERS / 100, Math.max(1, Math.ceil(chars / 100)))
|
|
1477
|
+
: 1;
|
|
1478
|
+
return `eleven-v3-tts-${buckets}00c`;
|
|
1479
|
+
}
|
|
1480
|
+
if (input.model === 'eleven-sfx') {
|
|
1481
|
+
const secs = clampAudioSeconds(input.durationSeconds ?? ELEVEN_SFX_DEFAULT_SECONDS, ELEVEN_SFX_MIN_SECONDS, ELEVEN_SFX_MAX_SECONDS, ELEVEN_SFX_DEFAULT_SECONDS);
|
|
1482
|
+
return `eleven-sfx-${secs}s`;
|
|
1483
|
+
}
|
|
1484
|
+
if (input.model === 'suno') {
|
|
1485
|
+
// Flat across every Suno model AND every duration up to 360s — measured
|
|
1486
|
+
// against the live sunoapi.org balance 2026-07-31 (12 credits for a
|
|
1487
|
+
// default call and for duration=240 alike). One call returns TWO songs.
|
|
1488
|
+
return 'suno-generate';
|
|
1489
|
+
}
|
|
1490
|
+
throw new Error(`Unknown audio model: ${input.model}`);
|
|
1491
|
+
}
|
|
1282
1492
|
/**
|
|
1283
1493
|
* Forgiving model-id resolver. Agents routinely paste registry COST keys
|
|
1284
1494
|
* ("kling-v3-standard-8s", "seedance-2-1080p-8s") into the `model` param —
|
|
@@ -1671,6 +1881,207 @@ export const generateVideo = {
|
|
|
1671
1881
|
};
|
|
1672
1882
|
},
|
|
1673
1883
|
};
|
|
1884
|
+
// ── Generate audio ──────────────────────────────────────────────
|
|
1885
|
+
export const generateAudio = {
|
|
1886
|
+
id: 'slates_generate_audio',
|
|
1887
|
+
description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Four surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s), eleven-v3 (verbatim text-to-speech in a named voice), eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop), suno (full music; every call returns TWO songs for one flat price). Which surface for which job: read the slates-model-selection skill. ' +
|
|
1888
|
+
'🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
|
|
1889
|
+
'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-suno). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
|
|
1890
|
+
'projectId is REQUIRED (no headless path). Cost > 17 credits returns requires_confirm — pass confirm=true after explicit user OK. No skill files installed? Call slates_get_prompting_guide first.',
|
|
1891
|
+
input: z.object({
|
|
1892
|
+
projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
|
|
1893
|
+
model: z
|
|
1894
|
+
.enum(AUDIO_MODELS)
|
|
1895
|
+
.describe('Audio surface. seed-audio = scene/ambience/beds (default choice), eleven-v3 = exact-script voiceover, eleven-sfx = one precise effect, suno = music. Routing doctrine: slates-model-selection skill.'),
|
|
1896
|
+
prompt: z
|
|
1897
|
+
.string()
|
|
1898
|
+
.min(1)
|
|
1899
|
+
.max(5000)
|
|
1900
|
+
.describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-v3: the SCRIPT, spoken verbatim — never put stage directions here. eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars. suno: a description in default mode, or the EXACT LYRICS when customMode=true and instrumental=false.'),
|
|
1901
|
+
durationSeconds: z
|
|
1902
|
+
.number()
|
|
1903
|
+
.optional()
|
|
1904
|
+
.describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic. suno 10-360, FREE (a 6-minute track costs the same as a default one) but only accepted on sunoModel=V5_5 with customMode=true. Ignored by eleven-v3, which bills per 100 characters of text.'),
|
|
1905
|
+
voice: z
|
|
1906
|
+
.string()
|
|
1907
|
+
.optional()
|
|
1908
|
+
.describe('seed-audio: a preset voice id (e.g. "cedric_en_zh") — leave unset to let the scene cast itself, which is usually right for background dialogue. eleven-v3: a preset name (Rachel default; Aria, Roger, Sarah, Laura, Charlie, George, Callum, River, Liam, Charlotte, Alice, Matilda, Will, Jessica, Eric, Chris, Brian, Daniel, Lily, Bill). Pick one and keep it for the whole piece. No voice cloning on this route.'),
|
|
1909
|
+
stability: z.number().min(0).max(1).optional().describe('eleven-v3 only. 0-1, default 0.5. Lower = more expressive and more variable take-to-take; higher = flatter and repeatable. Raise it for long narration.'),
|
|
1910
|
+
languageCode: z.string().optional().describe('eleven-v3 only — ISO 639-1 code to force a language when the text is ambiguous or code-switched.'),
|
|
1911
|
+
speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
|
|
1912
|
+
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
|
|
1913
|
+
pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
|
|
1914
|
+
multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
|
|
1915
|
+
loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
|
|
1916
|
+
promptInfluence: z.number().min(0).max(1).optional().describe('eleven-sfx only — 0-1, default 0.3. Higher hugs your wording with less variation between takes.'),
|
|
1917
|
+
audioReferenceAssetIds: z
|
|
1918
|
+
.array(z.string())
|
|
1919
|
+
.max(3)
|
|
1920
|
+
.optional()
|
|
1921
|
+
.describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId — the API rejects both.'),
|
|
1922
|
+
imageReferenceAssetId: z
|
|
1923
|
+
.string()
|
|
1924
|
+
.optional()
|
|
1925
|
+
.describe('seed-audio only — ONE image asset to score what is in frame. MUTUALLY EXCLUSIVE with audioReferenceAssetIds.'),
|
|
1926
|
+
sunoModel: z.enum(['V4', 'V4_5', 'V4_5PLUS', 'V4_5ALL', 'V5', 'V5_5']).optional().describe('suno only — wire model id (underscored). Default V5. Cost is FLAT across every version. duration needs V5_5.'),
|
|
1927
|
+
customMode: z
|
|
1928
|
+
.boolean()
|
|
1929
|
+
.optional()
|
|
1930
|
+
.describe('suno only. false (default) = prompt is a ≤500-char DESCRIPTION and lyrics get written for you. true = style + title required, and prompt becomes the EXACT LYRICS, sung as written. Putting a description in the prompt while customMode=true wastes a full generation.'),
|
|
1931
|
+
instrumental: z.boolean().optional().describe('suno only — score with no vocals. Usually right for a film bed: an unasked-for vocal fights dialogue.'),
|
|
1932
|
+
style: z.string().optional().describe('suno only — genre + era + instrumentation + tempo ("90s trip-hop, dusty breakbeat, Rhodes, 85 bpm"). Required in customMode. Steer here, not by piling adjectives into the prompt.'),
|
|
1933
|
+
title: z.string().optional().describe('suno only — track title. Required in customMode.'),
|
|
1934
|
+
negativeTags: z.string().optional().describe('suno only — comma-separated things to keep out ("brass, EDM drop, male vocal").'),
|
|
1935
|
+
vocalGender: z.enum(['m', 'f']).optional().describe('suno only — the WIRE values m/f, not "male"/"female".'),
|
|
1936
|
+
styleWeight: z.number().min(0).max(1).optional().describe('suno only — 0-1, how hard the track hugs `style`. Higher = more genre-obedient and more generic; lower = more room to surprise you.'),
|
|
1937
|
+
weirdnessConstraint: z.number().min(0).max(1).optional().describe('suno only — 0-1 experimentation dial. Low is safe and on-brief; high wanders. Leave unset for a film bed.'),
|
|
1938
|
+
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE + ' Recommended for suno (2-3 min renders).'),
|
|
1939
|
+
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate after explicit user OK.'),
|
|
1940
|
+
}),
|
|
1941
|
+
run: async (input, ctx) => {
|
|
1942
|
+
// ── Per-surface clarification + constraint gates ──
|
|
1943
|
+
const cfgDefaults = {
|
|
1944
|
+
'seed-audio': 15,
|
|
1945
|
+
'eleven-sfx': 4,
|
|
1946
|
+
'eleven-v3': undefined,
|
|
1947
|
+
suno: undefined,
|
|
1948
|
+
};
|
|
1949
|
+
const seconds = input.durationSeconds ?? cfgDefaults[input.model];
|
|
1950
|
+
if (input.model === 'seed-audio') {
|
|
1951
|
+
if (seconds == null || seconds < 3 || seconds > SEED_AUDIO_MAX_SECONDS) {
|
|
1952
|
+
return ok({
|
|
1953
|
+
requires_clarification: true,
|
|
1954
|
+
missing: ['durationSeconds'],
|
|
1955
|
+
message: `Seed Audio needs a durationSeconds of 3-${SEED_AUDIO_MAX_SECONDS}. It has NO duration parameter — the number is written into the prompt AND is what the user is billed, so it must be a deliberate choice. Ask the user how long the bed should be (a few seconds longer than the clip it sits under, so the edit has handles).`,
|
|
1956
|
+
});
|
|
1957
|
+
}
|
|
1958
|
+
if ((input.audioReferenceAssetIds?.length ?? 0) > 0 && input.imageReferenceAssetId) {
|
|
1959
|
+
throw new Error('Seed Audio takes audio references OR one image reference, never both — the provider rejects the combination.');
|
|
1960
|
+
}
|
|
1961
|
+
}
|
|
1962
|
+
if (input.model === 'eleven-sfx') {
|
|
1963
|
+
if (seconds == null || seconds < 1 || seconds > ELEVEN_SFX_MAX_SECONDS) {
|
|
1964
|
+
return ok({
|
|
1965
|
+
requires_clarification: true,
|
|
1966
|
+
missing: ['durationSeconds'],
|
|
1967
|
+
message: `Sound Effects needs a durationSeconds of 1-${ELEVEN_SFX_MAX_SECONDS}. It is billed per second and is never left for the model to pick (that would make the charge non-deterministic). Roughly: 0.5-1s for an impact, 2-4s for a whoosh, 8-22s for a loopable bed.`,
|
|
1968
|
+
});
|
|
1969
|
+
}
|
|
1970
|
+
if (input.prompt.length > 450) {
|
|
1971
|
+
throw new Error(`Sound Effects accepts up to 450 characters — this prompt is ${input.prompt.length}.`);
|
|
1972
|
+
}
|
|
1973
|
+
}
|
|
1974
|
+
if (input.model === 'eleven-v3' && input.prompt.length > ELEVEN_V3_MAX_CHARACTERS) {
|
|
1975
|
+
throw new Error(`Eleven v3 accepts up to ${ELEVEN_V3_MAX_CHARACTERS} characters — this script is ${input.prompt.length}.`);
|
|
1976
|
+
}
|
|
1977
|
+
if (input.model === 'suno' && input.customMode === true) {
|
|
1978
|
+
if (!input.style || !input.title) {
|
|
1979
|
+
return ok({
|
|
1980
|
+
requires_clarification: true,
|
|
1981
|
+
missing: [...(input.style ? [] : ['style']), ...(input.title ? [] : ['title'])],
|
|
1982
|
+
message: 'Suno custom mode requires style and title. Remember that in custom mode the prompt field is the EXACT LYRICS (unless instrumental=true, where it is ignored) — if you meant to describe a mood, use customMode=false instead.',
|
|
1983
|
+
});
|
|
1984
|
+
}
|
|
1985
|
+
}
|
|
1986
|
+
await ctx.desktop().requireCapability('audio-generation', 'audio generation');
|
|
1987
|
+
// ── Resolve asset refs at CALL time (UUIDs or badge codes) ──
|
|
1988
|
+
const refInputs = [];
|
|
1989
|
+
for (const ref of input.audioReferenceAssetIds ?? [])
|
|
1990
|
+
refInputs.push({ ref, role: 'audio reference' });
|
|
1991
|
+
if (input.imageReferenceAssetId)
|
|
1992
|
+
refInputs.push({ ref: input.imageReferenceAssetId, role: 'image reference' });
|
|
1993
|
+
const resolvedRefs = refInputs.length > 0
|
|
1994
|
+
? await resolveAssetRefs(ctx, input.projectId, refInputs.map((r) => r.ref))
|
|
1995
|
+
: new Map();
|
|
1996
|
+
const rid = (v) => (v ? (resolvedRefs.get(v)?.id ?? v) : v);
|
|
1997
|
+
const refEcho = refInputs.length > 0 ? describeResolvedRefs(refInputs, resolvedRefs) : '';
|
|
1998
|
+
// ── Cost — from the CLOUD registry, keyed by the SAME builder the desktop
|
|
1999
|
+
// and the proxy use. Never quote a price from memory. ──
|
|
2000
|
+
const cloud = ctx.cloud();
|
|
2001
|
+
const registry = await cloud.get('/api/agent/models');
|
|
2002
|
+
const costKey = audioCostKey({
|
|
2003
|
+
model: input.model,
|
|
2004
|
+
durationSeconds: seconds,
|
|
2005
|
+
characters: input.prompt.length,
|
|
2006
|
+
});
|
|
2007
|
+
const entry = registry.models.find((m) => m.model === costKey);
|
|
2008
|
+
if (!entry) {
|
|
2009
|
+
throw new Error(`Audio variant not in registry: ${costKey}. Available audio models: ${AUDIO_MODELS.join(' | ')}.`);
|
|
2010
|
+
}
|
|
2011
|
+
const totalCents = creditCost(entry);
|
|
2012
|
+
if (!input.confirm && totalCents > CONFIRM_CREDITS) {
|
|
2013
|
+
return ok({
|
|
2014
|
+
requires_confirm: true,
|
|
2015
|
+
model: input.model,
|
|
2016
|
+
variant: costKey,
|
|
2017
|
+
cost_credits: totalCents,
|
|
2018
|
+
cost_display: fmtCredits(totalCents),
|
|
2019
|
+
message: `${input.model} audio will cost ${fmtCredits(totalCents)}. ` +
|
|
2020
|
+
(input.model === 'seed-audio'
|
|
2021
|
+
? `You are billed for the ${seconds}s you requested regardless of the returned length. `
|
|
2022
|
+
: '') +
|
|
2023
|
+
(input.model === 'suno' ? 'This returns TWO songs for that one price. ' : '') +
|
|
2024
|
+
'Confirm with the user, then call again with confirm: true.',
|
|
2025
|
+
});
|
|
2026
|
+
}
|
|
2027
|
+
// ── Submit through the DESKTOP so the progress card, the project save,
|
|
2028
|
+
// and recovery all behave exactly like a UI-triggered run. ──
|
|
2029
|
+
const desktop = ctx.desktop();
|
|
2030
|
+
if (input.background) {
|
|
2031
|
+
await desktop.requireCapability('background-generation', 'background generation');
|
|
2032
|
+
}
|
|
2033
|
+
const result = await desktop.post('/agent/generation/audio', {
|
|
2034
|
+
projectId: input.projectId,
|
|
2035
|
+
model: input.model,
|
|
2036
|
+
prompt: input.prompt,
|
|
2037
|
+
durationSeconds: seconds,
|
|
2038
|
+
voice: input.voice,
|
|
2039
|
+
stability: input.stability,
|
|
2040
|
+
languageCode: input.languageCode,
|
|
2041
|
+
speed: input.speed,
|
|
2042
|
+
volume: input.volume,
|
|
2043
|
+
pitch: input.pitch,
|
|
2044
|
+
multilingual: input.multilingual,
|
|
2045
|
+
loop: input.loop,
|
|
2046
|
+
promptInfluence: input.promptInfluence,
|
|
2047
|
+
audioReferenceAssetIds: (input.audioReferenceAssetIds ?? []).map((r) => rid(r)),
|
|
2048
|
+
imageReferenceAssetId: rid(input.imageReferenceAssetId),
|
|
2049
|
+
sunoModel: input.sunoModel,
|
|
2050
|
+
customMode: input.customMode,
|
|
2051
|
+
instrumental: input.instrumental,
|
|
2052
|
+
style: input.style,
|
|
2053
|
+
title: input.title,
|
|
2054
|
+
negativeTags: input.negativeTags,
|
|
2055
|
+
vocalGender: input.vocalGender,
|
|
2056
|
+
styleWeight: input.styleWeight,
|
|
2057
|
+
weirdnessConstraint: input.weirdnessConstraint,
|
|
2058
|
+
background: input.background,
|
|
2059
|
+
});
|
|
2060
|
+
if (!result.success)
|
|
2061
|
+
throw new Error(result.error ?? 'Audio generation failed');
|
|
2062
|
+
if (result.background) {
|
|
2063
|
+
return backgroundSubmitted(`${input.model} audio generation`, [result.generationId].filter(Boolean), { model: input.model, variant: costKey, projectId: input.projectId, cost_credits: totalCents }, refEcho);
|
|
2064
|
+
}
|
|
2065
|
+
const siblings = result.siblingAssets ?? [];
|
|
2066
|
+
return {
|
|
2067
|
+
text: `Generated ${input.model} audio into project ${input.projectId} for ${fmtCredits(totalCents)}` +
|
|
2068
|
+
(siblings.length > 0 ? ` (${siblings.length + 1} tracks — Suno returns two variations)` : '') +
|
|
2069
|
+
`. Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
|
|
2070
|
+
(refEcho ? ` ${refEcho}` : ''),
|
|
2071
|
+
data: {
|
|
2072
|
+
model: input.model,
|
|
2073
|
+
variant: costKey,
|
|
2074
|
+
projectId: input.projectId,
|
|
2075
|
+
durationSeconds: seconds,
|
|
2076
|
+
cost_cents: totalCents,
|
|
2077
|
+
cost_credits: totalCents,
|
|
2078
|
+
asset: result.asset,
|
|
2079
|
+
...(siblings.length > 0 ? { siblingAssets: siblings } : {}),
|
|
2080
|
+
generationId: result.generationId,
|
|
2081
|
+
},
|
|
2082
|
+
};
|
|
2083
|
+
},
|
|
2084
|
+
};
|
|
1674
2085
|
// ── Generate lip-sync ───────────────────────────────────────────
|
|
1675
2086
|
export const generateLipSync = {
|
|
1676
2087
|
id: 'slates_generate_lip_sync',
|
|
@@ -2738,6 +3149,16 @@ function resolveGuideTopic(topic) {
|
|
|
2738
3149
|
return 'slates-prompting-kling-v3';
|
|
2739
3150
|
if (t.startsWith('kling-v3'))
|
|
2740
3151
|
return 'slates-prompting-kling-v3';
|
|
3152
|
+
// Audio — seed-audio BEFORE the seedance check: "seed-audio" also starts
|
|
3153
|
+
// with "seed", and falling through would hand the video guide to the audio
|
|
3154
|
+
// model (the exact class of aliasing bug this comment block warns about).
|
|
3155
|
+
if (t.startsWith('seed-audio') || t === 'seed audio' || t === 'audio')
|
|
3156
|
+
return 'slates-prompting-seed-audio';
|
|
3157
|
+
if (t.startsWith('eleven') || t.startsWith('elevenlabs') || t === 'tts' || t === 'sfx' || t === 'sound-effects' || t === 'sound effects' || t === 'voiceover') {
|
|
3158
|
+
return 'slates-prompting-elevenlabs';
|
|
3159
|
+
}
|
|
3160
|
+
if (t.startsWith('suno') || t === 'music')
|
|
3161
|
+
return 'slates-prompting-suno';
|
|
2741
3162
|
if (t.startsWith('seedance'))
|
|
2742
3163
|
return 'slates-prompting-seedance';
|
|
2743
3164
|
if (t.startsWith('avatar-') || t.includes('lip-sync'))
|
|
@@ -2764,7 +3185,7 @@ export const getPromptingGuide = {
|
|
|
2764
3185
|
topic: z
|
|
2765
3186
|
.string()
|
|
2766
3187
|
.min(1)
|
|
2767
|
-
.describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
|
|
3188
|
+
.describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-suno, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
|
|
2768
3189
|
}),
|
|
2769
3190
|
async run(input) {
|
|
2770
3191
|
const resolved = resolveGuideTopic(input.topic);
|
|
@@ -2796,6 +3217,9 @@ export const ALL_OPERATIONS = [
|
|
|
2796
3217
|
listFolders,
|
|
2797
3218
|
createFolder,
|
|
2798
3219
|
moveAssetsToFolder,
|
|
3220
|
+
moveAssetsToProject,
|
|
3221
|
+
copyAssetsToProject,
|
|
3222
|
+
moveEntityToProject,
|
|
2799
3223
|
listCharacters,
|
|
2800
3224
|
createCharacter,
|
|
2801
3225
|
setCharacterIdentity,
|
|
@@ -2810,6 +3234,7 @@ export const ALL_OPERATIONS = [
|
|
|
2810
3234
|
addFrame,
|
|
2811
3235
|
generateImage,
|
|
2812
3236
|
generateVideo,
|
|
3237
|
+
generateAudio,
|
|
2813
3238
|
generateLipSync,
|
|
2814
3239
|
generateMotionTransfer,
|
|
2815
3240
|
editVideo,
|
|
@@ -9,7 +9,8 @@
|
|
|
9
9
|
* that cannot match the portrait's, so the sheet would carry two competing
|
|
10
10
|
* identities and the model averages them. A back view has no face to compete
|
|
11
11
|
* with, and it is the only panel where hair fall reads. See the header comment
|
|
12
|
-
* for the receipt and for why the phrasing must stay
|
|
12
|
+
* for the receipt, and for why the phrasing must stay FRAMING rather than
|
|
13
|
+
* removal and must be SCOPED TO THE FACE rather than to the whole body.
|
|
13
14
|
*/
|
|
14
15
|
export declare const CHARACTER_SHEET_PANELS_DESC: string;
|
|
15
16
|
/** Panel identifiers, in sheet order. */
|