@slatesvideo/shared 0.6.4 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api-url.d.ts +5 -3
- package/dist/api-url.js +5 -3
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/manual/content.d.ts +2 -0
- package/dist/manual/content.js +3 -0
- package/dist/manual/index.d.ts +5 -0
- package/dist/manual/index.js +20 -0
- package/dist/operations/index.d.ts +21 -10
- package/dist/operations/index.js +145 -41
- package/dist/prompts/agent-doctrine.js +1 -0
- package/dist/prompts/shot-spec.d.ts +21 -8
- package/dist/prompts/shot-spec.js +24 -8
- package/dist/skills/content.js +5 -5
- package/package.json +1 -1
- package/skills/slates-model-selection.md +3 -2
- package/skills/slates-prompting-elevenlabs.md +1 -1
- package/skills/slates-prompting-inworld-tts.md +174 -166
- package/skills/slates-prompting-lip-sync.md +1 -1
- package/skills/slates-prompting-seed-audio.md +1 -1
package/dist/operations/index.js
CHANGED
|
@@ -14,6 +14,7 @@ import { SlatesCloudClient } from '../clients/cloud.js';
|
|
|
14
14
|
import { SlatesDesktopClient } from '../clients/desktop.js';
|
|
15
15
|
import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
|
|
16
16
|
import { SKILLS } from '../skills/content.js';
|
|
17
|
+
import { appManualSections } from '../manual/index.js';
|
|
17
18
|
// Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
|
|
18
19
|
// "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
|
|
19
20
|
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
|
|
@@ -183,6 +184,15 @@ export const TTS_MAX_CHARACTERS = (() => {
|
|
|
183
184
|
return max;
|
|
184
185
|
})();
|
|
185
186
|
export const TTS_BUCKET_COUNT = TTS_MAX_CHARACTERS / TTS_BUCKET_CHARS; // 8
|
|
187
|
+
/** The seat's cloning spec — reference-clip bounds, the design-prompt bounds
|
|
188
|
+
* and the workspace-wide clone rate — read from the SSOT for the same reason
|
|
189
|
+
* as the cap: every number in it was MEASURED and lives in exactly one row. */
|
|
190
|
+
export const TTS_VOICE_CLONE = (() => {
|
|
191
|
+
const spec = getModelCapability(TTS_MODEL)?.voiceClone;
|
|
192
|
+
if (!spec)
|
|
193
|
+
throw new Error(`MODEL_CAPABILITIES['${TTS_MODEL}'] must declare voiceClone`);
|
|
194
|
+
return spec;
|
|
195
|
+
})();
|
|
186
196
|
function creditCost(m) {
|
|
187
197
|
if (!m)
|
|
188
198
|
return 0;
|
|
@@ -219,7 +229,7 @@ const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it
|
|
|
219
229
|
const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
|
|
220
230
|
'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
|
|
221
231
|
const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
|
|
222
|
-
'slates_get_asset_video_frames for video.';
|
|
232
|
+
'slates_get_asset_video_frames for video. For audio, audition the saved file; metadata alone does not establish voice similarity or delivery quality.';
|
|
223
233
|
// The image saved, but reading it back off disk failed (best-effort fetch). The
|
|
224
234
|
// agent has an asset and NO pixels, which is the one state where a quality
|
|
225
235
|
// claim would be pure invention — so this branch has to say so rather than
|
|
@@ -2727,14 +2737,68 @@ export const generateVideo = {
|
|
|
2727
2737
|
};
|
|
2728
2738
|
},
|
|
2729
2739
|
};
|
|
2740
|
+
// ── The preset voice shelf ──────────────────────────────────────
|
|
2741
|
+
/**
|
|
2742
|
+
* AGENT PARITY for the voice picker. The desktop browses stock voices by
|
|
2743
|
+
* gender, accent and age and plays each one; until this op the agent could
|
|
2744
|
+
* only pass a `voiceId` it had no way to discover. Disk reads on the desktop,
|
|
2745
|
+
* never a vendor call — browsing is free on every surface.
|
|
2746
|
+
*/
|
|
2747
|
+
export const listVoices = {
|
|
2748
|
+
id: 'slates_list_voices',
|
|
2749
|
+
description: `Browse ${TTS_MODEL} preset voices. Pass a returned voiceId to slates_generate_audio. Filters AND together.`,
|
|
2750
|
+
input: z.object({
|
|
2751
|
+
gender: z.string().optional().describe('male | female'),
|
|
2752
|
+
accent: z.string().optional().describe('Region ("GB") or languageCode ("en-GB"); available accents come from the shelf.'),
|
|
2753
|
+
age: z.string().optional().describe('young | middle_aged | elderly'),
|
|
2754
|
+
query: z.string().optional().describe('Free text over name, description, tags.'),
|
|
2755
|
+
}),
|
|
2756
|
+
async run(input, ctx) {
|
|
2757
|
+
const desktop = ctx.desktop();
|
|
2758
|
+
await desktop.requireCapability('voices', 'the preset voice shelf');
|
|
2759
|
+
const shelf = await desktop.get('/agent/voices');
|
|
2760
|
+
if (!shelf.available) {
|
|
2761
|
+
return ok({ available: false, voices: [], message: 'This desktop build shipped without the preset shelf.' });
|
|
2762
|
+
}
|
|
2763
|
+
const terms = (input.query ?? '').toLowerCase().split(/\s+/).filter(Boolean);
|
|
2764
|
+
const region = input.accent?.includes('-') ? input.accent.split('-')[1] : input.accent;
|
|
2765
|
+
const voices = shelf.voices
|
|
2766
|
+
.filter((v) => !input.gender || v.gender === input.gender)
|
|
2767
|
+
.filter((v) => !region || v.languageCode.split('-')[1]?.toUpperCase() === region.toUpperCase())
|
|
2768
|
+
.filter((v) => !input.age || v.ageGroup === input.age)
|
|
2769
|
+
.filter((v) => {
|
|
2770
|
+
if (terms.length === 0)
|
|
2771
|
+
return true;
|
|
2772
|
+
const hay = [v.displayName, v.description, v.gender, v.ageGroup, v.languageCode, ...(v.tags ?? [])]
|
|
2773
|
+
.join(' ')
|
|
2774
|
+
.toLowerCase();
|
|
2775
|
+
return terms.every((t) => hay.includes(t));
|
|
2776
|
+
})
|
|
2777
|
+
.map(({ voiceId, displayName, description, tags, gender, ageGroup, languageCode }) => ({
|
|
2778
|
+
voiceId,
|
|
2779
|
+
displayName,
|
|
2780
|
+
description,
|
|
2781
|
+
tags,
|
|
2782
|
+
gender,
|
|
2783
|
+
ageGroup,
|
|
2784
|
+
languageCode,
|
|
2785
|
+
}));
|
|
2786
|
+
return ok({
|
|
2787
|
+
available: true,
|
|
2788
|
+
line: shelf.line,
|
|
2789
|
+
count: voices.length,
|
|
2790
|
+
voices,
|
|
2791
|
+
next: `Pass a voiceId to slates_generate_audio (model ${TTS_MODEL}) as voiceId. To keep one on a character for reuse, generate a clip with it and set slates_update_character voiceAssetId.`,
|
|
2792
|
+
});
|
|
2793
|
+
},
|
|
2794
|
+
};
|
|
2730
2795
|
// ── Generate audio ──────────────────────────────────────────────
|
|
2731
2796
|
export const generateAudio = {
|
|
2732
2797
|
id: 'slates_generate_audio',
|
|
2733
2798
|
billable: true,
|
|
2734
|
-
description:
|
|
2735
|
-
'
|
|
2736
|
-
'
|
|
2737
|
-
'projectId is REQUIRED (no headless path). ' +
|
|
2799
|
+
description: `Generate project audio using credits. Choose the surface via the model routing below. ` +
|
|
2800
|
+
'Read slates-cost-discipline and the matching prompting skill first (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). ' +
|
|
2801
|
+
'Seed Audio bills the requested duration, which is appended to the prompt regardless of output length. Kling "SFX:" / "Ambient noise:" syntax does not transfer. ' +
|
|
2738
2802
|
CONFIRM_GATE_SENTENCE +
|
|
2739
2803
|
' No skill files installed? Call slates_get_prompting_guide first.',
|
|
2740
2804
|
input: z.object({
|
|
@@ -2753,25 +2817,27 @@ export const generateAudio = {
|
|
|
2753
2817
|
durationSeconds: z
|
|
2754
2818
|
.number()
|
|
2755
2819
|
.optional()
|
|
2756
|
-
.describe(
|
|
2820
|
+
.describe(`seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (default ${SEED_AUDIO_DEFAULT_SECONDS}) — ⚠️ THIS IS THE BILL: appended to the prompt and charged whatever comes back. eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS} (default ${ELEVEN_SFX_DEFAULT_SECONDS}) — always sent explicitly so the per-second charge is deterministic. Not for ${TTS_MODEL}.`),
|
|
2757
2821
|
voice: z
|
|
2758
2822
|
.string()
|
|
2759
2823
|
.optional()
|
|
2760
|
-
.describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself
|
|
2824
|
+
.describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself. Agent-facing only.'),
|
|
2761
2825
|
voiceId: z
|
|
2762
2826
|
.string()
|
|
2763
2827
|
.optional()
|
|
2764
|
-
.describe(
|
|
2828
|
+
.describe(`${TTS_MODEL} — a preset voiceId from slates_list_voices, not a character or asset id. Exactly one voice source is required.`),
|
|
2765
2829
|
voiceReferenceAssetId: z
|
|
2766
2830
|
.string()
|
|
2767
2831
|
.optional()
|
|
2768
|
-
.describe(
|
|
2832
|
+
.describe(`${TTS_MODEL} — clone this AUDIO asset's voice for the take (${TTS_VOICE_CLONE.minSeconds}-${TTS_VOICE_CLONE.maxSeconds}s, one clean speaker). To speak AS a character pass its voiceAssetId. Cloning: ${TTS_VOICE_CLONE.clonesPerMinute} new voices/min across all of Slates; a burst waits.`),
|
|
2769
2833
|
voiceDescription: z
|
|
2770
2834
|
.string()
|
|
2835
|
+
.min(TTS_VOICE_CLONE.designPromptChars.min)
|
|
2836
|
+
.max(TTS_VOICE_CLONE.designPromptChars.max)
|
|
2771
2837
|
.optional()
|
|
2772
|
-
.describe(
|
|
2838
|
+
.describe(`${TTS_MODEL} — a voice from words (${TTS_VOICE_CLONE.designPromptChars.min}-${TTS_VOICE_CLONE.designPromptChars.max} chars) for a character with no recording; keep it via slates_update_character voiceAssetId.`),
|
|
2773
2839
|
speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
|
|
2774
|
-
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline
|
|
2840
|
+
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline fader for mix decisions.'),
|
|
2775
2841
|
pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
|
|
2776
2842
|
multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
|
|
2777
2843
|
loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
|
|
@@ -2780,7 +2846,7 @@ export const generateAudio = {
|
|
|
2780
2846
|
.array(z.string())
|
|
2781
2847
|
.max(3)
|
|
2782
2848
|
.optional()
|
|
2783
|
-
.describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId
|
|
2849
|
+
.describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId.'),
|
|
2784
2850
|
imageReferenceAssetId: z
|
|
2785
2851
|
.string()
|
|
2786
2852
|
.optional()
|
|
@@ -2823,7 +2889,7 @@ export const generateAudio = {
|
|
|
2823
2889
|
return ok({
|
|
2824
2890
|
requires_clarification: true,
|
|
2825
2891
|
missing: ['voiceId'],
|
|
2826
|
-
message: `${TTS_MODEL} needs a voice
|
|
2892
|
+
message: `${TTS_MODEL} needs a voice — exactly one of three. Speaking AS a character: pass its voiceAssetId (slates_list_characters) as voiceReferenceAssetId. A stock voice: slates_list_voices lists presets by gender, accent and age; pass one's voiceId. No recording of the voice: voiceDescription (words), or voiceReferenceAssetId with any clean clip of one speaker. A voice worth reusing can be kept on a character with slates_update_character, but nothing requires that — ask the user which they want only when the request does not say.`,
|
|
2827
2893
|
});
|
|
2828
2894
|
}
|
|
2829
2895
|
if (voiceSources.length > 1) {
|
|
@@ -3725,17 +3791,33 @@ export const setFolderCover = {
|
|
|
3725
3791
|
};
|
|
3726
3792
|
export const updateCharacter = {
|
|
3727
3793
|
id: 'slates_update_character',
|
|
3728
|
-
description: 'Update a character\'s name, description, or
|
|
3794
|
+
description: 'Update a character\'s name, description, style, or voice. Use slates_set_character_identity_asset for its canonical image.',
|
|
3729
3795
|
input: z.object({
|
|
3730
3796
|
characterId: z.string().uuid(),
|
|
3731
3797
|
name: z.string().min(1).max(120).optional(),
|
|
3732
3798
|
description: z.string().optional(),
|
|
3733
3799
|
style: z.string().max(200).optional().describe("Art style. Omit to inherit the reference's style (the default). Canonical styles: photoreal, anime, painterly, 3d-render, comic. Or pass any free-text instruction, e.g. 'turn this into a real person'."),
|
|
3800
|
+
// Agent parity for the character card's voice slot: the desktop route has
|
|
3801
|
+
// taken this since 2026-08-28; the op never exposed it, so an agent could
|
|
3802
|
+
// render a voice and had no way to keep it on the character.
|
|
3803
|
+
voiceAssetId: z
|
|
3804
|
+
.string()
|
|
3805
|
+
.uuid()
|
|
3806
|
+
.nullable()
|
|
3807
|
+
.optional()
|
|
3808
|
+
.describe("The AUDIO asset that is this character's voice (what inworld-tts-2 clones for its lines); null detaches, the clip stays."),
|
|
3734
3809
|
}),
|
|
3735
3810
|
async run(input, ctx) {
|
|
3736
3811
|
return ok(await ctx.desktop().post('/agent/characters/update', {
|
|
3737
3812
|
id: input.characterId,
|
|
3738
|
-
data: {
|
|
3813
|
+
data: {
|
|
3814
|
+
name: input.name,
|
|
3815
|
+
description: input.description,
|
|
3816
|
+
style: input.style,
|
|
3817
|
+
// Sent only when given: the route treats presence as intent, and an
|
|
3818
|
+
// explicit null is the detach.
|
|
3819
|
+
...(input.voiceAssetId !== undefined ? { voiceAssetId: input.voiceAssetId } : {}),
|
|
3820
|
+
},
|
|
3739
3821
|
}));
|
|
3740
3822
|
},
|
|
3741
3823
|
};
|
|
@@ -4021,8 +4103,8 @@ const SHOT_ASPECT_RATIOS = [...new Set([...VIDEO_ASPECT_RATIOS, ...IMAGE_ASPECT_
|
|
|
4021
4103
|
function shotParamsShape(described) {
|
|
4022
4104
|
const d = (node, text) => (described ? node.describe(text) : node);
|
|
4023
4105
|
return {
|
|
4024
|
-
aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the
|
|
4025
|
-
duration: d(z.number().int().min(1).max(360).optional(), 'Seconds
|
|
4106
|
+
aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the model; see slates_generate_video.'),
|
|
4107
|
+
duration: d(z.number().int().min(1).max(360).optional(), 'Seconds for video or duration-based audio; TTS uses text length.'),
|
|
4026
4108
|
videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the chosen model when the Shot is saved.'),
|
|
4027
4109
|
imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
|
|
4028
4110
|
gptQuality: d(z.enum(['medium', 'high']).optional(), 'gpt-image-2 only.'),
|
|
@@ -4031,6 +4113,11 @@ function shotParamsShape(described) {
|
|
|
4031
4113
|
sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
|
|
4032
4114
|
seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider at ~45% more."),
|
|
4033
4115
|
audioDurationSeconds: d(z.number().int().min(1).max(120).optional(), 'Audio lane. On seed-audio the requested duration IS the bill.'),
|
|
4116
|
+
// The TTS voice — the same three fields slates_generate_audio takes, so a
|
|
4117
|
+
// Shot is the audio call, serialized. Exactly one of them, enforced at fire.
|
|
4118
|
+
voiceId: d(z.string().optional(), `${TTS_MODEL}: preset voiceId (slates_list_voices).`),
|
|
4119
|
+
voiceReferenceAssetId: d(z.string().optional(), `${TTS_MODEL}: audio asset to clone.`),
|
|
4120
|
+
voiceDescription: d(z.string().optional(), `${TTS_MODEL}: the voice in words.`),
|
|
4034
4121
|
};
|
|
4035
4122
|
}
|
|
4036
4123
|
const shotParamsSchema = z.object(shotParamsShape(true)).optional();
|
|
@@ -4047,10 +4134,7 @@ const shotParamsSchemaTerse = z
|
|
|
4047
4134
|
* ship a field with no explanation — the same failure as a column nothing
|
|
4048
4135
|
* renders.
|
|
4049
4136
|
*
|
|
4050
|
-
*
|
|
4051
|
-
* surface; the prompt is the only thing the request carries. The one exception
|
|
4052
|
-
* is pre-existing: a multiShotSegment still prepends its own camera and
|
|
4053
|
-
* shotSize to its own segment prompt.
|
|
4137
|
+
* Script fields supply prompt prose when no authored prompt exists (shot-spec.ts).
|
|
4054
4138
|
*/
|
|
4055
4139
|
function shotScriptShape(described) {
|
|
4056
4140
|
const text = Object.fromEntries(SCRIPT_TEXT_FIELDS.map((field) => [
|
|
@@ -4151,6 +4235,8 @@ function shotRefInputs(input) {
|
|
|
4151
4235
|
out.push({ ref: input.firstFrameAssetId, role: 'first frame' });
|
|
4152
4236
|
if (input.lastFrameAssetId)
|
|
4153
4237
|
out.push({ ref: input.lastFrameAssetId, role: 'last frame' });
|
|
4238
|
+
if (input.params?.voiceReferenceAssetId)
|
|
4239
|
+
out.push({ ref: input.params.voiceReferenceAssetId, role: 'voice reference' });
|
|
4154
4240
|
return out;
|
|
4155
4241
|
}
|
|
4156
4242
|
/**
|
|
@@ -4187,7 +4273,10 @@ async function buildShotSpecInput(ctx, projectId, input) {
|
|
|
4187
4273
|
// later diverges from it and the desktop card says so — the prompt is
|
|
4188
4274
|
// never rewritten (that is prompt enhancement, deleted 2026-08-01).
|
|
4189
4275
|
authoredFor: input.model ?? null,
|
|
4190
|
-
params:
|
|
4276
|
+
params: {
|
|
4277
|
+
...shotParamsPatch(input.params),
|
|
4278
|
+
...(input.params?.voiceReferenceAssetId ? { voiceReferenceAssetId: rid(input.params.voiceReferenceAssetId) } : {}),
|
|
4279
|
+
},
|
|
4191
4280
|
mentions: {
|
|
4192
4281
|
characterIds: input.characterIds ?? [],
|
|
4193
4282
|
environmentIds: input.environmentIds ?? [],
|
|
@@ -4240,12 +4329,12 @@ function shotCostKey(detail) {
|
|
|
4240
4329
|
// has none, so it falls back to the raw ones and is announced as a floor.
|
|
4241
4330
|
const fires = detail.firesWith;
|
|
4242
4331
|
if (AUDIO_MODELS.includes(model)) {
|
|
4243
|
-
|
|
4244
|
-
|
|
4245
|
-
|
|
4246
|
-
|
|
4247
|
-
|
|
4248
|
-
|
|
4332
|
+
if (model === TTS_MODEL) {
|
|
4333
|
+
const text = detail.composedPrompt ?? (detail.rawPrompt.trim() || detail.line?.trim() || '');
|
|
4334
|
+
if (!text || text.length > TTS_MAX_CHARACTERS)
|
|
4335
|
+
return null;
|
|
4336
|
+
return audioCostKey({ model, characters: text.length });
|
|
4337
|
+
}
|
|
4249
4338
|
const seconds = fires?.audioDurationSeconds ?? p.audioDurationSeconds;
|
|
4250
4339
|
if (!seconds)
|
|
4251
4340
|
return null;
|
|
@@ -4453,8 +4542,17 @@ export const duplicateShot = {
|
|
|
4453
4542
|
spec.prompt = input.prompt;
|
|
4454
4543
|
if (input.model !== undefined)
|
|
4455
4544
|
spec.model = input.model;
|
|
4456
|
-
if (input.params !== undefined)
|
|
4457
|
-
|
|
4545
|
+
if (input.params !== undefined) {
|
|
4546
|
+
const voiceRef = input.params.voiceReferenceAssetId;
|
|
4547
|
+
if (voiceRef && !UUID_RE.test(voiceRef)) {
|
|
4548
|
+
const { shot } = await desktop.get('/agent/shots/get', { id: input.shotId });
|
|
4549
|
+
const built = await buildShotSpecInput(ctx, shot.projectId, { params: input.params });
|
|
4550
|
+
spec.params = built.spec.params;
|
|
4551
|
+
}
|
|
4552
|
+
else {
|
|
4553
|
+
spec.params = shotParamsPatch(input.params);
|
|
4554
|
+
}
|
|
4555
|
+
}
|
|
4458
4556
|
const r = await desktop.post('/agent/shots/duplicate', {
|
|
4459
4557
|
id: input.shotId,
|
|
4460
4558
|
name: input.name,
|
|
@@ -4565,8 +4663,8 @@ export const listShots = {
|
|
|
4565
4663
|
' depend on the reference set — Seedance reference-clip seconds and MiniMax reference images' +
|
|
4566
4664
|
' past the free five — are missing from it. slates_get_shot prices one exactly, and' +
|
|
4567
4665
|
' slates_generate_from_shots quotes the set exactly before it fires anything.' +
|
|
4568
|
-
(describeVarietyReport(r.variety) ? `
|
|
4569
|
-
|
|
4666
|
+
(describeVarietyReport(r.variety) ? `
|
|
4667
|
+
|
|
4570
4668
|
${describeVarietyReport(r.variety)}` : ''));
|
|
4571
4669
|
},
|
|
4572
4670
|
};
|
|
@@ -4715,7 +4813,7 @@ export const generateFromShots = {
|
|
|
4715
4813
|
// something to try again spends credits before anyone notices.
|
|
4716
4814
|
`\n${failedLines.join('\n')}\nThese were NOT retried. Read each error, fix the Shot, and re-fire only what you meant to.`
|
|
4717
4815
|
: '') +
|
|
4718
|
-
` ${
|
|
4816
|
+
` ${BACKGROUND_REVIEW_POINTER}`);
|
|
4719
4817
|
},
|
|
4720
4818
|
};
|
|
4721
4819
|
function resolveGuideTopic(topic) {
|
|
@@ -4874,15 +4972,16 @@ function describeGuideTopics() {
|
|
|
4874
4972
|
}
|
|
4875
4973
|
export const getPromptingGuide = {
|
|
4876
4974
|
id: 'slates_get_prompting_guide',
|
|
4877
|
-
description:
|
|
4878
|
-
|
|
4879
|
-
|
|
4880
|
-
|
|
4881
|
-
|
|
4882
|
-
|
|
4883
|
-
|
|
4884
|
-
|
|
4975
|
+
description: 'For app help and exact UI instructions use topic "app-manual" with a query such as "voice recording". This returns the canonical product manual, shared by every agent surface. ' +
|
|
4976
|
+
// 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
|
|
4977
|
+
// MEASURED at 13% compliance before and after the enforcement work — pointer
|
|
4978
|
+
// prose is the shape that does not move the agent. What replaced it is
|
|
4979
|
+
// structural: the never-use list rides the generate ops' descriptions and
|
|
4980
|
+
// the craft card rides the estimate result, so the facts arrive whether or
|
|
4981
|
+
// not this op is ever called.
|
|
4982
|
+
"Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
|
|
4885
4983
|
input: z.object({
|
|
4984
|
+
query: z.string().max(200).optional().describe('For app-manual: keywords to retrieve relevant UI sections. Omit for the entire manual.'),
|
|
4886
4985
|
topic: z
|
|
4887
4986
|
.string()
|
|
4888
4987
|
.min(1)
|
|
@@ -4890,6 +4989,10 @@ export const getPromptingGuide = {
|
|
|
4890
4989
|
depth: z.enum(['card', 'full']).optional().describe('"card" returns just the levers block (a few hundred words — the same card slates_estimate_generation_cost already attached, so usually redundant). "full" (default) returns the whole guide, up to several thousand words.'),
|
|
4891
4990
|
}),
|
|
4892
4991
|
async run(input) {
|
|
4992
|
+
if (input.topic.trim().toLowerCase() === 'app-manual') {
|
|
4993
|
+
const content = appManualSections(input.query);
|
|
4994
|
+
return { text: content, data: { topic: 'app-manual', bytes: Buffer.byteLength(content, 'utf8') } };
|
|
4995
|
+
}
|
|
4893
4996
|
const resolved = resolveGuideTopic(input.topic);
|
|
4894
4997
|
const content = resolved ? SKILLS[resolved] : undefined;
|
|
4895
4998
|
if (!resolved || content === undefined) {
|
|
@@ -5124,6 +5227,7 @@ export const ALL_OPERATIONS = [
|
|
|
5124
5227
|
generateImage,
|
|
5125
5228
|
generateVideo,
|
|
5126
5229
|
generateAudio,
|
|
5230
|
+
listVoices,
|
|
5127
5231
|
generateLipSync,
|
|
5128
5232
|
generateMotionTransfer,
|
|
5129
5233
|
editVideo,
|
|
@@ -79,6 +79,7 @@ export function buildSkillIndex() {
|
|
|
79
79
|
const PREAMBLE = fork(`You are the Slates Studio Agent — a production assistant living inside Slates, the AI video creation studio. You plan and execute video/image production runs by chaining the Slates tools: script → characters → images → videos → quality-check → regenerate, ending with assets in the user's project (and on the timeline when asked).`, `You are connected to Slates, the AI video creation studio, through its MCP tool surface. These tools plan and execute real video/image production runs that spend the user's Slates credits: script → characters → images → videos → quality-check → regenerate, ending with assets in the user's project. Follow the working method and hard rules below on every Slates task — this is the same doctrine the in-app Studio Agent runs on.`);
|
|
80
80
|
// ── The working method ─────────────────────────────────────────────
|
|
81
81
|
export const WORKING_METHOD = [
|
|
82
|
+
both(`For HOW/WHERE questions, load slates_get_prompting_guide with topic "app-manual" and relevant query keywords. Teach the documented buttons and tabs, preserving model-specific conditions; do not invent UI paths or mutate the project when the user only asks for instructions. Slates is a sandbox of optional tools, not a required pipeline.`),
|
|
82
83
|
both(`1. UNDERSTAND the outcome the user wants. If intent is clear, act with sane defaults — don't interrogate. If genuinely ambiguous, batch every question into ONE message.`),
|
|
83
84
|
both(`2. ORIENT: call slates_get_workspace_state once at the start of a workflow. Work in the user's CURRENT project — this chat lives inside it. NEVER create a new project unless explicitly asked; if there's no current project, ask which to use.`),
|
|
84
85
|
both(`3. LOAD KNOWLEDGE ON DEMAND: before prompting any model or running a multi-step workflow, load the matching guide with slates_get_prompting_guide (index below). Only the guides the task needs, when it needs them.`),
|
|
@@ -91,6 +91,15 @@ export interface ShotParams {
|
|
|
91
91
|
audioLoop?: boolean;
|
|
92
92
|
audioPromptInfluence?: number;
|
|
93
93
|
audioMultilingual?: boolean;
|
|
94
|
+
/**
|
|
95
|
+
* The VOICE a text-to-speech Shot speaks in — exactly one of the three, the
|
|
96
|
+
* same three `slates_generate_audio` takes. A Shot that carried the words but
|
|
97
|
+
* not the voice would fire in whatever voice happened to be on the bar, which
|
|
98
|
+
* is not the recipe that was saved.
|
|
99
|
+
*/
|
|
100
|
+
voiceId?: string;
|
|
101
|
+
voiceReferenceAssetId?: string;
|
|
102
|
+
voiceDescription?: string;
|
|
94
103
|
}
|
|
95
104
|
/** Prompt-owned identity — the entities the prompt text NAMES. */
|
|
96
105
|
export interface ShotMentions {
|
|
@@ -174,14 +183,14 @@ export interface ShotSpec {
|
|
|
174
183
|
* same failure as a column nothing renders.
|
|
175
184
|
*/
|
|
176
185
|
export declare const SCRIPT_FIELD_DESCRIPTION: {
|
|
177
|
-
readonly speaker: "
|
|
178
|
-
readonly line: "
|
|
179
|
-
readonly delivery: "
|
|
180
|
-
readonly action: "
|
|
181
|
-
readonly prop: "The
|
|
182
|
-
readonly shotSize: "
|
|
183
|
-
readonly camera: "
|
|
184
|
-
readonly continues: "True when this
|
|
186
|
+
readonly speaker: "Speaker: character id, bare name (including a new character), or \"VO\". Null when silent.";
|
|
187
|
+
readonly line: "Words spoken verbatim; no camera or scene instructions.";
|
|
188
|
+
readonly delivery: "Optional performance note. Leave null unless a specific direction is needed; do not fill every line with stock adjectives. Not sent to TTS: put supported inline cues in the spoken text using the selected model's prompting guide.";
|
|
189
|
+
readonly action: "Screenplay action covering everyone in frame.";
|
|
190
|
+
readonly prop: "The readable object carrying the beat.";
|
|
191
|
+
readonly shotSize: "Free-text framing, e.g. \"wide\" or \"long-lens CU, other head blurred\"; bucketed only for variety counts.";
|
|
192
|
+
readonly camera: "Free-text camera move, e.g. \"slow push in\"; never rejected or rewritten.";
|
|
193
|
+
readonly continues: "True when this line continues the previous row: one sentence across two cuts.";
|
|
185
194
|
};
|
|
186
195
|
/** The script fields, as a list. Sorted from the description map so the two
|
|
187
196
|
* cannot disagree — never a second hand-written array. */
|
|
@@ -223,6 +232,10 @@ export declare function effectivePrompt(spec: ShotSpec): string;
|
|
|
223
232
|
* overlays what it actually found, so a missing field is never `undefined`
|
|
224
233
|
* leaking into a request. */
|
|
225
234
|
export declare function emptyShotSpec(): ShotSpec;
|
|
235
|
+
/** The mutually exclusive TTS source fields, shared by readers and patch merging. */
|
|
236
|
+
export declare const VOICE_SOURCE_FIELDS: readonly ["voiceId", "voiceReferenceAssetId", "voiceDescription"];
|
|
237
|
+
/** Setting a voice replaces the previous source; unrelated parameter edits preserve it. */
|
|
238
|
+
export declare function mergeShotParams(existing: ShotParams, patch: Record<string, unknown>): ShotParams;
|
|
226
239
|
/**
|
|
227
240
|
* Read a `ShotSpec` out of whatever is on disk — a row written by an older
|
|
228
241
|
* build, a partial object from an op, `null`.
|
|
@@ -46,14 +46,14 @@ export const ORDERED_ATTACHMENT_ROLES = Object.keys(ORDERED_ROLE_EMISSION).sort(
|
|
|
46
46
|
* same failure as a column nothing renders.
|
|
47
47
|
*/
|
|
48
48
|
export const SCRIPT_FIELD_DESCRIPTION = {
|
|
49
|
-
speaker: '
|
|
50
|
-
line: '
|
|
51
|
-
delivery: '
|
|
52
|
-
action: '
|
|
53
|
-
prop: 'The
|
|
54
|
-
shotSize: '
|
|
55
|
-
camera: '
|
|
56
|
-
continues: 'True when this
|
|
49
|
+
speaker: 'Speaker: character id, bare name (including a new character), or "VO". Null when silent.',
|
|
50
|
+
line: 'Words spoken verbatim; no camera or scene instructions.',
|
|
51
|
+
delivery: 'Optional performance note. Leave null unless a specific direction is needed; do not fill every line with stock adjectives. Not sent to TTS: put supported inline cues in the spoken text using the selected model\'s prompting guide.',
|
|
52
|
+
action: 'Screenplay action covering everyone in frame.',
|
|
53
|
+
prop: 'The readable object carrying the beat.',
|
|
54
|
+
shotSize: 'Free-text framing, e.g. "wide" or "long-lens CU, other head blurred"; bucketed only for variety counts.',
|
|
55
|
+
camera: 'Free-text camera move, e.g. "slow push in"; never rejected or rewritten.',
|
|
56
|
+
continues: 'True when this line continues the previous row: one sentence across two cuts.',
|
|
57
57
|
};
|
|
58
58
|
/** The seven free-text script fields (everything but the `continues` flag) —
|
|
59
59
|
* the set an op accepts as `string | null` and a layer renders as text. */
|
|
@@ -151,6 +151,17 @@ export function emptyShotSpec() {
|
|
|
151
151
|
}
|
|
152
152
|
const str = (v) => (typeof v === 'string' && v.length > 0 ? v : null);
|
|
153
153
|
const strArray = (v) => Array.isArray(v) ? v.filter((x) => typeof x === 'string' && x.length > 0) : [];
|
|
154
|
+
/** The mutually exclusive TTS source fields, shared by readers and patch merging. */
|
|
155
|
+
export const VOICE_SOURCE_FIELDS = ['voiceId', 'voiceReferenceAssetId', 'voiceDescription'];
|
|
156
|
+
/** Setting a voice replaces the previous source; unrelated parameter edits preserve it. */
|
|
157
|
+
export function mergeShotParams(existing, patch) {
|
|
158
|
+
const next = { ...existing };
|
|
159
|
+
if (VOICE_SOURCE_FIELDS.some((k) => typeof patch[k] === 'string' && patch[k].trim())) {
|
|
160
|
+
for (const k of VOICE_SOURCE_FIELDS)
|
|
161
|
+
delete next[k];
|
|
162
|
+
}
|
|
163
|
+
return readParams({ ...next, ...patch });
|
|
164
|
+
}
|
|
154
165
|
function readParams(v) {
|
|
155
166
|
if (!v || typeof v !== 'object')
|
|
156
167
|
return {};
|
|
@@ -175,6 +186,9 @@ function readParams(v) {
|
|
|
175
186
|
s('audioLanguage');
|
|
176
187
|
s('audioAccent');
|
|
177
188
|
s('negativePrompt');
|
|
189
|
+
s('voiceId');
|
|
190
|
+
s('voiceReferenceAssetId');
|
|
191
|
+
s('voiceDescription');
|
|
178
192
|
n('duration');
|
|
179
193
|
n('imageQuantity');
|
|
180
194
|
n('audioDurationSeconds');
|
|
@@ -276,6 +290,8 @@ export function shotAssetIds(spec) {
|
|
|
276
290
|
ids.push(spec.firstFrameAssetId);
|
|
277
291
|
if (spec.lastFrameAssetId)
|
|
278
292
|
ids.push(spec.lastFrameAssetId);
|
|
293
|
+
if (spec.params.voiceReferenceAssetId)
|
|
294
|
+
ids.push(spec.params.voiceReferenceAssetId);
|
|
279
295
|
return [...new Set(ids.filter(Boolean))];
|
|
280
296
|
}
|
|
281
297
|
/** Total attachment count — what a list row shows without composing anything. */
|