@slatesvideo/shared 0.6.4 → 0.6.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,6 +14,7 @@ import { SlatesCloudClient } from '../clients/cloud.js';
14
14
  import { SlatesDesktopClient } from '../clients/desktop.js';
15
15
  import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
16
16
  import { SKILLS } from '../skills/content.js';
17
+ import { appManualSections } from '../manual/index.js';
17
18
  // Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
18
19
  // "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
19
20
  import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
@@ -183,6 +184,15 @@ export const TTS_MAX_CHARACTERS = (() => {
183
184
  return max;
184
185
  })();
185
186
  export const TTS_BUCKET_COUNT = TTS_MAX_CHARACTERS / TTS_BUCKET_CHARS; // 8
187
+ /** The seat's cloning spec — reference-clip bounds, the design-prompt bounds
188
+ * and the workspace-wide clone rate — read from the SSOT for the same reason
189
+ * as the cap: every number in it was MEASURED and lives in exactly one row. */
190
+ export const TTS_VOICE_CLONE = (() => {
191
+ const spec = getModelCapability(TTS_MODEL)?.voiceClone;
192
+ if (!spec)
193
+ throw new Error(`MODEL_CAPABILITIES['${TTS_MODEL}'] must declare voiceClone`);
194
+ return spec;
195
+ })();
186
196
  function creditCost(m) {
187
197
  if (!m)
188
198
  return 0;
@@ -219,7 +229,7 @@ const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it
219
229
  const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
220
230
  'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
221
231
  const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
222
- 'slates_get_asset_video_frames for video.';
232
+ 'slates_get_asset_video_frames for video. For audio, audition the saved file; metadata alone does not establish voice similarity or delivery quality.';
223
233
  // The image saved, but reading it back off disk failed (best-effort fetch). The
224
234
  // agent has an asset and NO pixels, which is the one state where a quality
225
235
  // claim would be pure invention — so this branch has to say so rather than
@@ -2727,14 +2737,68 @@ export const generateVideo = {
2727
2737
  };
2728
2738
  },
2729
2739
  };
2740
+ // ── The preset voice shelf ──────────────────────────────────────
2741
+ /**
2742
+ * AGENT PARITY for the voice picker. The desktop browses stock voices by
2743
+ * gender, accent and age and plays each one; until this op the agent could
2744
+ * only pass a `voiceId` it had no way to discover. Disk reads on the desktop,
2745
+ * never a vendor call — browsing is free on every surface.
2746
+ */
2747
+ export const listVoices = {
2748
+ id: 'slates_list_voices',
2749
+ description: `Browse ${TTS_MODEL} preset voices. Pass a returned voiceId to slates_generate_audio. Filters AND together.`,
2750
+ input: z.object({
2751
+ gender: z.string().optional().describe('male | female'),
2752
+ accent: z.string().optional().describe('Region ("GB") or languageCode ("en-GB"); available accents come from the shelf.'),
2753
+ age: z.string().optional().describe('young | middle_aged | elderly'),
2754
+ query: z.string().optional().describe('Free text over name, description, tags.'),
2755
+ }),
2756
+ async run(input, ctx) {
2757
+ const desktop = ctx.desktop();
2758
+ await desktop.requireCapability('voices', 'the preset voice shelf');
2759
+ const shelf = await desktop.get('/agent/voices');
2760
+ if (!shelf.available) {
2761
+ return ok({ available: false, voices: [], message: 'This desktop build shipped without the preset shelf.' });
2762
+ }
2763
+ const terms = (input.query ?? '').toLowerCase().split(/\s+/).filter(Boolean);
2764
+ const region = input.accent?.includes('-') ? input.accent.split('-')[1] : input.accent;
2765
+ const voices = shelf.voices
2766
+ .filter((v) => !input.gender || v.gender === input.gender)
2767
+ .filter((v) => !region || v.languageCode.split('-')[1]?.toUpperCase() === region.toUpperCase())
2768
+ .filter((v) => !input.age || v.ageGroup === input.age)
2769
+ .filter((v) => {
2770
+ if (terms.length === 0)
2771
+ return true;
2772
+ const hay = [v.displayName, v.description, v.gender, v.ageGroup, v.languageCode, ...(v.tags ?? [])]
2773
+ .join(' ')
2774
+ .toLowerCase();
2775
+ return terms.every((t) => hay.includes(t));
2776
+ })
2777
+ .map(({ voiceId, displayName, description, tags, gender, ageGroup, languageCode }) => ({
2778
+ voiceId,
2779
+ displayName,
2780
+ description,
2781
+ tags,
2782
+ gender,
2783
+ ageGroup,
2784
+ languageCode,
2785
+ }));
2786
+ return ok({
2787
+ available: true,
2788
+ line: shelf.line,
2789
+ count: voices.length,
2790
+ voices,
2791
+ next: `Pass a voiceId to slates_generate_audio (model ${TTS_MODEL}) as voiceId. To keep one on a character for reuse, generate a clip with it and set slates_update_character voiceAssetId.`,
2792
+ });
2793
+ },
2794
+ };
2730
2795
  // ── Generate audio ──────────────────────────────────────────────
2731
2796
  export const generateAudio = {
2732
2797
  id: 'slates_generate_audio',
2733
2798
  billable: true,
2734
- description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Three surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s), eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop), and inworld-tts-2 (one named voice saying one line; the prompt IS the words, billed per character). Which surface for which job: read the slates-model-selection skill. ' +
2735
- '🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
2736
- 'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
2737
- 'projectId is REQUIRED (no headless path). ' +
2799
+ description: `Generate project audio using credits. Choose the surface via the model routing below. ` +
2800
+ 'Read slates-cost-discipline and the matching prompting skill first (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). ' +
2801
+ 'Seed Audio bills the requested duration, which is appended to the prompt regardless of output length. Kling "SFX:" / "Ambient noise:" syntax does not transfer. ' +
2738
2802
  CONFIRM_GATE_SENTENCE +
2739
2803
  ' No skill files installed? Call slates_get_prompting_guide first.',
2740
2804
  input: z.object({
@@ -2753,25 +2817,27 @@ export const generateAudio = {
2753
2817
  durationSeconds: z
2754
2818
  .number()
2755
2819
  .optional()
2756
- .describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.'),
2820
+ .describe(`seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (default ${SEED_AUDIO_DEFAULT_SECONDS}) — ⚠️ THIS IS THE BILL: appended to the prompt and charged whatever comes back. eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS} (default ${ELEVEN_SFX_DEFAULT_SECONDS}) — always sent explicitly so the per-second charge is deterministic. Not for ${TTS_MODEL}.`),
2757
2821
  voice: z
2758
2822
  .string()
2759
2823
  .optional()
2760
- .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
2824
+ .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself. Agent-facing only.'),
2761
2825
  voiceId: z
2762
2826
  .string()
2763
2827
  .optional()
2764
- .describe('inworld-tts-2 — the voice to speak in. One of these three is required there.'),
2828
+ .describe(`${TTS_MODEL} — a preset voiceId from slates_list_voices, not a character or asset id. Exactly one voice source is required.`),
2765
2829
  voiceReferenceAssetId: z
2766
2830
  .string()
2767
2831
  .optional()
2768
- .describe('inworld-tts-2 — clone the voice from this AUDIO asset (5-15s of one clean speaker).'),
2832
+ .describe(`${TTS_MODEL} — clone this AUDIO asset's voice for the take (${TTS_VOICE_CLONE.minSeconds}-${TTS_VOICE_CLONE.maxSeconds}s, one clean speaker). To speak AS a character pass its voiceAssetId. Cloning: ${TTS_VOICE_CLONE.clonesPerMinute} new voices/min across all of Slates; a burst waits.`),
2769
2833
  voiceDescription: z
2770
2834
  .string()
2835
+ .min(TTS_VOICE_CLONE.designPromptChars.min)
2836
+ .max(TTS_VOICE_CLONE.designPromptChars.max)
2771
2837
  .optional()
2772
- .describe('inworld-tts-2 — build a voice from this description, for a character with no recording.'),
2838
+ .describe(`${TTS_MODEL} — a voice from words (${TTS_VOICE_CLONE.designPromptChars.min}-${TTS_VOICE_CLONE.designPromptChars.max} chars) for a character with no recording; keep it via slates_update_character voiceAssetId.`),
2773
2839
  speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
2774
- volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
2840
+ volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline fader for mix decisions.'),
2775
2841
  pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
2776
2842
  multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
2777
2843
  loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
@@ -2780,7 +2846,7 @@ export const generateAudio = {
2780
2846
  .array(z.string())
2781
2847
  .max(3)
2782
2848
  .optional()
2783
- .describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId — the API rejects both.'),
2849
+ .describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId.'),
2784
2850
  imageReferenceAssetId: z
2785
2851
  .string()
2786
2852
  .optional()
@@ -2823,7 +2889,7 @@ export const generateAudio = {
2823
2889
  return ok({
2824
2890
  requires_clarification: true,
2825
2891
  missing: ['voiceId'],
2826
- message: `${TTS_MODEL} needs a voice. Ask the user WHICH CHARACTER is speaking and pass that character's voice as voiceId — a voice is a field on a character, not a thing to pick at generation time. To make a NEW voice, pass voiceReferenceAssetId (a clip to clone) or voiceDescription (words, for a character with no recording).`,
2892
+ message: `${TTS_MODEL} needs a voice — exactly one of three. Speaking AS a character: pass its voiceAssetId (slates_list_characters) as voiceReferenceAssetId. A stock voice: slates_list_voices lists presets by gender, accent and age; pass one's voiceId. No recording of the voice: voiceDescription (words), or voiceReferenceAssetId with any clean clip of one speaker. A voice worth reusing can be kept on a character with slates_update_character, but nothing requires that — ask the user which they want only when the request does not say.`,
2827
2893
  });
2828
2894
  }
2829
2895
  if (voiceSources.length > 1) {
@@ -3725,17 +3791,33 @@ export const setFolderCover = {
3725
3791
  };
3726
3792
  export const updateCharacter = {
3727
3793
  id: 'slates_update_character',
3728
- description: 'Update a character\'s name, description, or style. Use slates_set_character_identity_asset for its canonical image.',
3794
+ description: 'Update a character\'s name, description, style, or voice. Use slates_set_character_identity_asset for its canonical image.',
3729
3795
  input: z.object({
3730
3796
  characterId: z.string().uuid(),
3731
3797
  name: z.string().min(1).max(120).optional(),
3732
3798
  description: z.string().optional(),
3733
3799
  style: z.string().max(200).optional().describe("Art style. Omit to inherit the reference's style (the default). Canonical styles: photoreal, anime, painterly, 3d-render, comic. Or pass any free-text instruction, e.g. 'turn this into a real person'."),
3800
+ // Agent parity for the character card's voice slot: the desktop route has
3801
+ // taken this since 2026-08-28; the op never exposed it, so an agent could
3802
+ // render a voice and had no way to keep it on the character.
3803
+ voiceAssetId: z
3804
+ .string()
3805
+ .uuid()
3806
+ .nullable()
3807
+ .optional()
3808
+ .describe("The AUDIO asset that is this character's voice (what inworld-tts-2 clones for its lines); null detaches, the clip stays."),
3734
3809
  }),
3735
3810
  async run(input, ctx) {
3736
3811
  return ok(await ctx.desktop().post('/agent/characters/update', {
3737
3812
  id: input.characterId,
3738
- data: { name: input.name, description: input.description, style: input.style },
3813
+ data: {
3814
+ name: input.name,
3815
+ description: input.description,
3816
+ style: input.style,
3817
+ // Sent only when given: the route treats presence as intent, and an
3818
+ // explicit null is the detach.
3819
+ ...(input.voiceAssetId !== undefined ? { voiceAssetId: input.voiceAssetId } : {}),
3820
+ },
3739
3821
  }));
3740
3822
  },
3741
3823
  };
@@ -4021,8 +4103,8 @@ const SHOT_ASPECT_RATIOS = [...new Set([...VIDEO_ASPECT_RATIOS, ...IMAGE_ASPECT_
4021
4103
  function shotParamsShape(described) {
4022
4104
  const d = (node, text) => (described ? node.describe(text) : node);
4023
4105
  return {
4024
- aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the chosen model when the Shot is saved — see slates_generate_video for the per-model sets.'),
4025
- duration: d(z.number().int().min(1).max(360).optional(), 'Seconds — video or audio. Required before a video Shot can be priced or fired; validated against the model when saved.'),
4106
+ aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the model; see slates_generate_video.'),
4107
+ duration: d(z.number().int().min(1).max(360).optional(), 'Seconds for video or duration-based audio; TTS uses text length.'),
4026
4108
  videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the chosen model when the Shot is saved.'),
4027
4109
  imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
4028
4110
  gptQuality: d(z.enum(['medium', 'high']).optional(), 'gpt-image-2 only.'),
@@ -4031,6 +4113,11 @@ function shotParamsShape(described) {
4031
4113
  sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
4032
4114
  seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider at ~45% more."),
4033
4115
  audioDurationSeconds: d(z.number().int().min(1).max(120).optional(), 'Audio lane. On seed-audio the requested duration IS the bill.'),
4116
+ // The TTS voice — the same three fields slates_generate_audio takes, so a
4117
+ // Shot is the audio call, serialized. Exactly one of them, enforced at fire.
4118
+ voiceId: d(z.string().optional(), `${TTS_MODEL}: preset voiceId (slates_list_voices).`),
4119
+ voiceReferenceAssetId: d(z.string().optional(), `${TTS_MODEL}: audio asset to clone.`),
4120
+ voiceDescription: d(z.string().optional(), `${TTS_MODEL}: the voice in words.`),
4034
4121
  };
4035
4122
  }
4036
4123
  const shotParamsSchema = z.object(shotParamsShape(true)).optional();
@@ -4047,10 +4134,7 @@ const shotParamsSchemaTerse = z
4047
4134
  * ship a field with no explanation — the same failure as a column nothing
4048
4135
  * renders.
4049
4136
  *
4050
- * 🚨 AND NONE OF THEM IS SENT TO A MODEL. They are a planning and counting
4051
- * surface; the prompt is the only thing the request carries. The one exception
4052
- * is pre-existing: a multiShotSegment still prepends its own camera and
4053
- * shotSize to its own segment prompt.
4137
+ * Script fields supply prompt prose when no authored prompt exists (shot-spec.ts).
4054
4138
  */
4055
4139
  function shotScriptShape(described) {
4056
4140
  const text = Object.fromEntries(SCRIPT_TEXT_FIELDS.map((field) => [
@@ -4151,6 +4235,8 @@ function shotRefInputs(input) {
4151
4235
  out.push({ ref: input.firstFrameAssetId, role: 'first frame' });
4152
4236
  if (input.lastFrameAssetId)
4153
4237
  out.push({ ref: input.lastFrameAssetId, role: 'last frame' });
4238
+ if (input.params?.voiceReferenceAssetId)
4239
+ out.push({ ref: input.params.voiceReferenceAssetId, role: 'voice reference' });
4154
4240
  return out;
4155
4241
  }
4156
4242
  /**
@@ -4187,7 +4273,10 @@ async function buildShotSpecInput(ctx, projectId, input) {
4187
4273
  // later diverges from it and the desktop card says so — the prompt is
4188
4274
  // never rewritten (that is prompt enhancement, deleted 2026-08-01).
4189
4275
  authoredFor: input.model ?? null,
4190
- params: shotParamsPatch(input.params),
4276
+ params: {
4277
+ ...shotParamsPatch(input.params),
4278
+ ...(input.params?.voiceReferenceAssetId ? { voiceReferenceAssetId: rid(input.params.voiceReferenceAssetId) } : {}),
4279
+ },
4191
4280
  mentions: {
4192
4281
  characterIds: input.characterIds ?? [],
4193
4282
  environmentIds: input.environmentIds ?? [],
@@ -4240,12 +4329,12 @@ function shotCostKey(detail) {
4240
4329
  // has none, so it falls back to the raw ones and is announced as a floor.
4241
4330
  const fires = detail.firesWith;
4242
4331
  if (AUDIO_MODELS.includes(model)) {
4243
- // The TTS seat prices on the TEXT, and a Shot carries no text field — so a
4244
- // Shot cannot be a TTS generation and cannot be quoted as one. Explicit,
4245
- // because the `!seconds` line below would also return null here and that
4246
- // would read as "duration missing" for a surface that has no duration.
4247
- if (model === TTS_MODEL)
4248
- return null;
4332
+ if (model === TTS_MODEL) {
4333
+ const text = detail.composedPrompt ?? (detail.rawPrompt.trim() || detail.line?.trim() || '');
4334
+ if (!text || text.length > TTS_MAX_CHARACTERS)
4335
+ return null;
4336
+ return audioCostKey({ model, characters: text.length });
4337
+ }
4249
4338
  const seconds = fires?.audioDurationSeconds ?? p.audioDurationSeconds;
4250
4339
  if (!seconds)
4251
4340
  return null;
@@ -4453,8 +4542,17 @@ export const duplicateShot = {
4453
4542
  spec.prompt = input.prompt;
4454
4543
  if (input.model !== undefined)
4455
4544
  spec.model = input.model;
4456
- if (input.params !== undefined)
4457
- spec.params = shotParamsPatch(input.params);
4545
+ if (input.params !== undefined) {
4546
+ const voiceRef = input.params.voiceReferenceAssetId;
4547
+ if (voiceRef && !UUID_RE.test(voiceRef)) {
4548
+ const { shot } = await desktop.get('/agent/shots/get', { id: input.shotId });
4549
+ const built = await buildShotSpecInput(ctx, shot.projectId, { params: input.params });
4550
+ spec.params = built.spec.params;
4551
+ }
4552
+ else {
4553
+ spec.params = shotParamsPatch(input.params);
4554
+ }
4555
+ }
4458
4556
  const r = await desktop.post('/agent/shots/duplicate', {
4459
4557
  id: input.shotId,
4460
4558
  name: input.name,
@@ -4565,8 +4663,8 @@ export const listShots = {
4565
4663
  ' depend on the reference set — Seedance reference-clip seconds and MiniMax reference images' +
4566
4664
  ' past the free five — are missing from it. slates_get_shot prices one exactly, and' +
4567
4665
  ' slates_generate_from_shots quotes the set exactly before it fires anything.' +
4568
- (describeVarietyReport(r.variety) ? `
4569
-
4666
+ (describeVarietyReport(r.variety) ? `
4667
+
4570
4668
  ${describeVarietyReport(r.variety)}` : ''));
4571
4669
  },
4572
4670
  };
@@ -4715,7 +4813,7 @@ export const generateFromShots = {
4715
4813
  // something to try again spends credits before anyone notices.
4716
4814
  `\n${failedLines.join('\n')}\nThese were NOT retried. Read each error, fix the Shot, and re-fire only what you meant to.`
4717
4815
  : '') +
4718
- ` ${VIDEO_REVIEW_POINTER}`);
4816
+ ` ${BACKGROUND_REVIEW_POINTER}`);
4719
4817
  },
4720
4818
  };
4721
4819
  function resolveGuideTopic(topic) {
@@ -4874,15 +4972,16 @@ function describeGuideTopics() {
4874
4972
  }
4875
4973
  export const getPromptingGuide = {
4876
4974
  id: 'slates_get_prompting_guide',
4877
- description:
4878
- // 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
4879
- // MEASURED at 13% compliance before and after the enforcement work — pointer
4880
- // prose is the shape that does not move the agent. What replaced it is
4881
- // structural: the never-use list rides the generate ops' descriptions and
4882
- // the craft card rides the estimate result, so the facts arrive whether or
4883
- // not this op is ever called.
4884
- "Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
4975
+ description: 'For app help and exact UI instructions use topic "app-manual" with a query such as "voice recording". This returns the canonical product manual, shared by every agent surface. ' +
4976
+ // 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
4977
+ // MEASURED at 13% compliance before and after the enforcement work — pointer
4978
+ // prose is the shape that does not move the agent. What replaced it is
4979
+ // structural: the never-use list rides the generate ops' descriptions and
4980
+ // the craft card rides the estimate result, so the facts arrive whether or
4981
+ // not this op is ever called.
4982
+ "Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
4885
4983
  input: z.object({
4984
+ query: z.string().max(200).optional().describe('For app-manual: keywords to retrieve relevant UI sections. Omit for the entire manual.'),
4886
4985
  topic: z
4887
4986
  .string()
4888
4987
  .min(1)
@@ -4890,6 +4989,10 @@ export const getPromptingGuide = {
4890
4989
  depth: z.enum(['card', 'full']).optional().describe('"card" returns just the levers block (a few hundred words — the same card slates_estimate_generation_cost already attached, so usually redundant). "full" (default) returns the whole guide, up to several thousand words.'),
4891
4990
  }),
4892
4991
  async run(input) {
4992
+ if (input.topic.trim().toLowerCase() === 'app-manual') {
4993
+ const content = appManualSections(input.query);
4994
+ return { text: content, data: { topic: 'app-manual', bytes: Buffer.byteLength(content, 'utf8') } };
4995
+ }
4893
4996
  const resolved = resolveGuideTopic(input.topic);
4894
4997
  const content = resolved ? SKILLS[resolved] : undefined;
4895
4998
  if (!resolved || content === undefined) {
@@ -5124,6 +5227,7 @@ export const ALL_OPERATIONS = [
5124
5227
  generateImage,
5125
5228
  generateVideo,
5126
5229
  generateAudio,
5230
+ listVoices,
5127
5231
  generateLipSync,
5128
5232
  generateMotionTransfer,
5129
5233
  editVideo,
@@ -79,6 +79,7 @@ export function buildSkillIndex() {
79
79
  const PREAMBLE = fork(`You are the Slates Studio Agent — a production assistant living inside Slates, the AI video creation studio. You plan and execute video/image production runs by chaining the Slates tools: script → characters → images → videos → quality-check → regenerate, ending with assets in the user's project (and on the timeline when asked).`, `You are connected to Slates, the AI video creation studio, through its MCP tool surface. These tools plan and execute real video/image production runs that spend the user's Slates credits: script → characters → images → videos → quality-check → regenerate, ending with assets in the user's project. Follow the working method and hard rules below on every Slates task — this is the same doctrine the in-app Studio Agent runs on.`);
80
80
  // ── The working method ─────────────────────────────────────────────
81
81
  export const WORKING_METHOD = [
82
+ both(`For HOW/WHERE questions, load slates_get_prompting_guide with topic "app-manual" and relevant query keywords. Teach the documented buttons and tabs, preserving model-specific conditions; do not invent UI paths or mutate the project when the user only asks for instructions. Slates is a sandbox of optional tools, not a required pipeline.`),
82
83
  both(`1. UNDERSTAND the outcome the user wants. If intent is clear, act with sane defaults — don't interrogate. If genuinely ambiguous, batch every question into ONE message.`),
83
84
  both(`2. ORIENT: call slates_get_workspace_state once at the start of a workflow. Work in the user's CURRENT project — this chat lives inside it. NEVER create a new project unless explicitly asked; if there's no current project, ask which to use.`),
84
85
  both(`3. LOAD KNOWLEDGE ON DEMAND: before prompting any model or running a multi-step workflow, load the matching guide with slates_get_prompting_guide (index below). Only the guides the task needs, when it needs them.`),
@@ -91,6 +91,15 @@ export interface ShotParams {
91
91
  audioLoop?: boolean;
92
92
  audioPromptInfluence?: number;
93
93
  audioMultilingual?: boolean;
94
+ /**
95
+ * The VOICE a text-to-speech Shot speaks in — exactly one of the three, the
96
+ * same three `slates_generate_audio` takes. A Shot that carried the words but
97
+ * not the voice would fire in whatever voice happened to be on the bar, which
98
+ * is not the recipe that was saved.
99
+ */
100
+ voiceId?: string;
101
+ voiceReferenceAssetId?: string;
102
+ voiceDescription?: string;
94
103
  }
95
104
  /** Prompt-owned identity — the entities the prompt text NAMES. */
96
105
  export interface ShotMentions {
@@ -174,14 +183,14 @@ export interface ShotSpec {
174
183
  * same failure as a column nothing renders.
175
184
  */
176
185
  export declare const SCRIPT_FIELD_DESCRIPTION: {
177
- readonly speaker: "Who says the line — a character id, a bare name (a character that does not exist yet is fine), or \"VO\". Null for a shot with no words.";
178
- readonly line: "What is SAID, verbatim. Never camera, scene or prompt language — this is the half a person reads aloud.";
179
- readonly delivery: "The parenthetical: how it is said. \"(flat, exhausted)\"";
180
- readonly action: "What happens in the shot, screenplay-style. One line covering everyone in frame.";
181
- readonly prop: "The one readable object carrying the beat.";
182
- readonly shotSize: "Framing, in your own words — \"wide\", \"long-lens CU, other head blurred\". FREE TEXT: it is bucketed for the variety count and never rejected or rewritten.";
183
- readonly camera: "Camera move, in your own words — \"slow push in\", \"through the rearview, eyes only\". FREE TEXT, same rule as shotSize.";
184
- readonly continues: "True when this row's line runs on from the previous row's — one sentence split across two cuts. The signature VO move; set it deliberately.";
186
+ readonly speaker: "Speaker: character id, bare name (including a new character), or \"VO\". Null when silent.";
187
+ readonly line: "Words spoken verbatim; no camera or scene instructions.";
188
+ readonly delivery: "Optional performance note. Leave null unless a specific direction is needed; do not fill every line with stock adjectives. Not sent to TTS: put supported inline cues in the spoken text using the selected model's prompting guide.";
189
+ readonly action: "Screenplay action covering everyone in frame.";
190
+ readonly prop: "The readable object carrying the beat.";
191
+ readonly shotSize: "Free-text framing, e.g. \"wide\" or \"long-lens CU, other head blurred\"; bucketed only for variety counts.";
192
+ readonly camera: "Free-text camera move, e.g. \"slow push in\"; never rejected or rewritten.";
193
+ readonly continues: "True when this line continues the previous row: one sentence across two cuts.";
185
194
  };
186
195
  /** The script fields, as a list. Sorted from the description map so the two
187
196
  * cannot disagree — never a second hand-written array. */
@@ -223,6 +232,10 @@ export declare function effectivePrompt(spec: ShotSpec): string;
223
232
  * overlays what it actually found, so a missing field is never `undefined`
224
233
  * leaking into a request. */
225
234
  export declare function emptyShotSpec(): ShotSpec;
235
+ /** The mutually exclusive TTS source fields, shared by readers and patch merging. */
236
+ export declare const VOICE_SOURCE_FIELDS: readonly ["voiceId", "voiceReferenceAssetId", "voiceDescription"];
237
+ /** Setting a voice replaces the previous source; unrelated parameter edits preserve it. */
238
+ export declare function mergeShotParams(existing: ShotParams, patch: Record<string, unknown>): ShotParams;
226
239
  /**
227
240
  * Read a `ShotSpec` out of whatever is on disk — a row written by an older
228
241
  * build, a partial object from an op, `null`.
@@ -46,14 +46,14 @@ export const ORDERED_ATTACHMENT_ROLES = Object.keys(ORDERED_ROLE_EMISSION).sort(
46
46
  * same failure as a column nothing renders.
47
47
  */
48
48
  export const SCRIPT_FIELD_DESCRIPTION = {
49
- speaker: 'Who says the line — a character id, a bare name (a character that does not exist yet is fine), or "VO". Null for a shot with no words.',
50
- line: 'What is SAID, verbatim. Never camera, scene or prompt language — this is the half a person reads aloud.',
51
- delivery: 'The parenthetical: how it is said. "(flat, exhausted)"',
52
- action: 'What happens in the shot, screenplay-style. One line covering everyone in frame.',
53
- prop: 'The one readable object carrying the beat.',
54
- shotSize: 'Framing, in your own words — "wide", "long-lens CU, other head blurred". FREE TEXT: it is bucketed for the variety count and never rejected or rewritten.',
55
- camera: 'Camera move, in your own words — "slow push in", "through the rearview, eyes only". FREE TEXT, same rule as shotSize.',
56
- continues: 'True when this row\'s line runs on from the previous row\'s — one sentence split across two cuts. The signature VO move; set it deliberately.',
49
+ speaker: 'Speaker: character id, bare name (including a new character), or "VO". Null when silent.',
50
+ line: 'Words spoken verbatim; no camera or scene instructions.',
51
+ delivery: 'Optional performance note. Leave null unless a specific direction is needed; do not fill every line with stock adjectives. Not sent to TTS: put supported inline cues in the spoken text using the selected model\'s prompting guide.',
52
+ action: 'Screenplay action covering everyone in frame.',
53
+ prop: 'The readable object carrying the beat.',
54
+ shotSize: 'Free-text framing, e.g. "wide" or "long-lens CU, other head blurred"; bucketed only for variety counts.',
55
+ camera: 'Free-text camera move, e.g. "slow push in"; never rejected or rewritten.',
56
+ continues: 'True when this line continues the previous row: one sentence across two cuts.',
57
57
  };
58
58
  /** The seven free-text script fields (everything but the `continues` flag) —
59
59
  * the set an op accepts as `string | null` and a layer renders as text. */
@@ -151,6 +151,17 @@ export function emptyShotSpec() {
151
151
  }
152
152
  const str = (v) => (typeof v === 'string' && v.length > 0 ? v : null);
153
153
  const strArray = (v) => Array.isArray(v) ? v.filter((x) => typeof x === 'string' && x.length > 0) : [];
154
+ /** The mutually exclusive TTS source fields, shared by readers and patch merging. */
155
+ export const VOICE_SOURCE_FIELDS = ['voiceId', 'voiceReferenceAssetId', 'voiceDescription'];
156
+ /** Setting a voice replaces the previous source; unrelated parameter edits preserve it. */
157
+ export function mergeShotParams(existing, patch) {
158
+ const next = { ...existing };
159
+ if (VOICE_SOURCE_FIELDS.some((k) => typeof patch[k] === 'string' && patch[k].trim())) {
160
+ for (const k of VOICE_SOURCE_FIELDS)
161
+ delete next[k];
162
+ }
163
+ return readParams({ ...next, ...patch });
164
+ }
154
165
  function readParams(v) {
155
166
  if (!v || typeof v !== 'object')
156
167
  return {};
@@ -175,6 +186,9 @@ function readParams(v) {
175
186
  s('audioLanguage');
176
187
  s('audioAccent');
177
188
  s('negativePrompt');
189
+ s('voiceId');
190
+ s('voiceReferenceAssetId');
191
+ s('voiceDescription');
178
192
  n('duration');
179
193
  n('imageQuantity');
180
194
  n('audioDurationSeconds');
@@ -276,6 +290,8 @@ export function shotAssetIds(spec) {
276
290
  ids.push(spec.firstFrameAssetId);
277
291
  if (spec.lastFrameAssetId)
278
292
  ids.push(spec.lastFrameAssetId);
293
+ if (spec.params.voiceReferenceAssetId)
294
+ ids.push(spec.params.voiceReferenceAssetId);
279
295
  return [...new Set(ids.filter(Boolean))];
280
296
  }
281
297
  /** Total attachment count — what a list row shows without composing anything. */