@slatesvideo/shared 0.6.3 → 0.6.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/dist/api-url.d.ts +11 -0
  2. package/dist/api-url.js +11 -0
  3. package/dist/auth.d.ts +13 -1
  4. package/dist/auth.js +9 -5
  5. package/dist/clients/cloud.d.ts +3 -0
  6. package/dist/clients/cloud.js +34 -3
  7. package/dist/clients/desktop.js +3 -0
  8. package/dist/index.d.ts +7 -2
  9. package/dist/index.js +20 -2
  10. package/dist/manual/content.d.ts +2 -0
  11. package/dist/manual/content.js +3 -0
  12. package/dist/manual/index.d.ts +5 -0
  13. package/dist/manual/index.js +20 -0
  14. package/dist/operations/index.d.ts +253 -30
  15. package/dist/operations/index.js +1387 -141
  16. package/dist/operations/surface.d.ts +69 -0
  17. package/dist/operations/surface.js +227 -0
  18. package/dist/prompts/agent-doctrine.js +8 -0
  19. package/dist/prompts/asset-label.d.ts +23 -0
  20. package/dist/prompts/asset-label.js +70 -0
  21. package/dist/prompts/banned-tokens.d.ts +15 -3
  22. package/dist/prompts/banned-tokens.js +76 -9
  23. package/dist/prompts/character-sheet.d.ts +0 -2
  24. package/dist/prompts/character-sheet.js +0 -2
  25. package/dist/prompts/craft-cards.d.ts +20 -0
  26. package/dist/prompts/craft-cards.js +82 -0
  27. package/dist/prompts/environment-sheet.js +16 -0
  28. package/dist/prompts/index.d.ts +1 -0
  29. package/dist/prompts/index.js +4 -0
  30. package/dist/prompts/model-capabilities.d.ts +52 -0
  31. package/dist/prompts/model-capabilities.js +42 -0
  32. package/dist/prompts/model-facts.d.ts +0 -4
  33. package/dist/prompts/model-facts.js +8 -4
  34. package/dist/prompts/partials.generated.js +2 -1
  35. package/dist/prompts/prompting-tips.d.ts +1 -1
  36. package/dist/prompts/prompting-tips.js +58 -0
  37. package/dist/prompts/reference-composer.d.ts +36 -7
  38. package/dist/prompts/reference-composer.js +75 -20
  39. package/dist/prompts/reference-rules.d.ts +15 -26
  40. package/dist/prompts/reference-rules.js +15 -93
  41. package/dist/prompts/shot-grammar.d.ts +154 -0
  42. package/dist/prompts/shot-grammar.js +184 -0
  43. package/dist/prompts/shot-spec.d.ts +278 -0
  44. package/dist/prompts/shot-spec.js +319 -0
  45. package/dist/skills/content.js +25 -23
  46. package/exports/slates-prompt-builder/generated/reference-content-policy.md +6 -0
  47. package/exports/slates-prompt-builder/generated/reference-kling.md +22 -0
  48. package/exports/slates-prompt-builder/generated/reference-nano-banana.md +17 -0
  49. package/exports/slates-prompt-builder/generated/reference-seedance.md +17 -0
  50. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +15 -15
  51. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  52. package/package.json +83 -73
  53. package/skills/_partials/decision-log.md +5 -4
  54. package/skills/_partials/thresholds.md +19 -0
  55. package/skills/slates-content-policy.md +15 -1
  56. package/skills/slates-cost-discipline.md +26 -4
  57. package/skills/slates-model-selection.md +4 -3
  58. package/skills/slates-one-prompt-film.md +20 -12
  59. package/skills/slates-project-organization.md +1 -1
  60. package/skills/slates-prompting-elevenlabs.md +61 -2
  61. package/skills/slates-prompting-flux-2-max.md +39 -0
  62. package/skills/slates-prompting-gpt-image-2.md +109 -70
  63. package/skills/slates-prompting-inworld-tts.md +174 -0
  64. package/skills/slates-prompting-kling-v3.md +39 -0
  65. package/skills/slates-prompting-lip-sync.md +38 -0
  66. package/skills/slates-prompting-ltx-2-5.md +38 -0
  67. package/skills/slates-prompting-minimax-h3.md +39 -0
  68. package/skills/slates-prompting-motion-transfer.md +38 -0
  69. package/skills/slates-prompting-nano-banana-2.md +26 -0
  70. package/skills/slates-prompting-omni-flash.md +41 -0
  71. package/skills/slates-prompting-seed-audio.md +39 -1
  72. package/skills/slates-prompting-seedance-2-5.md +38 -0
  73. package/skills/slates-prompting-seedance.md +26 -0
  74. package/skills/slates-prompting-seedream-5-lite.md +38 -0
  75. package/skills/slates-prompting-veo-3.md +39 -0
  76. package/skills/slates-shot-variety.md +53 -0
  77. package/skills/slates-storyboard-from-script.md +31 -15
  78. package/skills/slates-style-prompting.md +1 -1
  79. package/skills/slates-vision-feedback-loop.md +1 -1
@@ -14,6 +14,7 @@ import { SlatesCloudClient } from '../clients/cloud.js';
14
14
  import { SlatesDesktopClient } from '../clients/desktop.js';
15
15
  import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
16
16
  import { SKILLS } from '../skills/content.js';
17
+ import { appManualSections } from '../manual/index.js';
17
18
  // Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
18
19
  // "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
19
20
  import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
@@ -38,13 +39,42 @@ import { AGENT_ROUTE_PROVIDER, aspectRatioUnion, videoResolutionUnion, durationB
38
39
  // no call to skip, no discretion. `bannedTokenWarning` then reports what the
39
40
  // submitted prompt actually contained, in the result, without blocking it.
40
41
  // Never hand-type one of these tokens here; edit the skill.
41
- import { describeBannedTokens, bannedTokenWarning } from '../prompts/banned-tokens.js';
42
+ import { describeBannedTokens, bannedTokenWarning, describeBannedTokensForSkill, } from '../prompts/banned-tokens.js';
43
+ // 🚨 THE OTHER HALF OF THE SAME LESSON. The banned list is the NEGATIVE half and
44
+ // it rides an op description. The CRAFT CARD is the POSITIVE half — the levers
45
+ // that make a shot good rather than merely un-bad — and it rides the estimate
46
+ // RESULT, which the doctrine already makes the agent call before generating.
47
+ // Zero prefix bytes, present at the moment the model has just been named.
48
+ import { describeCraftCard } from '../prompts/craft-cards.js';
49
+ // ONE captioning rule for a generated asset, shared with the desktop gallery.
50
+ import { assetCaption } from '../prompts/asset-label.js';
51
+ // 🚨 THE ONE ROLE LIST + the Shot shape, mirrored into the desktop. The op
52
+ // surface DERIVES its per-role params from these rather than naming roles by
53
+ // hand — a hand-typed bucket name is what `attachmentRoles` exists to kill.
54
+ import { ORDERED_ATTACHMENT_ROLES, ATTACHMENT_ROLE_DESCRIPTION, SCRIPT_FIELD_DESCRIPTION, SCRIPT_TEXT_FIELDS, } from '../prompts/shot-spec.js';
55
+ import { SHOT_SIZE_BUCKETS, CAMERA_MOVE_BUCKETS, SPEECH_RATE, } from '../prompts/shot-grammar.js';
56
+ // Annotations, tiers and the ONE schema renderer. Declared next door so this
57
+ // module never hand-sets a hint or a tier per op: `annotate()` derives all four
58
+ // from the id and the lockstep check re-derives them from the transport verbs.
59
+ import { annotate, groupFor, tierFor, toolDefinitions, GROUP_SUMMARY, OPERATION_GROUPS, } from './surface.js';
42
60
  export function defaultContext() {
43
61
  return {
44
62
  cloud: () => new SlatesCloudClient(),
45
63
  desktop: () => new SlatesDesktopClient(),
46
64
  };
47
65
  }
66
+ /** Thrown by an op that noticed `ctx.signal` had aborted mid-work. */
67
+ export class OperationCancelledError extends Error {
68
+ code = 'OPERATION_CANCELLED';
69
+ constructor(detail) {
70
+ super(`Cancelled by the user${detail ? ` — ${detail}` : ''}.`);
71
+ }
72
+ }
73
+ /** Throw if the caller has cancelled. Call between billable items, never inside one. */
74
+ function throwIfCancelled(ctx, detail) {
75
+ if (ctx.signal?.aborted)
76
+ throw new OperationCancelledError(detail);
77
+ }
48
78
  // ── Helpers ─────────────────────────────────────────────────────
49
79
  /**
50
80
  * `z.enum` over a GENERATED list. Zod's signature wants a non-empty tuple
@@ -72,8 +102,97 @@ function ok(data, text) {
72
102
  // the same credit value); balances the same. Read via creditCost(); display
73
103
  // via fmtCredits(). The confirm gate fires above CONFIRM_CREDITS (≈ the old
74
104
  // $0.50 gate at the 3¢/credit peg).
75
- const CONFIRM_CREDITS = 17;
105
+ export const CONFIRM_CREDITS = 17;
76
106
  const CENTS_PER_CREDIT = 3; // peg: 1 credit = 3¢ billed = 2¢ COGS (mirror of slates-api)
107
+ /**
108
+ * The desktop deviation guard's ceiling multiplier: the Studio Agent pauses and
109
+ * re-asks when projected generation spend exceeds the approved ledger by more
110
+ * than this factor.
111
+ *
112
+ * 🚨 EXPORTED BECAUSE THE NUMBER IS QUOTED DOWNSTREAM. `slates-cost-discipline`
113
+ * told the agent the threshold was 25% while `loop.ts` paused at 20% — a skill
114
+ * contradicting the code on the one number that decides whether a run stops.
115
+ * The skill now renders it from here through `_partials/thresholds.md`; the
116
+ * desktop loop imports it instead of declaring its own.
117
+ */
118
+ export const DEVIATION_FACTOR = 1.2;
119
+ /**
120
+ * The confirm-gate sentence, rendered ONCE from CONFIRM_CREDITS.
121
+ *
122
+ * It was hand-typed three different ways for the same constant — "$0.50" on the
123
+ * image and video ops, "17 credits" on audio — so a rate change would have had
124
+ * to find three wordings. Byte-stable (a template over a literal), so the
125
+ * desktop's prompt-cached prefix is unaffected.
126
+ */
127
+ const CONFIRM_GATE_SENTENCE = `Cost above ${CONFIRM_CREDITS} credits returns requires_confirm — pass confirm=true after explicit user OK.`;
128
+ // Declared HERE, above every op, because `slates_estimate_generation_cost`
129
+ // renders them into its `duration` description at MODULE LOAD — a const
130
+ // declared below the first schema that reads it is a temporal-dead-zone
131
+ // crash, not a lint nit (the same rule the VIDEO_MODELS block states).
132
+ /**
133
+ * Per-surface bounds and defaults.
134
+ *
135
+ * 🚨 THESE FOUR NUMBERS PER SURFACE LIVE IN THREE REPOS. A change is a
136
+ * three-site edit, every time:
137
+ * 1. HERE (`audioCostKey`, the agent's pre-flight quote)
138
+ * 2. `slate/src/shared/pricing.ts` → MODEL_REGISTRY `audio.durationSeconds`
139
+ * (min/max/default), read by `clampAudioDuration` + `audioCreditKey`
140
+ * 3. `slates-api/src/lib/audio-keys.ts` → the server's fail-closed bounds
141
+ * `slates-api/scripts/pricing-consistency-check.mjs` §4 asserts 1 and 2 agree
142
+ * at EVERY value including out-of-range ones; the gate check covers 3.
143
+ *
144
+ * The MINs used to be missing here, and the clamp floor was a hardcoded 1. That
145
+ * made `slates_estimate_generation_cost({model:'seed-audio', duration:2})`
146
+ * quote a real `seed-audio-2s` price for a generation the desktop would bill as
147
+ * 3s and the proxy would REJECT outright. Same for an omitted duration, which
148
+ * quoted `seed-audio-1s` against the desktop's `seed-audio-15s`.
149
+ */
150
+ export const SEED_AUDIO_MIN_SECONDS = 3;
151
+ export const SEED_AUDIO_MAX_SECONDS = 120;
152
+ export const SEED_AUDIO_DEFAULT_SECONDS = 15;
153
+ export const ELEVEN_SFX_MIN_SECONDS = 1;
154
+ export const ELEVEN_SFX_MAX_SECONDS = 22;
155
+ export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
156
+ // ── The TTS character bucket ────────────────────────────────────────────────
157
+ //
158
+ // TTS is the ONE audio surface that does not bill per second: speech length
159
+ // falls out of the text, so there is no duration to charge for. It bills on a
160
+ // bucket of CHARACTERS instead, and the bucket exists because of the minimum
161
+ // billable floor rather than for tidiness.
162
+ //
163
+ // At $20.8/M characters and a 1.5× markup, a 200-character line is $0.006 of
164
+ // basis — under `MIN_AUDIO_BILLABLE_DOLLARS` ($0.01), so it bills the floor.
165
+ // The floor stops biting at 321 characters. A bucket SMALLER than that would be
166
+ // entirely floor-bound (every bucket the same price, so the displayed rate stops
167
+ // tracking cost and becomes a lie); a much LARGER one over-bills the short lines
168
+ // this feature is mostly for. 250 splits that difference and divides 2,000
169
+ // exactly, giving eight buckets and no ragged last one.
170
+ //
171
+ // 🚨 THE CAP IS READ FROM THE CAPABILITY SSOT, NEVER TYPED. 2,000 is the
172
+ // vendor's MEASURED limit (the API rejects 2,001 by name), and this module
173
+ // already imports MODEL_CAPABILITIES — so typing the number here would be a
174
+ // second copy of a fact one file away, which is exactly what that SSOT exists
175
+ // to delete. `getModelCapability` throws nothing on a miss, so the `??` would
176
+ // hide a renamed model: assert instead, at module load, where it fails the
177
+ // build rather than shipping a bucket ceiling of `undefined`.
178
+ export const TTS_MODEL = 'inworld-tts-2';
179
+ export const TTS_BUCKET_CHARS = 250;
180
+ export const TTS_MAX_CHARACTERS = (() => {
181
+ const max = getModelCapability(TTS_MODEL)?.maxCharacters;
182
+ if (!max)
183
+ throw new Error(`MODEL_CAPABILITIES['${TTS_MODEL}'] must declare maxCharacters`);
184
+ return max;
185
+ })();
186
+ export const TTS_BUCKET_COUNT = TTS_MAX_CHARACTERS / TTS_BUCKET_CHARS; // 8
187
+ /** The seat's cloning spec — reference-clip bounds, the design-prompt bounds
188
+ * and the workspace-wide clone rate — read from the SSOT for the same reason
189
+ * as the cap: every number in it was MEASURED and lives in exactly one row. */
190
+ export const TTS_VOICE_CLONE = (() => {
191
+ const spec = getModelCapability(TTS_MODEL)?.voiceClone;
192
+ if (!spec)
193
+ throw new Error(`MODEL_CAPABILITIES['${TTS_MODEL}'] must declare voiceClone`);
194
+ return spec;
195
+ })();
77
196
  function creditCost(m) {
78
197
  if (!m)
79
198
  return 0;
@@ -89,9 +208,10 @@ function creditsFromDollars(dollars) {
89
208
  const cents = Math.round(dollars * 100);
90
209
  return cents <= 0 ? 0 : Math.max(1, Math.ceil(cents / CENTS_PER_CREDIT));
91
210
  }
92
- // Shared describe-text for the background flag on every generate_* op.
93
- const BACKGROUND_DESCRIBE = 'Submit and return immediately with generationId(s) instead of blocking until the file is saved. ' +
94
- 'Poll with slates_get_generation_status. Recommended for video (1-5 min renders).';
211
+ // Shared describe-text for the background flag on every generate_* op. ONE
212
+ // sentence: it is repeated verbatim on seven ops, so every word costs seven
213
+ // times, and `slates_get_generation_status` explains the polling itself.
214
+ const BACKGROUND_DESCRIBE = 'Return generationId(s) immediately instead of blocking; poll slates_get_generation_status. Recommended for video.';
95
215
  // ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
96
216
  //
97
217
  // QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
@@ -109,7 +229,7 @@ const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it
109
229
  const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
110
230
  'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
111
231
  const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
112
- 'slates_get_asset_video_frames for video.';
232
+ 'slates_get_asset_video_frames for video. For audio, audition the saved file; metadata alone does not establish voice similarity or delivery quality.';
113
233
  // The image saved, but reading it back off disk failed (best-effort fetch). The
114
234
  // agent has an asset and NO pixels, which is the one state where a quality
115
235
  // claim would be pure invention — so this branch has to say so rather than
@@ -133,17 +253,40 @@ function backgroundSubmitted(kind, ids, extra, note) {
133
253
  // ── Workspace + identity ────────────────────────────────────────
134
254
  export const getWorkspaceState = {
135
255
  id: 'slates_get_workspace_state',
136
- description: 'Snapshot of the user\'s Slates workspace: projects list, optional active project detail. Call once at the start of a workflow to seed your understanding.',
137
- input: z.object({ projectId: z.string().optional() }),
256
+ description: 'Snapshot of the user\'s Slates workspace: the project list (most recent first) plus the active project in full when you name one. Call once at the start of a workflow to seed your understanding.',
257
+ input: z.object({
258
+ projectId: z.string().optional(),
259
+ limit: z.number().int().min(1).max(200).optional().describe('How many projects to list, newest first. Default 40.'),
260
+ }),
138
261
  async run(input, ctx) {
139
262
  const desktop = ctx.desktop();
140
263
  const { projects } = await desktop.get('/agent/projects');
264
+ // 🚨 COMPACT ROWS, AND A CAP. This returned every project's FULL row — fine
265
+ // at fourteen projects and unbounded by construction, on the one op the
266
+ // doctrine tells the agent to call first in every session. The fields kept
267
+ // are the ones a routing decision actually uses; the full row is one
268
+ // slates_get_project away.
269
+ const limit = input.limit ?? 40;
270
+ const rows = (projects ?? []);
271
+ const compact = rows.slice(0, limit).map((p) => ({
272
+ id: p.id,
273
+ name: p.name,
274
+ asset_count: p.assetCount ?? p.asset_count ?? undefined,
275
+ updated_at: p.updatedAt ?? p.updated_at ?? undefined,
276
+ }));
141
277
  let activeProject = undefined;
142
278
  if (input.projectId) {
143
279
  const r = await desktop.get('/agent/projects/get', { id: input.projectId });
144
280
  activeProject = r.project;
145
281
  }
146
- return ok({ projects, activeProject });
282
+ return ok({
283
+ projects: compact,
284
+ project_count: rows.length,
285
+ ...(rows.length > compact.length
286
+ ? { truncated: `${rows.length - compact.length} more — raise limit or call slates_list_projects.` }
287
+ : {}),
288
+ activeProject,
289
+ });
147
290
  },
148
291
  };
149
292
  export const getMe = {
@@ -351,8 +494,14 @@ export const estimateGenerationCost = {
351
494
  // hand-typed "Video 3-15" here was wrong the day seedance-2.5 (4-30s)
352
495
  // shipped, and "Seedance defaults to 1080p" was wrong for 2.5, which has no
353
496
  // 1080p at all.
354
- duration: z.number().int().min(1).max(360).optional().describe(`Seconds; cost scales linearly. Required with a video or audio base id. Video per model: ${describeDurations(VIDEO_MODELS)}. Audio: seed-audio 3-120 (⚠️ the requested duration IS the bill), eleven-sfx 1-22.`),
355
- videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe(`Video only. Omitted, each model quotes at its own default. Per model: ${describeVideoResolutions(VIDEO_MODELS)}`),
497
+ // 🚨 THE PER-MODEL TABLES ARE NOT REPEATED HERE. `slates_generate_video`
498
+ // carries `describeDurations` and `describeVideoResolutions` and is always
499
+ // in context beside this op; a second copy is 1.2 KB of the same generated
500
+ // text on every turn, and it would go stale in exactly one direction — the
501
+ // one where someone edits a table and forgets there were two.
502
+ duration: z.number().int().min(1).max(360).optional().describe(`Seconds; cost scales linearly. Required with a video or PER-SECOND audio base id. Per-model windows: see slates_generate_video's duration. Audio: seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (⚠️ the requested duration IS the bill), eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. ⛔ NOT for ${TTS_MODEL} — pass \`characters\`.`),
503
+ characters: z.number().int().min(1).max(TTS_MAX_CHARACTERS).optional().describe(`${TTS_MODEL} only — the LENGTH OF THE TEXT to speak (${TTS_BUCKET_CHARS}-char buckets).`),
504
+ videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe('Video only. Omitted, each model quotes at its own default. Per-model ladders: see slates_generate_video\'s videoResolution.'),
356
505
  resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
357
506
  quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
358
507
  sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
@@ -367,7 +516,11 @@ export const estimateGenerationCost = {
367
516
  let key = byKey.has(input.model) ? input.model : null;
368
517
  // 2) image base id + resolution (+ quality for gpt-image-2)
369
518
  if (!key) {
370
- const img = ['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2', 'flux-2-max', 'seedream-5-lite'].find((m) => m === input.model);
519
+ // IMAGE_MODELS, never a second hand-typed copy: this list is declared
520
+ // below (a runtime read, so no temporal-dead-zone hazard) and is the same
521
+ // enum `slates_generate_image` accepts. Two copies is how the estimate op
522
+ // would quietly stop pricing the seventh image model.
523
+ const img = IMAGE_MODELS.find((m) => m === input.model);
371
524
  if (img)
372
525
  key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
373
526
  }
@@ -377,7 +530,27 @@ export const estimateGenerationCost = {
377
530
  // a seedance spelling.
378
531
  if (!key && AUDIO_MODELS.includes(input.model)) {
379
532
  const m = input.model;
380
- if (!input.duration) {
533
+ // The TTS seat prices on CHARACTERS, so it leaves this per-second lane
534
+ // entirely. Asking for a duration here would quote a number that does not
535
+ // exist for it.
536
+ if (m === TTS_MODEL) {
537
+ if (!input.characters) {
538
+ return ok({
539
+ requires_clarification: true,
540
+ missing: ['characters'],
541
+ message: `${TTS_MODEL} bills per character, not per second — there is no duration to pass. Send the LENGTH OF THE TEXT to be spoken as \`characters\` (1-${TTS_MAX_CHARACTERS}).`,
542
+ });
543
+ }
544
+ if (input.characters > TTS_MAX_CHARACTERS) {
545
+ return ok({
546
+ requires_clarification: true,
547
+ missing: ['characters'],
548
+ message: `${TTS_MODEL} accepts up to ${TTS_MAX_CHARACTERS} characters in one take — ${input.characters} would be refused at generation time. Split the text and quote each line.`,
549
+ });
550
+ }
551
+ key = audioCostKey({ model: m, characters: input.characters });
552
+ }
553
+ if (!key && !input.duration) {
381
554
  return ok({
382
555
  requires_clarification: true,
383
556
  missing: ['duration'],
@@ -392,20 +565,26 @@ export const estimateGenerationCost = {
392
565
  // real 3s price and no hint that 2s is not a thing it can order. The
393
566
  // generate op gates the same way; the two must agree or the quote is a
394
567
  // promise the generation refuses to keep.
568
+ // Only the PER-SECOND surfaces have duration bounds. Keyed by those two
569
+ // ids rather than by AudioModel so adding a third surface on a different
570
+ // unit is a compile error here instead of a silent `undefined` bounds
571
+ // lookup that would throw at quote time.
395
572
  const audioBounds = {
396
573
  'seed-audio': { min: SEED_AUDIO_MIN_SECONDS, max: SEED_AUDIO_MAX_SECONDS },
397
574
  'eleven-sfx': { min: ELEVEN_SFX_MIN_SECONDS, max: ELEVEN_SFX_MAX_SECONDS },
398
575
  };
399
- const bounds = audioBounds[m];
400
- if (input.duration < bounds.min || input.duration > bounds.max) {
401
- return ok({
402
- requires_clarification: true,
403
- missing: ['duration'],
404
- message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
405
- 'Re-ask with a duration in range.',
406
- });
576
+ if (!key && m !== TTS_MODEL && input.duration) {
577
+ const bounds = audioBounds[m];
578
+ if (input.duration < bounds.min || input.duration > bounds.max) {
579
+ return ok({
580
+ requires_clarification: true,
581
+ missing: ['duration'],
582
+ message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
583
+ 'Re-ask with a duration in range.',
584
+ });
585
+ }
586
+ key = audioCostKey({ model: m, durationSeconds: input.duration });
407
587
  }
408
- key = audioCostKey({ model: m, durationSeconds: input.duration });
409
588
  }
410
589
  // 2b) Kling O3 edit base id + duration (ceiled source-clip length)
411
590
  if (!key && (input.model === 'kling-v3.0-omni-edit' || input.model === 'kling-v3.0-omni-pro-edit')) {
@@ -461,10 +640,23 @@ export const estimateGenerationCost = {
461
640
  }
462
641
  const perCredits = key != null ? byKey.get(key) : undefined;
463
642
  if (key == null || perCredits == null) {
464
- throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | nano-banana-2 | flux-2-max | seedream-5-lite) plus duration/resolution params, or use slates_list_available_models with a filter.`);
643
+ // Every id in the error comes from the SSOT arrays. The image half was
644
+ // hand-typed and named three of six, so an agent that mis-spelled
645
+ // `gpt-image-2` was told the model did not exist.
646
+ throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | ${IMAGE_MODELS.join(' | ')}) plus duration/resolution params, or use slates_list_available_models with a filter.`);
465
647
  }
466
648
  const qty = input.quantity ?? 1;
467
649
  const totalCredits = perCredits * qty;
650
+ // 🚨 THE CRAFT CARD RIDES THIS RESULT. Measured 2026-08-30: the never-use
651
+ // list inlined where the agent could not skip it moved compliance 0/8 →
652
+ // 30/32, while the same guidance behind `slates_get_prompting_guide` sat at
653
+ // 13% before AND after. This is that placement applied to the POSITIVE half.
654
+ // It is here rather than in a param description because a description is
655
+ // paid for on every turn of every session and read once — this is paid for
656
+ // only by the call that is about to use the model.
657
+ const skill = promptingSkillFor(input.model);
658
+ const card = describeCraftCard(skill);
659
+ const banned = describeBannedTokensForSkill(skill);
468
660
  return ok({
469
661
  model: input.model,
470
662
  cost_key: key,
@@ -472,7 +664,11 @@ export const estimateGenerationCost = {
472
664
  cost_per_credits: perCredits,
473
665
  total_credits: totalCredits,
474
666
  requires_confirm: totalCredits > CONFIRM_CREDITS,
475
- }, `${input.model}${qty > 1 ? ` ×${qty}` : ''}: ${fmtCredits(totalCredits)} (${key}).`);
667
+ ...(card ? { craft_card: card, craft_card_skill: skill } : {}),
668
+ ...(banned ? { banned_tokens: banned } : {}),
669
+ }, `${input.model}${qty > 1 ? ` ×${qty}` : ''}: ${fmtCredits(totalCredits)} (${key}).` +
670
+ (card ? `\n\n--- HOW TO PROMPT ${input.model} ---\n${card}` : '') +
671
+ (banned ? `\n\n${banned}` : ''));
476
672
  },
477
673
  };
478
674
  // ── Projects ────────────────────────────────────────────────────
@@ -510,11 +706,17 @@ export const getProject = {
510
706
  * tool result). Every op that embeds asset lists uses this. */
511
707
  function compactAsset(a) {
512
708
  const r = a;
513
- const prompt = typeof r.prompt === 'string' ? r.prompt : '';
514
709
  return {
515
710
  id: r.id,
516
711
  code: r.code ?? null,
517
- label: r.label ?? (prompt ? prompt.slice(0, 40) : null),
712
+ // ONE captioning rule, shared with the desktop gallery — see
713
+ // prompts/asset-label.ts for why the first 40 characters of a prompt is not
714
+ // a label on reference-driven work.
715
+ label: assetCaption({
716
+ label: typeof r.label === 'string' ? r.label : null,
717
+ shotName: typeof r.shotName === 'string' ? r.shotName : null,
718
+ prompt: typeof r.prompt === 'string' ? r.prompt : null,
719
+ }),
518
720
  type: r.type,
519
721
  created_at: r.createdAt ?? r.created_at ?? undefined,
520
722
  };
@@ -913,6 +1115,7 @@ export const createEnvironment = {
913
1115
  };
914
1116
  export const generateCharacterIdentity = {
915
1117
  id: 'slates_generate_character_identity',
1118
+ billable: true,
916
1119
  description: "Generate one character identity sheet from a base portrait asset and bind it as the character's canonical reference. Call after slates_create_character. Read slates-character-identity before calling and quote the cost from slates_estimate_generation_cost.",
917
1120
  input: z.object({
918
1121
  characterId: z.string().uuid(),
@@ -936,6 +1139,7 @@ export const generateCharacterIdentity = {
936
1139
  };
937
1140
  export const generateEnvironmentPlate = {
938
1141
  id: 'slates_generate_environment_plate',
1142
+ billable: true,
939
1143
  description: "Generate one clean establishing image from an optional base image and bind it as the environment's canonical reference. Call after slates_create_environment and quote the cost from slates_estimate_generation_cost.",
940
1144
  input: z.object({
941
1145
  environmentId: z.string().uuid(),
@@ -979,7 +1183,11 @@ export const createStoryboard = {
979
1183
  };
980
1184
  export const getStoryboardWithFrames = {
981
1185
  id: 'slates_get_storyboard_with_frames',
982
- description: 'Deep-fetch a storyboard with all scenes and frames.',
1186
+ // 🚨 ONE BOARD, TWO READERS. Every frame carries its Shots — the same rows,
1187
+ // in the same order, that the user is looking at. Before this the agent got
1188
+ // scenes and frames with no Shots, so the user read an arranged board while
1189
+ // the agent read an unordered pile.
1190
+ description: "Deep-fetch a storyboard: every scene, every slot, and the SHOT in each slot — its script line, references, model, params and takes — plus the piece's variety distribution. This is the same board, in the same order, that the user is reading.",
983
1191
  input: z.object({ storyboardId: z.string().uuid() }),
984
1192
  async run(input, ctx) {
985
1193
  const desktop = ctx.desktop();
@@ -987,7 +1195,7 @@ export const getStoryboardWithFrames = {
987
1195
  desktop.get('/agent/storyboards/get', { id: input.storyboardId }),
988
1196
  desktop.get('/agent/scenes/full', { storyboardId: input.storyboardId }),
989
1197
  ]);
990
- return ok({ ...storyboard, scenes: scenes.scenes });
1198
+ return ok({ ...storyboard, scenes: scenes.scenes, variety: scenes.variety }, describeVarietyReport(scenes.variety));
991
1199
  },
992
1200
  };
993
1201
  export const addScene = {
@@ -1004,7 +1212,11 @@ export const addScene = {
1004
1212
  };
1005
1213
  export const addFrame = {
1006
1214
  id: 'slates_add_frame',
1007
- description: 'Add a frame (asset reference) to a scene.',
1215
+ // ⚠️ THIS OP AND `slates_create_shot({ frameId })` NOW OVERLAP: two doors to
1216
+ // "put something in a scene". Naming the wart is cheaper than merging them
1217
+ // mid-release. Use THIS one only to park an existing picture; use
1218
+ // slates_create_shot to write a beat, which is almost always what you want.
1219
+ description: 'Park an EXISTING image in a scene as a new slot. It creates a Shot for that picture automatically, because a scene is a list of Shots. To write a beat — line, references, model, prompt — use slates_create_shot instead; it needs no image and files itself.',
1008
1220
  input: z.object({
1009
1221
  projectId: z.string().uuid(),
1010
1222
  storyboardId: z.string().uuid(),
@@ -1140,13 +1352,16 @@ function imageCostKey(model, resolution, quality = 'medium') {
1140
1352
  }
1141
1353
  export const generateImage = {
1142
1354
  id: 'slates_generate_image',
1355
+ billable: true,
1143
1356
  description: 'Generate an image via Slates credits.\n' +
1144
1357
  // GENERATED from MODEL_FACTS — the hand-typed model list that stood here
1145
1358
  // was a third copy of the routing doctrine, and it had already gone stale
1146
1359
  // (it still described nano-banana-2-lite by a capability the param owns).
1147
1360
  `${describeRouting('image')}\n` +
1148
1361
  'Full table: the slates-model-selection skill. ' +
1149
- 'Pass projectId to save into a Slates project (recommended — asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use. ' +
1362
+ 'Pass projectId to save into a Slates project (recommended — asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). ' +
1363
+ CONFIRM_GATE_SENTENCE +
1364
+ ' MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use. ' +
1150
1365
  // GENERATED from the skill file's own never-use list -- the one piece of
1151
1366
  // prompting doctrine that is ALWAYS in context, because the agent has
1152
1367
  // demonstrably skipped the call that would have taught it.
@@ -1257,7 +1472,7 @@ export const generateImage = {
1257
1472
  if (!entry)
1258
1473
  throw new Error(`Model not in registry: ${costKey}`);
1259
1474
  const totalCents = creditCost(entry) * (input.count ?? 1);
1260
- // Confirm gate. Fires on cost > $0.50, AND (look-first, mirroring
1475
+ // Confirm gate. Fires above CONFIRM_CREDITS, AND (look-first, mirroring
1261
1476
  // slates_generate_video) whenever reference assets are involved
1262
1477
  // regardless of cost — the LLM must see what it's referencing before
1263
1478
  // committing spend.
@@ -1481,6 +1696,7 @@ async function pollProxyJob(cloud, jobId, options = {}) {
1481
1696
  // ── Edit image ──────────────────────────────────────────────────
1482
1697
  export const editImage = {
1483
1698
  id: 'slates_edit_image',
1699
+ billable: true,
1484
1700
  description: 'Surgically edit an existing image asset with a text instruction (e.g. \'remove the lamppost\', \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The edited result is saved as a NEW asset in the project (prompt prefixed \'[Edit]\'); the source is untouched. Default model nano-banana-2 (only model that also accepts referenceAssetIds); flux-2-max / seedream-5-lite use their own edit endpoints and ignore references. Before first use call slates_get_prompting_guide with topic \'slates-edit-and-iterate\'.',
1485
1701
  input: z.object({
1486
1702
  projectId: z.string().uuid(),
@@ -1763,31 +1979,7 @@ export function seedanceEditCostKey(input) {
1763
1979
  // Exported: the exact `model` ids slates_generate_audio accepts — consumed
1764
1980
  // by the desktop Studio Agent system prompt (SSOT; never restate these ids
1765
1981
  // in prose that can drift). Mirrors VIDEO_MODELS for the third media type.
1766
- export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx'];
1767
- /**
1768
- * Per-surface bounds and defaults.
1769
- *
1770
- * 🚨 THESE FOUR NUMBERS PER SURFACE LIVE IN THREE REPOS. A change is a
1771
- * three-site edit, every time:
1772
- * 1. HERE (`audioCostKey`, the agent's pre-flight quote)
1773
- * 2. `slate/src/shared/pricing.ts` → MODEL_REGISTRY `audio.durationSeconds`
1774
- * (min/max/default), read by `clampAudioDuration` + `audioCreditKey`
1775
- * 3. `slates-api/src/lib/audio-keys.ts` → the server's fail-closed bounds
1776
- * `slates-api/scripts/pricing-consistency-check.mjs` §4 asserts 1 and 2 agree
1777
- * at EVERY value including out-of-range ones; the gate check covers 3.
1778
- *
1779
- * The MINs used to be missing here, and the clamp floor was a hardcoded 1. That
1780
- * made `slates_estimate_generation_cost({model:'seed-audio', duration:2})`
1781
- * quote a real `seed-audio-2s` price for a generation the desktop would bill as
1782
- * 3s and the proxy would REJECT outright. Same for an omitted duration, which
1783
- * quoted `seed-audio-1s` against the desktop's `seed-audio-15s`.
1784
- */
1785
- export const SEED_AUDIO_MIN_SECONDS = 3;
1786
- export const SEED_AUDIO_MAX_SECONDS = 120;
1787
- export const SEED_AUDIO_DEFAULT_SECONDS = 15;
1788
- export const ELEVEN_SFX_MIN_SECONDS = 1;
1789
- export const ELEVEN_SFX_MAX_SECONDS = 22;
1790
- export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
1982
+ export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx', 'inworld-tts-2'];
1791
1983
  /**
1792
1984
  * Byte-for-byte the desktop's `clampAudioDuration` in slate/src/shared/pricing.ts,
1793
1985
  * INCLUDING the non-finite arm — that one matters: `Math.max(min, NaN)` is NaN,
@@ -1811,6 +2003,21 @@ function clampInt(value, min, max) {
1811
2003
  // injects "... N seconds" into the prompt and bills seed-audio-{N}s, so
1812
2004
  // display == billing with no amendment to the pricing law. The server probes
1813
2005
  // the returned audio.duration afterwards and logs SEED AUDIO BILLING DRIFT.
2006
+ /**
2007
+ * Characters → the billed bucket, byte-identical to the desktop's
2008
+ * `clampTtsCharacters`. Rounds UP to the next 250 and clamps to [250, 2000].
2009
+ *
2010
+ * The non-finite arm matters for the same reason it does on the duration side:
2011
+ * `Math.ceil(NaN)` is NaN, which would build the key `inworld-tts-2-NaNc` here
2012
+ * while the desktop quotes the first bucket. Divergence at a value neither side
2013
+ * can bill is still divergence.
2014
+ */
2015
+ function clampTtsCharacters(value) {
2016
+ if (value == null || !Number.isFinite(value))
2017
+ return TTS_BUCKET_CHARS;
2018
+ const buckets = Math.ceil(value / TTS_BUCKET_CHARS);
2019
+ return Math.min(TTS_BUCKET_COUNT, Math.max(1, buckets)) * TTS_BUCKET_CHARS;
2020
+ }
1814
2021
  export function audioCostKey(input) {
1815
2022
  if (input.model === 'seed-audio') {
1816
2023
  // `?? default` before the clamp, not `?? 0` — the desktop resolves a missing
@@ -1823,6 +2030,11 @@ export function audioCostKey(input) {
1823
2030
  const secs = clampAudioSeconds(input.durationSeconds ?? ELEVEN_SFX_DEFAULT_SECONDS, ELEVEN_SFX_MIN_SECONDS, ELEVEN_SFX_MAX_SECONDS, ELEVEN_SFX_DEFAULT_SECONDS);
1824
2031
  return `eleven-sfx-${secs}s`;
1825
2032
  }
2033
+ if (input.model === TTS_MODEL) {
2034
+ // `c` for characters, so the key can never be mistaken for a seconds key by
2035
+ // the `-\d+s$` tests the proxy and the desktop both run.
2036
+ return `${TTS_MODEL}-${clampTtsCharacters(input.characters)}c`;
2037
+ }
1826
2038
  throw new Error(`Unknown audio model: ${input.model}`);
1827
2039
  }
1828
2040
  /**
@@ -1941,33 +2153,36 @@ function spokenTextByAssetId(assetIds, spoken) {
1941
2153
  });
1942
2154
  return Object.keys(out).length > 0 ? out : undefined;
1943
2155
  }
1944
- // Maps a video model id to its bundled prompting skill (frontmatter `name:`),
1945
- // so guidance text points at a skill that actually exists. Deriving the name
1946
- // via model.split('-')[0] produced 'slates-prompting-kling' / '...-veo', which
1947
- // match no file — only seedance happened to line up.
2156
+ // Maps a model id to its bundled prompting skill (frontmatter `name:`), so
2157
+ // guidance text points at a skill that actually exists.
2158
+ //
2159
+ // 🚨 ONE RESOLVER. This was a second hand-typed alias table beside
2160
+ // `resolveGuideTopic()`, and it had already fallen behind: it knew nothing
2161
+ // about LTX-2.5, so every LTX generation was told to read the cost skill
2162
+ // instead of its own guide. Delegate; the ordering traps (2.5 before seedance,
2163
+ // seed-audio before seedance, minimax before both) are solved once, there.
1948
2164
  function promptingSkillFor(model) {
1949
- if (model.startsWith('kling'))
1950
- return 'slates-prompting-kling-v3';
1951
- if (model.startsWith('veo'))
1952
- return 'slates-prompting-veo-3';
1953
- // 2.5 BEFORE the generic seedance test — "seedance-2.5" also starts with
1954
- // "seedance", and falling through hands 2.0's guide to a model with different
1955
- // limits, a different resolution ladder and an extra task type.
1956
- if (model.startsWith('seedance-2.5'))
1957
- return 'slates-prompting-seedance-2-5';
1958
- if (model.startsWith('seedance'))
1959
- return 'slates-prompting-seedance';
1960
- if (model.startsWith('omni-flash'))
1961
- return 'slates-prompting-omni-flash';
1962
- // ONE skill covers both H3 seats — the prompt grammar is identical and only
1963
- // the ladder and the reference transport differ — so a prefix is right here.
1964
- if (model.startsWith('minimax-h3'))
1965
- return 'slates-prompting-minimax-h3';
1966
- return 'slates-cost-discipline';
2165
+ return resolveGuideTopic(model) ?? 'slates-cost-discipline';
1967
2166
  }
2167
+ /**
2168
+ * The per-model prompting guides for the whole video roster, DERIVED.
2169
+ *
2170
+ * `slates_generate_video`'s description hand-typed five of them and omitted
2171
+ * `slates-prompting-ltx-2-5` for as long as LTX shipped — an agent reading the
2172
+ * description could not learn the guide existed. A hand-typed index of a
2173
+ * generated corpus is a stale index; it is only a matter of when.
2174
+ */
2175
+ const VIDEO_MODEL_GUIDES = [
2176
+ ...new Set(VIDEO_MODELS.map((m) => promptingSkillFor(m))),
2177
+ ].join(' / ');
1968
2178
  export const generateVideo = {
1969
2179
  id: 'slates_generate_video',
1970
- description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3 / slates-prompting-minimax-h3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8"). ' +
2180
+ billable: true,
2181
+ description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (' +
2182
+ VIDEO_MODEL_GUIDES +
2183
+ ') — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). ' +
2184
+ CONFIRM_GATE_SENTENCE +
2185
+ ' Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8"). ' +
1971
2186
  // GENERATED from the skill's own slop-token list. Always in context on both
1972
2187
  // surfaces, so it survives an agent that skips slates_get_prompting_guide.
1973
2188
  describeBannedTokens('video'),
@@ -1996,32 +2211,39 @@ export const generateVideo = {
1996
2211
  aspectRatio: zEnum(VIDEO_ASPECT_RATIOS).optional().describe(`NOT every model takes every value — an out-of-set ratio is REFUSED before submit, not silently ignored. Per model: ${describeAspectRatios(VIDEO_MODELS, AGENT_ROUTE_PROVIDER)}`),
1997
2212
  duration: z.number().int().min(VIDEO_DURATION_BOUNDS.min).max(VIDEO_DURATION_BOUNDS.max).optional().describe(`Seconds. Cost scales linearly — a 30s seedance-2.5 take is several hundred credits, so be explicit rather than defaulting. Per model: ${describeDurations(VIDEO_MODELS)}`),
1998
2213
  videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe(`An unsupported resolution is REJECTED, never downgraded. Per model: ${describeVideoResolutions(VIDEO_MODELS)}. 4K video is Pro-only (the server returns PRO_REQUIRED for a base-tier account).`),
1999
- firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video: asset UUID or badge code ("IMG-A8") — codes resolve against the project at call time, so a code the user just spoke is always safe to pass.'),
2000
- lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code). Veo and Seedance only. Pairs with firstFrameAssetId for guided transitions.'),
2214
+ // 🚨 EVERY DESCRIPTION BELOW IS A CONSTRAINT OR A GENERATED TABLE — never
2215
+ // craft, never rationale, never a worked example. This op is the single
2216
+ // largest thing in the desktop's prompt-cached prefix (measured 15,357 of
2217
+ // 112,114 bytes on 2026-09-02, 13.7%), and almost all of the excess was
2218
+ // reasoning that belongs in slates-prompting-* where it is read once, on
2219
+ // demand, by the one session that needs it. If you are about to explain
2220
+ // WHY here, you are writing the skill in the wrong file.
2221
+ firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video (UUID or badge code, resolved at call time).'),
2222
+ lastFrameAssetId: z.string().optional().describe('Ending frame. Veo and Seedance only; pairs with firstFrameAssetId.'),
2001
2223
  ingredientAssetIds: z.array(z.string()).max(30).optional().describe(
2002
2224
  // Caps DERIVED from MODEL_CAPABILITIES — the hand-typed list omitted
2003
2225
  // kling-v3.0-omni-pro entirely and read as if 7 were an Omni Flash-only rule.
2004
- `Visual reference / ingredient assets (UUIDs or badge codes). Cap per model (combined across ingredient/character/environment/style params): ${describeReferenceImageCaps(VIDEO_MODELS)}. More is not better: 2-4 strong references beat both extremes, and past 4 reference PEOPLE output stability drops on Seedance regardless of the cap.`),
2005
- characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets (UUIDs or badge codes) — keeps a character consistent across the shot.'),
2006
- environmentAssetIds: z.array(z.string()).optional().describe('Environment reference assets (UUIDs or badge codes) — keeps a location/setting consistent across the shot.'),
2007
- styleAssetIds: z.array(z.string()).optional().describe('Style reference assets (UUIDs or badge codes) — locks the visual style of the shot.'),
2008
- videoReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into videoReferenceAssetIds; prefer that for anything new. A single VIDEO asset (UUID or badge code) used as a reference. Kept working forever: installed CLI and MCP builds send this shape.'),
2009
- videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach. Required with videoReferenceAssetId: that clip\'s duration in seconds.'),
2010
- audioReferenceAssetId: z.string().optional().describe('DEPRECATED — forwarded into audioReferenceAssetIds; prefer that. A single AUDIO asset (UUID or badge code) used as a reference.'),
2226
+ `Visual reference / ingredient assets. Cap per model, combined across the ingredient/character/environment/style params: ${describeReferenceImageCaps(VIDEO_MODELS)}. 2-4 strong references beat both extremes.`),
2227
+ characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets — keeps a character consistent.'),
2228
+ environmentAssetIds: z.array(z.string()).optional().describe('Environment references — keeps a location consistent.'),
2229
+ styleAssetIds: z.array(z.string()).optional().describe('Style references — locks the look.'),
2230
+ videoReferenceAssetId: z.string().optional().describe('DEPRECATED — use videoReferenceAssetIds. Kept working: shipped CLI/MCP builds send this shape.'),
2231
+ videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach.'),
2232
+ audioReferenceAssetId: z.string().optional().describe('DEPRECATED — use audioReferenceAssetIds. Carries no spoken text.'),
2011
2233
  // ── Multimodal references, the plural surface ──
2012
2234
  // The capacity sentences are DERIVED from MODEL_FACTS (see
2013
2235
  // multimodalRefSummary) rather than hand-typed, so a cap change in one
2014
2236
  // place cannot leave a stale number in a description an LLM reads.
2015
- videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS (UUIDs or badge codes) read alongside the images and audio in the same generation — own-footage restyle, MOTION TRANSFER ("the character from image 1 performs the motion from video 1"), or dialogue conditioning. Cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; ignored elsewhere. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to combined input+output seconds (the vref key) — pass videoReferenceSecondsEach so the quote is right. If any clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Over the cap is REFUSED, never trimmed: a dropped clip would already have been priced in.`),
2016
- videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — the bill is Σceil(each) + output seconds. The server re-derives this by probing every uploaded clip, so an understated value just gets corrected upward.'),
2017
- audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO clips (UUIDs or badge codes) read alongside the images and video — e.g. lip-sync a character to a line ("the character in image 1 speaks the dialogue from audio 1"). Cited as "audio 1", "audio 2"… in the order given. No billing surcharge (Seedance audio is included). ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
2018
- audioReferenceSpokenText: z.array(z.string()).optional().describe('STRONGLY RECOMMENDED whenever a reference clip contains SPEECH. Same order and length as audioReferenceAssetIds; use "" for a clip with no words (music, ambience, room tone). The model RE-TRANSCRIBES a supplied take rather than using it verbatim — a field test heard "an app called Slates" come back as "a map called Slates" — so the audio decides the VOICE, the ACCENT and the TIMING while only text decides the WORDS. Give the exact line here and it is quoted into the prompt beside the citation. Omit it and the words are a guess. Pairs with the plural audioReferenceAssetIds; the deprecated singular audioReferenceAssetId carries no text.'),
2237
+ videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS, cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to the vref key (input+output seconds) — pass videoReferenceSecondsEach. Over the cap is REFUSED, never trimmed.`),
2238
+ videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each clip\'s duration in seconds. Feeds the vref cost key; the server re-probes and corrects an understated value upward.'),
2239
+ audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO, cited as "audio 1", "audio 2"… in the order given. No billing surcharge. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
2240
+ audioReferenceSpokenText: z.array(z.string()).optional().describe('The exact words in each reference clip — same order and length as audioReferenceAssetIds, "" for a clip with no speech. The model RE-TRANSCRIBES a take rather than using it verbatim, so audio decides voice/accent/timing and only this decides the WORDS. Omit it and the words are a guess.'),
2019
2241
  sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
2020
2242
  audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
2021
2243
  generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
2022
- seedanceFace: z.boolean().optional().describe('Seedance ONLY: set true when a reference/ingredient shows an AI-character\'s FACE. Faces are blocked on the default (cheaper) Seedance route, so this reroutes the gen to a face-capable provider at ~45% more (the cost key becomes seedance-2-face-*). Leave false/unset for faceless or object-only references. No effect on Kling/Veo. A REAL person\'s photo is rejected on this route — the failure message contains [REAL_FACE_DETECTED]; see seedanceRealFace.'),
2023
- seedanceRealFace: z.boolean().optional().describe('Seedance ONLY: the reference shows a REAL person (a photo of an actual human, not an AI character). Routes to the premium real-face provider (cost key seedance-2-realface-*, roughly 2x the AI-face price — quote it via slates_estimate_generation_cost first). REQUIRES realFaceConsent=true. Typical flow: a seedanceFace gen fails with [REAL_FACE_DETECTED] → ask the user to confirm consent + the higher price → retry with seedanceRealFace=true + realFaceConsent=true.'),
2024
- realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace: set true ONLY after the user has explicitly confirmed they hold the rights/consent to this person\'s likeness and it doesn\'t impersonate or misrepresent them. The generation is refused without it. Public figures/celebrities fail on every route.'),
2244
+ seedanceFace: z.boolean().optional().describe('Seedance ONLY: a reference shows an AI CHARACTER\'s face. Faces are blocked on the default route, so this reroutes to a face-capable provider at ~45% more. A REAL person fails here with [REAL_FACE_DETECTED] — see seedanceRealFace.'),
2245
+ seedanceRealFace: z.boolean().optional().describe('Seedance ONLY: a reference shows a REAL person. Premium route, roughly 2x the AI-face price — quote it first. REQUIRES realFaceConsent=true.'),
2246
+ realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace: true ONLY after the user has explicitly confirmed they hold rights/consent to this likeness and it does not impersonate or misrepresent them. Refused without it; public figures fail on every route.'),
2025
2247
  negativePrompt: z.string().optional(),
2026
2248
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2027
2249
  confirm: z.boolean().optional().describe('Set true after explicit user OK to bypass the confirm gate (which fires for almost every video gen since they\'re expensive).'),
@@ -2374,7 +2596,7 @@ export const generateVideo = {
2374
2596
  }
2375
2597
  const totalCents = creditCost(entry);
2376
2598
  // Pre-flight confirm gate. Fires when:
2377
- // (a) cost > $0.50 (the cost gate), OR
2599
+ // (a) cost above CONFIRM_CREDITS (the cost gate), OR
2378
2600
  // (b) any reference assets are involved (the look-first gate)
2379
2601
  // When references are present, the response inlines them as image
2380
2602
  // content blocks (and video keyframes for video refs) so the LLM
@@ -2515,13 +2737,70 @@ export const generateVideo = {
2515
2737
  };
2516
2738
  },
2517
2739
  };
2740
+ // ── The preset voice shelf ──────────────────────────────────────
2741
+ /**
2742
+ * AGENT PARITY for the voice picker. The desktop browses stock voices by
2743
+ * gender, accent and age and plays each one; until this op the agent could
2744
+ * only pass a `voiceId` it had no way to discover. Disk reads on the desktop,
2745
+ * never a vendor call — browsing is free on every surface.
2746
+ */
2747
+ export const listVoices = {
2748
+ id: 'slates_list_voices',
2749
+ description: `Browse ${TTS_MODEL} preset voices. Pass a returned voiceId to slates_generate_audio. Filters AND together.`,
2750
+ input: z.object({
2751
+ gender: z.string().optional().describe('male | female'),
2752
+ accent: z.string().optional().describe('Region ("GB") or languageCode ("en-GB"); available accents come from the shelf.'),
2753
+ age: z.string().optional().describe('young | middle_aged | elderly'),
2754
+ query: z.string().optional().describe('Free text over name, description, tags.'),
2755
+ }),
2756
+ async run(input, ctx) {
2757
+ const desktop = ctx.desktop();
2758
+ await desktop.requireCapability('voices', 'the preset voice shelf');
2759
+ const shelf = await desktop.get('/agent/voices');
2760
+ if (!shelf.available) {
2761
+ return ok({ available: false, voices: [], message: 'This desktop build shipped without the preset shelf.' });
2762
+ }
2763
+ const terms = (input.query ?? '').toLowerCase().split(/\s+/).filter(Boolean);
2764
+ const region = input.accent?.includes('-') ? input.accent.split('-')[1] : input.accent;
2765
+ const voices = shelf.voices
2766
+ .filter((v) => !input.gender || v.gender === input.gender)
2767
+ .filter((v) => !region || v.languageCode.split('-')[1]?.toUpperCase() === region.toUpperCase())
2768
+ .filter((v) => !input.age || v.ageGroup === input.age)
2769
+ .filter((v) => {
2770
+ if (terms.length === 0)
2771
+ return true;
2772
+ const hay = [v.displayName, v.description, v.gender, v.ageGroup, v.languageCode, ...(v.tags ?? [])]
2773
+ .join(' ')
2774
+ .toLowerCase();
2775
+ return terms.every((t) => hay.includes(t));
2776
+ })
2777
+ .map(({ voiceId, displayName, description, tags, gender, ageGroup, languageCode }) => ({
2778
+ voiceId,
2779
+ displayName,
2780
+ description,
2781
+ tags,
2782
+ gender,
2783
+ ageGroup,
2784
+ languageCode,
2785
+ }));
2786
+ return ok({
2787
+ available: true,
2788
+ line: shelf.line,
2789
+ count: voices.length,
2790
+ voices,
2791
+ next: `Pass a voiceId to slates_generate_audio (model ${TTS_MODEL}) as voiceId. To keep one on a character for reuse, generate a clip with it and set slates_update_character voiceAssetId.`,
2792
+ });
2793
+ },
2794
+ };
2518
2795
  // ── Generate audio ──────────────────────────────────────────────
2519
2796
  export const generateAudio = {
2520
2797
  id: 'slates_generate_audio',
2521
- description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Two surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s) and eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop). Which surface for which job: read the slates-model-selection skill. ' +
2522
- '🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
2523
- 'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
2524
- 'projectId is REQUIRED (no headless path). Cost > 17 credits returns requires_confirm — pass confirm=true after explicit user OK. No skill files installed? Call slates_get_prompting_guide first.',
2798
+ billable: true,
2799
+ description: `Generate project audio using credits. Choose the surface via the model routing below. ` +
2800
+ 'Read slates-cost-discipline and the matching prompting skill first (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). ' +
2801
+ 'Seed Audio bills the requested duration, which is appended to the prompt regardless of output length. Kling "SFX:" / "Ambient noise:" syntax does not transfer. ' +
2802
+ CONFIRM_GATE_SENTENCE +
2803
+ ' No skill files installed? Call slates_get_prompting_guide first.',
2525
2804
  input: z.object({
2526
2805
  projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
2527
2806
  model: z
@@ -2534,17 +2813,31 @@ export const generateAudio = {
2534
2813
  .string()
2535
2814
  .min(1)
2536
2815
  .max(5000)
2537
- .describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars.'),
2816
+ .describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars. inworld-tts-2: THE WORDS TO SPEAK, verbatim, max ' + TTS_MAX_CHARACTERS + ' — its length is the bill.'),
2538
2817
  durationSeconds: z
2539
2818
  .number()
2540
2819
  .optional()
2541
- .describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.'),
2820
+ .describe(`seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (default ${SEED_AUDIO_DEFAULT_SECONDS}) — ⚠️ THIS IS THE BILL: appended to the prompt and charged whatever comes back. eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS} (default ${ELEVEN_SFX_DEFAULT_SECONDS}) — always sent explicitly so the per-second charge is deterministic. Not for ${TTS_MODEL}.`),
2542
2821
  voice: z
2543
2822
  .string()
2544
2823
  .optional()
2545
- .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
2824
+ .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself. Agent-facing only.'),
2825
+ voiceId: z
2826
+ .string()
2827
+ .optional()
2828
+ .describe(`${TTS_MODEL} — a preset voiceId from slates_list_voices, not a character or asset id. Exactly one voice source is required.`),
2829
+ voiceReferenceAssetId: z
2830
+ .string()
2831
+ .optional()
2832
+ .describe(`${TTS_MODEL} — clone this AUDIO asset's voice for the take (${TTS_VOICE_CLONE.minSeconds}-${TTS_VOICE_CLONE.maxSeconds}s, one clean speaker). To speak AS a character pass its voiceAssetId. Cloning: ${TTS_VOICE_CLONE.clonesPerMinute} new voices/min across all of Slates; a burst waits.`),
2833
+ voiceDescription: z
2834
+ .string()
2835
+ .min(TTS_VOICE_CLONE.designPromptChars.min)
2836
+ .max(TTS_VOICE_CLONE.designPromptChars.max)
2837
+ .optional()
2838
+ .describe(`${TTS_MODEL} — a voice from words (${TTS_VOICE_CLONE.designPromptChars.min}-${TTS_VOICE_CLONE.designPromptChars.max} chars) for a character with no recording; keep it via slates_update_character voiceAssetId.`),
2546
2839
  speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
2547
- volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
2840
+ volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline fader for mix decisions.'),
2548
2841
  pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
2549
2842
  multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
2550
2843
  loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
@@ -2553,7 +2846,7 @@ export const generateAudio = {
2553
2846
  .array(z.string())
2554
2847
  .max(3)
2555
2848
  .optional()
2556
- .describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId — the API rejects both.'),
2849
+ .describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId.'),
2557
2850
  imageReferenceAssetId: z
2558
2851
  .string()
2559
2852
  .optional()
@@ -2563,13 +2856,48 @@ export const generateAudio = {
2563
2856
  }),
2564
2857
  run: async (input, ctx) => {
2565
2858
  // ── Per-surface clarification + constraint gates ──
2859
+ //
2860
+ // `null` for the TTS seat is load-bearing rather than a placeholder: that
2861
+ // surface has NO duration dimension at all (speech length falls out of the
2862
+ // text), so there is no default that would be honest. A number here would
2863
+ // flow into `audioCostKey` and quote a per-second price for a per-character
2864
+ // generation.
2566
2865
  const cfgDefaults = {
2567
2866
  'seed-audio': SEED_AUDIO_DEFAULT_SECONDS,
2568
2867
  'eleven-sfx': ELEVEN_SFX_DEFAULT_SECONDS,
2868
+ 'inworld-tts-2': null,
2569
2869
  };
2570
- const seconds = input.durationSeconds ?? cfgDefaults[input.model];
2870
+ const seconds = input.durationSeconds ?? cfgDefaults[input.model] ?? undefined;
2871
+ if (input.model === TTS_MODEL) {
2872
+ // The text IS the prompt on this surface — there is no separate field,
2873
+ // because "the words that get spoken" and "what you asked for" are the
2874
+ // same thing here. That also means the character count is knowable
2875
+ // client-side, which is what makes the quote below exact.
2876
+ if (input.prompt.length > TTS_MAX_CHARACTERS) {
2877
+ throw new Error(`${TTS_MODEL} accepts up to ${TTS_MAX_CHARACTERS} characters in one take — this text is ${input.prompt.length}. Split it into separate lines and generate each one.`);
2878
+ }
2879
+ if (input.durationSeconds != null) {
2880
+ throw new Error(`${TTS_MODEL} has no duration parameter — speech length falls out of the text, and it bills per character. Drop durationSeconds.`);
2881
+ }
2882
+ // EXACTLY ONE voice source. Zero is a clarification (the agent should ask
2883
+ // who is speaking, not guess); two or more is an error, because the three
2884
+ // paths are genuinely different generations and silently picking one would
2885
+ // spend credits on a voice the caller did not ask for.
2886
+ const voiceSources = [input.voiceId, input.voiceReferenceAssetId, input.voiceDescription]
2887
+ .filter((v) => typeof v === 'string' && v.length > 0);
2888
+ if (voiceSources.length === 0) {
2889
+ return ok({
2890
+ requires_clarification: true,
2891
+ missing: ['voiceId'],
2892
+ message: `${TTS_MODEL} needs a voice — exactly one of three. Speaking AS a character: pass its voiceAssetId (slates_list_characters) as voiceReferenceAssetId. A stock voice: slates_list_voices lists presets by gender, accent and age; pass one's voiceId. No recording of the voice: voiceDescription (words), or voiceReferenceAssetId with any clean clip of one speaker. A voice worth reusing can be kept on a character with slates_update_character, but nothing requires that — ask the user which they want only when the request does not say.`,
2893
+ });
2894
+ }
2895
+ if (voiceSources.length > 1) {
2896
+ throw new Error(`${TTS_MODEL} takes exactly one voice source — pass voiceId OR voiceReferenceAssetId OR voiceDescription, not ${voiceSources.length} of them.`);
2897
+ }
2898
+ }
2571
2899
  if (input.model === 'seed-audio') {
2572
- if (seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
2900
+ if (seconds == null || seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
2573
2901
  return ok({
2574
2902
  requires_clarification: true,
2575
2903
  missing: ['durationSeconds'],
@@ -2581,7 +2909,7 @@ export const generateAudio = {
2581
2909
  }
2582
2910
  }
2583
2911
  if (input.model === 'eleven-sfx') {
2584
- if (seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
2912
+ if (seconds == null || seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
2585
2913
  return ok({
2586
2914
  requires_clarification: true,
2587
2915
  missing: ['durationSeconds'],
@@ -2597,6 +2925,10 @@ export const generateAudio = {
2597
2925
  const refInputs = [];
2598
2926
  for (const ref of input.audioReferenceAssetIds ?? [])
2599
2927
  refInputs.push({ ref, role: 'audio reference' });
2928
+ // Resolved through the SAME resolver as every other asset ref, so a badge
2929
+ // code ("AUD-S1") works here exactly as it does everywhere else.
2930
+ if (input.voiceReferenceAssetId)
2931
+ refInputs.push({ ref: input.voiceReferenceAssetId, role: 'voice reference' });
2600
2932
  if (input.imageReferenceAssetId)
2601
2933
  refInputs.push({ ref: input.imageReferenceAssetId, role: 'image reference' });
2602
2934
  const resolvedRefs = refInputs.length > 0
@@ -2611,6 +2943,7 @@ export const generateAudio = {
2611
2943
  const costKey = audioCostKey({
2612
2944
  model: input.model,
2613
2945
  durationSeconds: seconds,
2946
+ characters: input.model === TTS_MODEL ? input.prompt.length : undefined,
2614
2947
  });
2615
2948
  const entry = registry.models.find((m) => m.model === costKey);
2616
2949
  if (!entry) {
@@ -2643,6 +2976,9 @@ export const generateAudio = {
2643
2976
  prompt: input.prompt,
2644
2977
  durationSeconds: seconds,
2645
2978
  voice: input.voice,
2979
+ voiceId: input.voiceId,
2980
+ voiceReferenceAssetId: rid(input.voiceReferenceAssetId),
2981
+ voiceDescription: input.voiceDescription,
2646
2982
  speed: input.speed,
2647
2983
  volume: input.volume,
2648
2984
  pitch: input.pitch,
@@ -2678,7 +3014,8 @@ export const generateAudio = {
2678
3014
  // ── Generate lip-sync ───────────────────────────────────────────
2679
3015
  export const generateLipSync = {
2680
3016
  id: 'slates_generate_lip_sync',
2681
- description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip (~$0.11 / 5s); sourceType=image animates a still avatar (avatar-standard ~$0.42 / 5s; avatar-pro ~$0.86 / 5s). Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
3017
+ billable: true,
3018
+ description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip, sourceType=image animates a still avatar (avatar-standard, or avatar-pro for the premium seat). Quote each with slates_estimate_generation_cost rather than from memory. Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
2682
3019
  input: z.object({
2683
3020
  projectId: z.string().uuid().describe('Slates project the source asset lives in. The new lip-synced video lands here.'),
2684
3021
  sourceAssetId: z.string().uuid().describe('Asset id of the still image (avatar flow) or video clip (lip-sync flow). Must already exist in the project — use slates_upload_reference_image or slates_generate_image / slates_generate_video first if needed.'),
@@ -2797,12 +3134,14 @@ export const generateLipSync = {
2797
3134
  // ── Generate motion transfer ────────────────────────────────────
2798
3135
  export const generateMotionTransfer = {
2799
3136
  id: 'slates_generate_motion_transfer',
2800
- description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std ($0.95 / 5s) or kling-mc-pro ($1.26 / 5s), structured skeleton/depth retargeting, always 5s. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. Both tiers hit the >$0.50 confirm gate.',
3137
+ billable: true,
3138
+ description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std or kling-mc-pro, structured skeleton/depth retargeting, always 5s. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. ' +
3139
+ CONFIRM_GATE_SENTENCE,
2801
3140
  input: z.object({
2802
3141
  projectId: z.string().uuid().describe('Slates project. Both source and target assets must live here.'),
2803
3142
  sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s.'),
2804
3143
  targetImageAssetId: z.string().uuid().describe('Asset id of the target image (the character that will perform the motion). Must already exist in the project.'),
2805
- motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std (~32 credits) general motion; kling-mc-pro (~42 credits) cleaner anatomy — default.'),
3144
+ motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std general motion; kling-mc-pro cleaner anatomy — default. Quote both with slates_estimate_generation_cost.'),
2806
3145
  characterOrientation: z.enum(['video', 'image']).optional().describe('"video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
2807
3146
  prompt: z.string().optional().describe('Optional refinement. Read slates-prompting-motion-transfer.'),
2808
3147
  klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. "fal" (default) uses Slates credits.'),
@@ -2837,7 +3176,15 @@ export const generateMotionTransfer = {
2837
3176
  target_ref: target,
2838
3177
  message: `Cost: ${fmtCredits(totalCents)} for 5s ${motionModel} (${costKey}). ` +
2839
3178
  `Transferring motion from ${source} onto ${target}. ` +
2840
- `Re-call with confirm=true after the user explicitly OKs the spend, or pick kling-mc-std to save ~10 credits. ` +
3179
+ // The saving is READ from the registry, never guessed: "~10 credits"
3180
+ // was hand-typed and is a rate change away from being a lie.
3181
+ `Re-call with confirm=true after the user explicitly OKs the spend${motionModel === 'kling-mc-pro'
3182
+ ? (() => {
3183
+ const std = registry.models.find((m) => m.model === 'kling-mc-std-5s');
3184
+ const saving = std ? totalCents - creditCost(std) : 0;
3185
+ return saving > 0 ? `, or pick kling-mc-std to save ${fmtCredits(saving)}` : '';
3186
+ })()
3187
+ : ''}. ` +
2841
3188
  `When discussing with the user, refer to the assets by those codes — they'll match the gallery badges.`,
2842
3189
  });
2843
3190
  }
@@ -2891,7 +3238,15 @@ export const generateMotionTransfer = {
2891
3238
  // ── Edit video (Kling O3 video-to-video) ────────────────────────
2892
3239
  export const editVideo = {
2893
3240
  id: 'slates_edit_video',
2894
- description: 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3–15s clips, 720–3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3–10s clips, 720p output, PROMPT-ONLY — no refs, cheapest seat), or seedance-2.5-edit (4–30s clips — the ONLY engine that takes a clip over 15s; up to 1080p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (≈ clip length, rounded UP to the next second): omni-flash-edit ≈ 19¢/s ≈ kling-v3.0-omni-edit ≈ 19¢/s, kling-v3.0-omni-pro-edit ≈ 25¢/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds — always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
3241
+ billable: true,
3242
+ description:
3243
+ // 🚨 NO PRICES, NO HAND-TYPED WINDOWS. This description carried five dollar
3244
+ // figures and three duration/resolution claims. The prices contradicted the
3245
+ // agent's own REAL NUMBERS ONLY rule (it may not repeat a figure it cannot
3246
+ // point to in a tool result) and go stale on the next rate change; the
3247
+ // windows are owned by MODEL_CAPABILITIES and are generated below into the
3248
+ // params that enforce them.
3249
+ 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default — subject/style refs via elements), omni-flash-edit (PROMPT-ONLY, no refs, the cheapest seat), or seedance-2.5-edit (the only engine that takes a clip over 15s; seedanceFace:true for AI-character faces). Clip-length and resolution windows per engine are on the `model` param. Cost is per second of OUTPUT (≈ clip length, rounded up), and an edit bills input + output seconds on every provider — read the quote from the confirm gate or slates_estimate_generation_cost, never from memory. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; Seedance edit/relocate for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
2895
3250
  input: z.object({
2896
3251
  projectId: z.string().uuid().describe('Project the source clip lives in.'),
2897
3252
  sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
@@ -3436,17 +3791,33 @@ export const setFolderCover = {
3436
3791
  };
3437
3792
  export const updateCharacter = {
3438
3793
  id: 'slates_update_character',
3439
- description: 'Update a character\'s name, description, or style. Use slates_set_character_identity_asset for its canonical image.',
3794
+ description: 'Update a character\'s name, description, style, or voice. Use slates_set_character_identity_asset for its canonical image.',
3440
3795
  input: z.object({
3441
3796
  characterId: z.string().uuid(),
3442
3797
  name: z.string().min(1).max(120).optional(),
3443
3798
  description: z.string().optional(),
3444
3799
  style: z.string().max(200).optional().describe("Art style. Omit to inherit the reference's style (the default). Canonical styles: photoreal, anime, painterly, 3d-render, comic. Or pass any free-text instruction, e.g. 'turn this into a real person'."),
3800
+ // Agent parity for the character card's voice slot: the desktop route has
3801
+ // taken this since 2026-08-28; the op never exposed it, so an agent could
3802
+ // render a voice and had no way to keep it on the character.
3803
+ voiceAssetId: z
3804
+ .string()
3805
+ .uuid()
3806
+ .nullable()
3807
+ .optional()
3808
+ .describe("The AUDIO asset that is this character's voice (what inworld-tts-2 clones for its lines); null detaches, the clip stays."),
3445
3809
  }),
3446
3810
  async run(input, ctx) {
3447
3811
  return ok(await ctx.desktop().post('/agent/characters/update', {
3448
3812
  id: input.characterId,
3449
- data: { name: input.name, description: input.description, style: input.style },
3813
+ data: {
3814
+ name: input.name,
3815
+ description: input.description,
3816
+ style: input.style,
3817
+ // Sent only when given: the route treats presence as intent, and an
3818
+ // explicit null is the detach.
3819
+ ...(input.voiceAssetId !== undefined ? { voiceAssetId: input.voiceAssetId } : {}),
3820
+ },
3450
3821
  }));
3451
3822
  },
3452
3823
  };
@@ -3591,7 +3962,12 @@ export const reorderScenes = {
3591
3962
  };
3592
3963
  export const updateFrame = {
3593
3964
  id: 'slates_update_frame',
3594
- description: 'Update a frame: shot label, notes, bound asset (assetId=null unbinds), scene, position, frameType (first/last/ingredient, null clears), or motion prompt (null clears).',
3965
+ description:
3966
+ // ⚠️ `frameType` and `motionPrompt` are GONE. They were a weaker duplicate
3967
+ // of what the Shot in this slot already encodes — the image's role and the
3968
+ // beat's words — and they were backfilled into Shots on 2026-08-31. Use
3969
+ // slates_update_shot for either.
3970
+ 'Update a slot: its shot label, notes, bound asset (assetId=null unbinds), scene or position. The BEAT — its line, references, model, prompt and framing — lives on the Shot in this slot; use slates_update_shot for that.',
3595
3971
  input: z.object({
3596
3972
  frameId: z.string().uuid(),
3597
3973
  shotLabel: z.string().optional(),
@@ -3599,8 +3975,6 @@ export const updateFrame = {
3599
3975
  assetId: z.string().uuid().nullable().optional(),
3600
3976
  sceneId: z.string().uuid().nullable().optional(),
3601
3977
  position: z.number().int().min(0).optional(),
3602
- frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
3603
- motionPrompt: z.string().nullable().optional(),
3604
3978
  }),
3605
3979
  async run(input, ctx) {
3606
3980
  return ok(await ctx.desktop().post('/agent/frames/update', {
@@ -3611,8 +3985,6 @@ export const updateFrame = {
3611
3985
  assetId: input.assetId,
3612
3986
  sceneId: input.sceneId,
3613
3987
  position: input.position,
3614
- frameType: input.frameType,
3615
- motionPrompt: input.motionPrompt,
3616
3988
  },
3617
3989
  }));
3618
3990
  },
@@ -3632,7 +4004,7 @@ export const updateFrame = {
3632
4004
  */
3633
4005
  export const batchUpdateFrames = {
3634
4006
  id: 'slates_batch_update_frames',
3635
- description: 'Update MANY frames in one call — motion prompts, shot labels, notes, asset binding, scene/position, frameType. Prefer this over repeated slates_update_frame when writing a scene or a whole storyboard: it is one round-trip and one UI refresh. Every id is validated before anything is written, so the batch never lands half-applied.',
4007
+ description: 'Update MANY slots in one call — shot labels, notes, asset binding, scene/position. Prefer this over repeated slates_update_frame when re-arranging a scene: it is one round-trip and one UI refresh, and every id is validated before anything is written, so the batch never lands half-applied. To write the BEATS themselves, use slates_create_shot / slates_update_shot.',
3636
4008
  input: z.object({
3637
4009
  updates: z
3638
4010
  .array(z.object({
@@ -3642,8 +4014,6 @@ export const batchUpdateFrames = {
3642
4014
  assetId: z.string().uuid().nullable().optional(),
3643
4015
  sceneId: z.string().uuid().nullable().optional(),
3644
4016
  position: z.number().int().min(0).optional(),
3645
- frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
3646
- motionPrompt: z.string().nullable().optional(),
3647
4017
  }))
3648
4018
  .min(1),
3649
4019
  }),
@@ -3657,8 +4027,6 @@ export const batchUpdateFrames = {
3657
4027
  assetId: u.assetId,
3658
4028
  sceneId: u.sceneId,
3659
4029
  position: u.position,
3660
- frameType: u.frameType,
3661
- motionPrompt: u.motionPrompt,
3662
4030
  },
3663
4031
  })),
3664
4032
  }));
@@ -3675,6 +4043,779 @@ export const deleteFrame = {
3675
4043
  // ── Prompting guides (local lookup — no transport) ──────────────
3676
4044
  // Model-id → guide-name aliasing. Order matters: kling-mc-* (motion
3677
4045
  // transfer) must match before the generic kling-v3* check.
4046
+ // ── Shots — the prompt bar, serialized ──────────────────────────
4047
+ //
4048
+ // A Shot is a NAMED generation recipe: what to make, with what, on which model,
4049
+ // at what settings. It exists to remove one pipeline constraint from this
4050
+ // surface — `slates_add_frame` requires a non-null `assetId`, so an agent could
4051
+ // not plan a shot before its image existed. A Shot takes no asset at all.
4052
+ //
4053
+ // 🚨 IT STORES A RAW PROMPT AND REFERENCES, NEVER A COMPOSED PROMPT. The
4054
+ // composer is the only thing that numbers anything; a stored "image 3" is a lie
4055
+ // the moment a reference moves. `slates_get_shot` returns the composed prompt so
4056
+ // the agent can audit its own work through the exact resolver the request uses.
4057
+ /** Role → its `string[]` param, GENERATED from the role list. Never hand-typed:
4058
+ * `attachmentRoles`/`shot-spec` is the ONE role list, and a sixth role has to
4059
+ * appear here without anyone remembering to add it. */
4060
+ function shotRefShape(described) {
4061
+ return Object.fromEntries(ORDERED_ATTACHMENT_ROLES.map((role) => [
4062
+ role,
4063
+ described
4064
+ ? z
4065
+ .array(z.string())
4066
+ .optional()
4067
+ .describe(`${ATTACHMENT_ROLE_DESCRIPTION[role]} UUIDs or badge codes ("IMG-A8").`)
4068
+ : z.array(z.string()).optional(),
4069
+ ]));
4070
+ }
4071
+ const SHOT_REFS_LEAD = 'Attachments by ROLE, ordered within each role. The role decides the sentence the model is told, so a subject reference and a plain one are not interchangeable.';
4072
+ const shotRefsSchema = z.object(shotRefShape(true)).optional().describe(SHOT_REFS_LEAD);
4073
+ /**
4074
+ * 🚨 THE SAME SHAPE, DESCRIBED ONCE.
4075
+ *
4076
+ * `params`, `refs` and the script fields are identical across create / update /
4077
+ * duplicate, and `zodToJsonSchema` inlines every description into all three —
4078
+ * 2.7 KB of the same prose, three times, in the desktop's prompt-cached prefix
4079
+ * on every turn. `slates_create_shot` is the op that documents the shape and it
4080
+ * is always in context beside these; repeating the table here bought nothing
4081
+ * but bytes. The Zod ENUMS stay on both, so enforcement is unchanged — only the
4082
+ * prose is deduplicated.
4083
+ */
4084
+ const SEE_CREATE_SHOT = 'Same shape as slates_create_shot — see it for what each field means. ';
4085
+ const shotRefsSchemaTerse = z
4086
+ .object(shotRefShape(false))
4087
+ .optional()
4088
+ .describe(SEE_CREATE_SHOT + SHOT_REFS_LEAD);
4089
+ // The full ratio vocabulary a Shot can hold — image OR video, because a Shot is
4090
+ // any generation. Per-model narrowing happens at create time for video (the same
4091
+ // `assertVideoCapabilities` gate `slates_generate_video` uses) and at generate
4092
+ // time for everything.
4093
+ const SHOT_ASPECT_RATIOS = [...new Set([...VIDEO_ASPECT_RATIOS, ...IMAGE_ASPECT_RATIOS])];
4094
+ // 🚨 THE PER-MODEL CAPABILITY TABLES ARE DELIBERATELY NOT REPEATED HERE.
4095
+ // `slates_generate_video`'s param descriptions already carry them and are always
4096
+ // in context on both surfaces; embedding them again — in three shot ops, each
4097
+ // taking these same params — would put FOUR more copies of a table that grows on
4098
+ // every model addition into the desktop's cached prefix. Measured: it was 15.3%
4099
+ // of the whole tool surface, almost all of it those three strings.
4100
+ // The vocabulary is still ENFORCED (a Zod enum built from MODEL_CAPABILITIES),
4101
+ // and the per-model narrowing is enforced by `assertShotCapabilities` at save
4102
+ // time — which is stronger than prose, not weaker.
4103
+ function shotParamsShape(described) {
4104
+ const d = (node, text) => (described ? node.describe(text) : node);
4105
+ return {
4106
+ aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the model; see slates_generate_video.'),
4107
+ duration: d(z.number().int().min(1).max(360).optional(), 'Seconds for video or duration-based audio; TTS uses text length.'),
4108
+ videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the chosen model when the Shot is saved.'),
4109
+ imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
4110
+ gptQuality: d(z.enum(['medium', 'high']).optional(), 'gpt-image-2 only.'),
4111
+ imageQuantity: d(z.number().int().min(1).max(4).optional(), 'Image models only — how many to make per fire.'),
4112
+ negativePrompt: z.string().optional(),
4113
+ sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
4114
+ seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider at ~45% more."),
4115
+ audioDurationSeconds: d(z.number().int().min(1).max(120).optional(), 'Audio lane. On seed-audio the requested duration IS the bill.'),
4116
+ // The TTS voice — the same three fields slates_generate_audio takes, so a
4117
+ // Shot is the audio call, serialized. Exactly one of them, enforced at fire.
4118
+ voiceId: d(z.string().optional(), `${TTS_MODEL}: preset voiceId (slates_list_voices).`),
4119
+ voiceReferenceAssetId: d(z.string().optional(), `${TTS_MODEL}: audio asset to clone.`),
4120
+ voiceDescription: d(z.string().optional(), `${TTS_MODEL}: the voice in words.`),
4121
+ };
4122
+ }
4123
+ const shotParamsSchema = z.object(shotParamsShape(true)).optional();
4124
+ const shotParamsSchemaTerse = z
4125
+ .object(shotParamsShape(false))
4126
+ .optional()
4127
+ .describe(SEE_CREATE_SHOT + 'Generation settings for the Shot.');
4128
+ /**
4129
+ * The script layer, as op params — GENERATED from `SCRIPT_FIELD_DESCRIPTION`.
4130
+ *
4131
+ * 🚨 THE PROSE IS NEVER HAND-TYPED HERE. `shot-spec.ts` owns what each field
4132
+ * MEANS, with `satisfies Record<ScriptField, string>` making a ninth field a
4133
+ * compile error in the descriptions too. An op that spelled these out would
4134
+ * ship a field with no explanation — the same failure as a column nothing
4135
+ * renders.
4136
+ *
4137
+ * Script fields supply prompt prose when no authored prompt exists (shot-spec.ts).
4138
+ */
4139
+ function shotScriptShape(described) {
4140
+ const text = Object.fromEntries(SCRIPT_TEXT_FIELDS.map((field) => [
4141
+ field,
4142
+ described
4143
+ ? z.string().max(2000).nullable().optional().describe(SCRIPT_FIELD_DESCRIPTION[field])
4144
+ : z.string().max(2000).nullable().optional(),
4145
+ ]));
4146
+ return {
4147
+ ...text,
4148
+ continues: described
4149
+ ? z.boolean().optional().describe(SCRIPT_FIELD_DESCRIPTION.continues)
4150
+ : z.boolean().optional(),
4151
+ };
4152
+ }
4153
+ const shotScriptSchema = shotScriptShape(true);
4154
+ /** Same fields, described on `slates_create_shot` only — see SEE_CREATE_SHOT. */
4155
+ const shotScriptSchemaTerse = shotScriptShape(false);
4156
+ /** The framing vocabulary an agent should know about — generated from the
4157
+ * bucket lists so a seventh bucket cannot ship undescribed, and worded so it
4158
+ * is unmistakably a COUNTING aid rather than a closed set. */
4159
+ const FRAMING_NOTE = `shotSize and camera are FREE TEXT and are never rejected or rewritten. ` +
4160
+ `They are bucketed only for the variety count — shot size into ` +
4161
+ `${SHOT_SIZE_BUCKETS.join(' / ')}, camera into ${CAMERA_MOVE_BUCKETS.join(' / ')} — ` +
4162
+ `and anything unrecognised counts as "other", which is a fine answer.`;
4163
+ /** What the caller actually sent for the script half, by omission. Same rule
4164
+ * as `shotParamsPatch`: an `undefined` value is not an absent key, and a
4165
+ * spread of undefineds is silent data loss on a merging route. */
4166
+ function shotScriptPatch(input) {
4167
+ const out = {};
4168
+ const raw = input;
4169
+ for (const field of SCRIPT_TEXT_FIELDS) {
4170
+ if (raw[field] !== undefined)
4171
+ out[field] = raw[field];
4172
+ }
4173
+ if (input.continues !== undefined)
4174
+ out.continues = input.continues;
4175
+ return out;
4176
+ }
4177
+ /**
4178
+ * The `params` half of a spec patch — ONLY the keys the caller actually named.
4179
+ *
4180
+ * 🚨 AN `undefined` VALUE IS NOT AN ABSENT KEY, AND THE DIFFERENCE IS SILENT
4181
+ * DATA LOSS. `/agent/shots/update` merges `params` one level deep so that
4182
+ * "anything you omit is left exactly as it was" — but a spread copies keys
4183
+ * whose value is `undefined` too, so a hand-built object listing all ten fields
4184
+ * overwrote the nine the caller never mentioned with `undefined`, and the
4185
+ * tolerant reader on the far side then dropped them. `params: { duration: 9 }`
4186
+ * silently cleared the aspect ratio, the resolution, the negative prompt and
4187
+ * the face route. Building the object by omission is what keeps that promise.
4188
+ *
4189
+ * ONE builder, three callers (create / update / duplicate) — the ten field
4190
+ * names were hand-listed twice before this, which is the same near-miss in
4191
+ * waiting.
4192
+ */
4193
+ function shotParamsPatch(p) {
4194
+ const out = {};
4195
+ if (!p)
4196
+ return out;
4197
+ for (const [k, v] of Object.entries(p))
4198
+ if (v !== undefined)
4199
+ out[k] = v;
4200
+ return out;
4201
+ }
4202
+ /**
4203
+ * `audioRefSpokenText` pairs POSITIONALLY with the audio references, so it can
4204
+ * only be sent alongside them.
4205
+ *
4206
+ * Refused rather than truncated or silently re-keyed: the whole point of the
4207
+ * field is that the WORDS are exact, and a misaligned array attaches one clip's
4208
+ * line to another clip with nothing on screen to say so. Same rule
4209
+ * `slates_generate_video` already applies.
4210
+ */
4211
+ function checkSpokenTextAlignment(input) {
4212
+ if (input.audioRefSpokenText === undefined)
4213
+ return null;
4214
+ const clips = input.refs?.['audio-reference'];
4215
+ if (!clips || clips.length !== input.audioRefSpokenText.length) {
4216
+ return ok({
4217
+ requires_clarification: true,
4218
+ missing: ['refs["audio-reference"]'],
4219
+ message: `audioRefSpokenText pairs by position with refs["audio-reference"], so send both together ` +
4220
+ `and at the same length (${input.audioRefSpokenText.length} line(s) vs ${clips?.length ?? 0} clip(s)). ` +
4221
+ `Use "" for a clip with no speech.`,
4222
+ });
4223
+ }
4224
+ return null;
4225
+ }
4226
+ /** Every asset reference a Shot input carries, flat, for one `resolveAssetRefs`
4227
+ * pass. Derived from the role list so a new role resolves badge codes too. */
4228
+ function shotRefInputs(input) {
4229
+ const out = [];
4230
+ for (const role of ORDERED_ATTACHMENT_ROLES) {
4231
+ for (const r of input.refs?.[role] ?? [])
4232
+ out.push({ ref: r, role });
4233
+ }
4234
+ if (input.firstFrameAssetId)
4235
+ out.push({ ref: input.firstFrameAssetId, role: 'first frame' });
4236
+ if (input.lastFrameAssetId)
4237
+ out.push({ ref: input.lastFrameAssetId, role: 'last frame' });
4238
+ if (input.params?.voiceReferenceAssetId)
4239
+ out.push({ ref: input.params.voiceReferenceAssetId, role: 'voice reference' });
4240
+ return out;
4241
+ }
4242
+ /**
4243
+ * Op input → the `ShotSpec` shape the desktop stores.
4244
+ *
4245
+ * Badge codes resolve to ids HERE, at call time, against the project as it
4246
+ * stands — never a mapping remembered from earlier in the conversation. A Shot
4247
+ * that recorded a guessed id would look fine in a listing and compose to
4248
+ * something else entirely.
4249
+ */
4250
+ async function buildShotSpecInput(ctx, projectId, input) {
4251
+ const refInputs = shotRefInputs(input);
4252
+ const resolvedRefs = await resolveAssetRefs(ctx, projectId, refInputs.map((r) => r.ref));
4253
+ const rid = (v) => (v ? (resolvedRefs.get(v)?.id ?? v) : null);
4254
+ const refs = {};
4255
+ for (const role of ORDERED_ATTACHMENT_ROLES) {
4256
+ refs[role] = (input.refs?.[role] ?? []).map((v) => resolvedRefs.get(v)?.id ?? v);
4257
+ }
4258
+ // Positional in, KEYED out — the same re-keying `slates_generate_video` does,
4259
+ // and for the same reason: an index means different clips depending on how the
4260
+ // list was assembled, while an asset id cannot drift.
4261
+ const audioIds = refs['audio-reference'] ?? [];
4262
+ const spoken = {};
4263
+ (input.audioRefSpokenText ?? []).forEach((text, i) => {
4264
+ const id = audioIds[i];
4265
+ if (id && text?.trim())
4266
+ spoken[id] = text.trim();
4267
+ });
4268
+ return {
4269
+ spec: {
4270
+ prompt: input.prompt,
4271
+ model: input.model ?? null,
4272
+ // The prompt is recorded as written FOR this model. Swapping the model
4273
+ // later diverges from it and the desktop card says so — the prompt is
4274
+ // never rewritten (that is prompt enhancement, deleted 2026-08-01).
4275
+ authoredFor: input.model ?? null,
4276
+ params: {
4277
+ ...shotParamsPatch(input.params),
4278
+ ...(input.params?.voiceReferenceAssetId ? { voiceReferenceAssetId: rid(input.params.voiceReferenceAssetId) } : {}),
4279
+ },
4280
+ mentions: {
4281
+ characterIds: input.characterIds ?? [],
4282
+ environmentIds: input.environmentIds ?? [],
4283
+ styleIds: input.styleIds ?? [],
4284
+ },
4285
+ refs,
4286
+ firstFrameAssetId: rid(input.firstFrameAssetId),
4287
+ lastFrameAssetId: rid(input.lastFrameAssetId),
4288
+ audioRefSpokenText: spoken,
4289
+ },
4290
+ refEcho: describeResolvedRefs(refInputs, resolvedRefs),
4291
+ };
4292
+ }
4293
+ /**
4294
+ * The capability gate, applied to a SAVED recipe.
4295
+ *
4296
+ * It enforces exactly what the matching generate op enforces and no more: video
4297
+ * goes through `assertVideoCapabilities` (the MODEL_CAPABILITIES SSOT), image
4298
+ * does not, because `slates_generate_image`'s own per-model ratio check is the
4299
+ * named open follow-up in the capability plan. A Shot that refused what
4300
+ * `slates_generate_image` accepts would be a THIRD opinion about the same
4301
+ * model, which is worse than the gap.
4302
+ */
4303
+ function assertShotCapabilities(model, params) {
4304
+ if (!model || !VIDEO_MODELS.includes(model))
4305
+ return null;
4306
+ const err = assertVideoCapabilities({
4307
+ model,
4308
+ aspectRatio: params?.aspectRatio,
4309
+ videoResolution: params?.videoResolution,
4310
+ duration: params?.duration,
4311
+ });
4312
+ return err ? ok(err) : null;
4313
+ }
4314
+ /**
4315
+ * The registry cost key for a saved Shot, through the SAME builders every quote
4316
+ * in this file uses (`videoCostKey` / `imageCostKey` / `audioCostKey`).
4317
+ *
4318
+ * Returns null when the Shot cannot be priced — no model, or a model this
4319
+ * surface does not carry. The caller REPORTS that rather than quoting zero: a
4320
+ * missing price displayed as free is the failure mode the whole pricing
4321
+ * contract exists to prevent.
4322
+ */
4323
+ function shotCostKey(detail) {
4324
+ const model = detail.model;
4325
+ if (!model)
4326
+ return null;
4327
+ const p = detail.params;
4328
+ // Prefer the CLAMPED values the desktop will actually fire with; a listing row
4329
+ // has none, so it falls back to the raw ones and is announced as a floor.
4330
+ const fires = detail.firesWith;
4331
+ if (AUDIO_MODELS.includes(model)) {
4332
+ if (model === TTS_MODEL) {
4333
+ const text = detail.composedPrompt ?? (detail.rawPrompt.trim() || detail.line?.trim() || '');
4334
+ if (!text || text.length > TTS_MAX_CHARACTERS)
4335
+ return null;
4336
+ return audioCostKey({ model, characters: text.length });
4337
+ }
4338
+ const seconds = fires?.audioDurationSeconds ?? p.audioDurationSeconds;
4339
+ if (!seconds)
4340
+ return null;
4341
+ return audioCostKey({ model: model, durationSeconds: seconds });
4342
+ }
4343
+ if (VIDEO_MODELS.includes(model)) {
4344
+ const duration = fires?.duration ?? p.duration;
4345
+ if (!duration)
4346
+ return null;
4347
+ const billed = (d) => (d > 0 ? Math.ceil(d - 0.05) : 0);
4348
+ return videoCostKey({
4349
+ model: model,
4350
+ duration,
4351
+ videoResolution: fires?.videoResolution ??
4352
+ p.videoResolution ??
4353
+ defaultVideoResolutionFor(model),
4354
+ sound: p.sound,
4355
+ seedanceFace: p.seedanceFace,
4356
+ // `references` is absent on a LISTING row (it does not compose), so both
4357
+ // of these read 0 there. That is why a listing quote is announced as a
4358
+ // floor and `slates_get_shot` is the exact one.
4359
+ referenceImages: (detail.references ?? []).filter((r) => r.kind === 'image').length,
4360
+ videoRefSeconds: (detail.references ?? [])
4361
+ .filter((r) => r.kind === 'video')
4362
+ .reduce((n, r) => n + billed(r.durationSeconds ?? 0), 0),
4363
+ });
4364
+ }
4365
+ if (IMAGE_MODELS.includes(model)) {
4366
+ return imageCostKey(model, (fires?.imageResolution ?? p.imageResolution) ??
4367
+ (model === 'nano-banana-2-lite' ? '1k' : '2k'), p.gptQuality ?? 'medium');
4368
+ }
4369
+ return null;
4370
+ }
4371
+ /** Credits for one Shot, and how many generations it fires.
4372
+ *
4373
+ * `imageQuantity` multiplies IMAGE models only — the same condition the
4374
+ * desktop's `estimateCostFor` applies and the only lane `/agent/shots/*` sends
4375
+ * a `count` for. Multiplying it blindly would quote a video Shot 3× for a
4376
+ * param its request never carries, and the card beside it would say ×1. */
4377
+ function shotQuote(detail, byKey) {
4378
+ const key = shotCostKey(detail);
4379
+ const isImage = !!detail.model && IMAGE_MODELS.includes(detail.model);
4380
+ const quantity = isImage ? (detail.params.imageQuantity ?? 1) || 1 : 1;
4381
+ const per = key != null ? byKey.get(key) : undefined;
4382
+ return { key, credits: (per ?? 0) * quantity, quantity };
4383
+ }
4384
+ export const createShot = {
4385
+ id: 'slates_create_shot',
4386
+ description: 'Write one beat of the piece — a Shot: its script line, its references with their roles, its model and params, and the prompt that fires. It needs NO image to exist, so a whole film can be written, arranged and priced before anything is generated. It lands in the storyboard automatically (the open scene, else the most recent storyboard) — never unfiled. ' +
4387
+ FRAMING_NOTE,
4388
+ input: z.object({
4389
+ projectId: z.string().uuid(),
4390
+ name: z.string().max(120).optional().describe('What to call it. Shown on the card; the prompt supplies one if you omit it.'),
4391
+ prompt: z.string().min(1).max(4000).describe('The RAW prompt, @mentions intact. Never write "image 1" yourself — the composer numbers references, and a hand-written number is wrong the moment one moves.'),
4392
+ model: z.string().optional().describe('Model id — the same ids slates_generate_image / slates_generate_video / slates_generate_audio take, and their descriptions carry the routing. Optional: a Shot can be planned before the model is decided.'),
4393
+ params: shotParamsSchema,
4394
+ refs: shotRefsSchema,
4395
+ firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video (UUID or badge code).'),
4396
+ lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code).'),
4397
+ audioRefSpokenText: z.array(z.string()).optional().describe('Same order and length as refs["audio-reference"]; use "" for a clip with no words. The model RE-TRANSCRIBES a supplied take, so only text decides the words.'),
4398
+ characterIds: z.array(z.string().uuid()).optional().describe('Characters the prompt @mentions — stored as ENTITY ids, so updating the character updates every Shot that names it.'),
4399
+ environmentIds: z.array(z.string().uuid()).optional(),
4400
+ styleIds: z.array(z.string().uuid()).optional(),
4401
+ frameId: z.string().uuid().optional().describe('Put it in this exact frame. Optional — omit it and the Shot files itself into a scene, creating a storyboard named after the project if there is none.'),
4402
+ sceneId: z.string().uuid().optional().describe('File it into this scene. Optional; ignored when frameId is given.'),
4403
+ ...shotScriptSchema,
4404
+ }),
4405
+ async run(input, ctx) {
4406
+ const capErr = assertShotCapabilities(input.model, input.params);
4407
+ if (capErr)
4408
+ return capErr;
4409
+ const alignErr = checkSpokenTextAlignment(input);
4410
+ if (alignErr)
4411
+ return alignErr;
4412
+ const desktop = ctx.desktop();
4413
+ await desktop.requireCapability('shots', 'saved Shots');
4414
+ const { spec, refEcho } = await buildShotSpecInput(ctx, input.projectId, input);
4415
+ const r = await desktop.post('/agent/shots', {
4416
+ projectId: input.projectId,
4417
+ name: input.name,
4418
+ spec: { ...spec, ...shotScriptPatch(input) },
4419
+ frameId: input.frameId ?? null,
4420
+ sceneId: input.sceneId ?? null,
4421
+ });
4422
+ // The CODE is the address the user sees on the row — say it back so the
4423
+ // next call, and the next sentence to the user, can point at it.
4424
+ return ok(r.shot, `${r.shot?.code || 'Shot'} — "${r.shot?.name || 'Untitled'}". ${refEcho}`.trim());
4425
+ },
4426
+ };
4427
+ export const updateShot = {
4428
+ id: 'slates_update_shot',
4429
+ description: 'Change part of a Shot, or attach/detach it from a storyboard frame — anything you omit is left exactly as it was. ' +
4430
+ FRAMING_NOTE,
4431
+ input: z.object({
4432
+ shotId: z.string().describe('The Shot id, or its SHOT-A code as shown on the row.'),
4433
+ projectId: z.string().uuid().describe('The Shot\'s project — badge codes and entity ids resolve against it.'),
4434
+ name: z.string().max(120).optional(),
4435
+ prompt: z.string().max(4000).optional(),
4436
+ model: z.string().optional(),
4437
+ params: shotParamsSchemaTerse,
4438
+ refs: shotRefsSchemaTerse,
4439
+ firstFrameAssetId: z.string().optional(),
4440
+ lastFrameAssetId: z.string().optional(),
4441
+ audioRefSpokenText: z.array(z.string()).optional(),
4442
+ characterIds: z.array(z.string().uuid()).optional(),
4443
+ environmentIds: z.array(z.string().uuid()).optional(),
4444
+ styleIds: z.array(z.string().uuid()).optional(),
4445
+ attachFrameId: z.string().uuid().optional().describe('Attach this Shot to a storyboard frame.'),
4446
+ detachFrameId: z.string().uuid().optional().describe('Detach it from a frame. The Shot itself survives.'),
4447
+ posterAssetId: z.string().nullable().optional().describe('Which reference represents this Shot as a thumbnail. Defaulted automatically (first frame, else the first image reference, else the newest take) — only set it to OVERRIDE, and pass null to go back to the default.'),
4448
+ ...shotScriptSchemaTerse,
4449
+ }),
4450
+ async run(input, ctx) {
4451
+ const capErr = assertShotCapabilities(input.model, input.params);
4452
+ if (capErr)
4453
+ return capErr;
4454
+ const alignErr = checkSpokenTextAlignment(input);
4455
+ if (alignErr)
4456
+ return alignErr;
4457
+ const desktop = ctx.desktop();
4458
+ await desktop.requireCapability('shots', 'saved Shots');
4459
+ const { spec } = await buildShotSpecInput(ctx, input.projectId, input);
4460
+ // Only send the halves the caller actually named; the route merges a PARTIAL
4461
+ // spec over the stored one, so an omitted field is never silently cleared.
4462
+ const patch = {};
4463
+ if (input.prompt !== undefined)
4464
+ patch.prompt = input.prompt;
4465
+ if (input.model !== undefined) {
4466
+ patch.model = input.model;
4467
+ // 🚨 `authoredFor` is NOT re-stamped on a model swap. The whole point of
4468
+ // recording it is that the prompt stays written for the model it was
4469
+ // written for — the grammars genuinely differ — so the card can say so.
4470
+ }
4471
+ if (input.params !== undefined)
4472
+ patch.params = spec.params;
4473
+ // 🚨 ONLY THE ROLES THE CALLER NAMED. `buildShotSpecInput` always returns a
4474
+ // COMPLETE refs record, and the route merges one level deep — so sending all
4475
+ // five would clear every role the caller never mentioned. Clearing a role is
4476
+ // explicit: send `[]`.
4477
+ if (input.refs !== undefined) {
4478
+ const built = spec.refs;
4479
+ const named = {};
4480
+ for (const role of ORDERED_ATTACHMENT_ROLES) {
4481
+ if (input.refs[role] !== undefined)
4482
+ named[role] = built[role];
4483
+ }
4484
+ if (Object.keys(named).length > 0)
4485
+ patch.refs = named;
4486
+ }
4487
+ if (input.firstFrameAssetId !== undefined)
4488
+ patch.firstFrameAssetId = spec.firstFrameAssetId;
4489
+ if (input.lastFrameAssetId !== undefined)
4490
+ patch.lastFrameAssetId = spec.lastFrameAssetId;
4491
+ if (input.audioRefSpokenText !== undefined)
4492
+ patch.audioRefSpokenText = spec.audioRefSpokenText;
4493
+ Object.assign(patch, shotScriptPatch(input));
4494
+ // Same rule for the three mention lists — naming one must not clear the
4495
+ // other two.
4496
+ {
4497
+ const built = spec.mentions;
4498
+ const named = {};
4499
+ if (input.characterIds !== undefined)
4500
+ named.characterIds = built.characterIds;
4501
+ if (input.environmentIds !== undefined)
4502
+ named.environmentIds = built.environmentIds;
4503
+ if (input.styleIds !== undefined)
4504
+ named.styleIds = built.styleIds;
4505
+ if (Object.keys(named).length > 0)
4506
+ patch.mentions = named;
4507
+ }
4508
+ const r = await desktop.post('/agent/shots/update', {
4509
+ id: input.shotId,
4510
+ data: {
4511
+ name: input.name,
4512
+ ...(Object.keys(patch).length > 0 ? { spec: patch } : {}),
4513
+ attachFrameId: input.attachFrameId,
4514
+ detachFrameId: input.detachFrameId,
4515
+ posterAssetId: input.posterAssetId,
4516
+ },
4517
+ });
4518
+ return ok(r.shot);
4519
+ },
4520
+ };
4521
+ export const duplicateShot = {
4522
+ id: 'slates_duplicate_shot',
4523
+ description: 'Fork a saved Shot, changing the prompt, the model or any param on the COPY in the same call — make one, fork it five times, change one thing on each. The original is never touched.',
4524
+ input: z.object({
4525
+ shotId: z.string().describe('The Shot id, or its SHOT-A code.'),
4526
+ name: z.string().max(120).optional().describe('Name for the copy (default: the original plus "copy").'),
4527
+ prompt: z.string().max(4000).optional().describe('Replace the prompt on the copy. Omit to keep the original\'s.'),
4528
+ model: z.string().optional().describe('Point the copy at a different model — the A/B lever. The prompt is NOT rewritten, and the copy records which model it was written for.'),
4529
+ params: shotParamsSchemaTerse,
4530
+ frameId: z.string().uuid().nullable().optional().describe('Attach the copy to this frame. Omit to keep the original\'s frame; pass null to leave it unattached.'),
4531
+ }),
4532
+ async run(input, ctx) {
4533
+ const capErr = assertShotCapabilities(input.model, input.params);
4534
+ if (capErr)
4535
+ return capErr;
4536
+ const desktop = ctx.desktop();
4537
+ await desktop.requireCapability('shots', 'saved Shots');
4538
+ // Only the halves actually named. The route merges two levels deep, so an
4539
+ // omitted param on the copy keeps the original's value rather than clearing it.
4540
+ const spec = {};
4541
+ if (input.prompt !== undefined)
4542
+ spec.prompt = input.prompt;
4543
+ if (input.model !== undefined)
4544
+ spec.model = input.model;
4545
+ if (input.params !== undefined) {
4546
+ const voiceRef = input.params.voiceReferenceAssetId;
4547
+ if (voiceRef && !UUID_RE.test(voiceRef)) {
4548
+ const { shot } = await desktop.get('/agent/shots/get', { id: input.shotId });
4549
+ const built = await buildShotSpecInput(ctx, shot.projectId, { params: input.params });
4550
+ spec.params = built.spec.params;
4551
+ }
4552
+ else {
4553
+ spec.params = shotParamsPatch(input.params);
4554
+ }
4555
+ }
4556
+ const r = await desktop.post('/agent/shots/duplicate', {
4557
+ id: input.shotId,
4558
+ name: input.name,
4559
+ ...(Object.keys(spec).length > 0 ? { spec } : {}),
4560
+ frameId: input.frameId,
4561
+ });
4562
+ return ok(r.shot, `Forked into "${r.shot?.name || 'Untitled'}".`);
4563
+ },
4564
+ };
4565
+ /**
4566
+ * The variety strip, in words. GENERATED from the report — never hand-typed,
4567
+ * and never a judgement: it states what is there and stops.
4568
+ *
4569
+ * 🔑 IT RIDES THE OP RESULT, NOT A SKILL. Measured on the 2026-08-30 eval
4570
+ * harness: a rule inlined into an op description moved compliance from 0/8 to
4571
+ * 30/32, while the same guidance behind `slates_get_prompting_guide` sat at 13%
4572
+ * before and 13% after. Guidance the agent must CHOOSE to fetch does not reach
4573
+ * it — so the counts arrive in the result it is already reading.
4574
+ *
4575
+ * Slates counts; the agent judges. No suggested shot size, no auto-varied
4576
+ * camera, no "we changed this for you".
4577
+ */
4578
+ function describeVarietyReport(v) {
4579
+ if (!v || v.cuts === 0)
4580
+ return '';
4581
+ const parts = [];
4582
+ const topSize = v.shotSizes.find((b) => b.bucket !== 'other');
4583
+ if (topSize && topSize.count >= 2)
4584
+ parts.push(`${topSize.count}/${v.cuts} ${topSize.bucket}`);
4585
+ const topMove = v.cameraMoves.find((b) => b.bucket !== 'other');
4586
+ if (topMove && topMove.count >= 2)
4587
+ parts.push(`${topMove.count} ${topMove.bucket}`);
4588
+ for (const run of v.runs)
4589
+ parts.push(`${run.length} ${run.bucket} in a row`);
4590
+ // 🚨 TWO DENOMINATORS, NAMED. Rhythm is counted in CUTS and money in
4591
+ // GENERATIONS; several cuts routinely live inside one generation.
4592
+ const head = `${v.generations} generation(s) · ${v.cuts} cut(s) · ` +
4593
+ `${v.runtimeSeconds == null ? 'no runtime yet' : `${v.runtimeSeconds}s`}` +
4594
+ (v.cutsWithoutDuration > 0 ? ` (${v.cutsWithoutDuration} cut(s) have no duration)` : '');
4595
+ const variety = parts.length > 0 ? `\nVARIETY: ${parts.join(' · ')}` : '';
4596
+ const words = v.words > 0
4597
+ ? `\nWORDS: ${v.words} written` +
4598
+ (v.wordBudget != null
4599
+ ? `, ~${v.wordBudget} fit the runtime at ${SPEECH_RATE.conversational.wpm} wpm` +
4600
+ ` (measured across ${SPEECH_RATE.conversational.n} real ads)`
4601
+ : '')
4602
+ : '';
4603
+ const overLong = v.overLongLines.length > 0
4604
+ ? `\n⚠ ${v.overLongLines.length} line(s) cannot fit their cut at ANY plausible delivery` +
4605
+ ` (over ${SPEECH_RATE.ceiling.wpm} wpm, the fastest read in the corpus).` +
4606
+ ` Split the line or merge the cut — slates_split_shot / slates_merge_shots.`
4607
+ : '';
4608
+ return `${head}${variety}${words}${overLong}`;
4609
+ }
4610
+ export const listShots = {
4611
+ id: 'slates_list_shots',
4612
+ description: "Read the shot list — every Shot as a compact row IN BOARD ORDER (scene, then position), with the piece's cut count, runtime, credit floor and its variety distribution. Read this before firing a set: if one shot size is the plurality or three cuts in a row share a camera move, the batch is wrong before a credit is spent.",
4613
+ input: z.object({
4614
+ projectId: z.string().uuid(),
4615
+ storyboardId: z.string().uuid().optional().describe('Only Shots attached to a frame in this storyboard.'),
4616
+ frameId: z.string().uuid().optional().describe('Only Shots attached to this frame.'),
4617
+ }),
4618
+ async run(input, ctx) {
4619
+ const desktop = ctx.desktop();
4620
+ await desktop.requireCapability('shots', 'saved Shots');
4621
+ const r = await desktop.get('/agent/shots', {
4622
+ projectId: input.projectId,
4623
+ storyboardId: input.storyboardId,
4624
+ frameId: input.frameId,
4625
+ });
4626
+ const rows = r.shots ?? [];
4627
+ // Deliberately does NOT compose each Shot — that is what slates_get_shot is
4628
+ // for. A listing that composed every row would make browsing cost as much as
4629
+ // auditing.
4630
+ const registry = await ctx.cloud().get('/api/agent/models');
4631
+ const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
4632
+ let total = 0;
4633
+ let unpriced = 0;
4634
+ const shots = rows.map((s) => {
4635
+ const q = shotQuote(s, byKey);
4636
+ if (q.key == null || !byKey.has(q.key))
4637
+ unpriced += 1;
4638
+ total += q.credits;
4639
+ return {
4640
+ id: s.id,
4641
+ code: s.code,
4642
+ name: s.name,
4643
+ scene: s.sceneName,
4644
+ position: s.position,
4645
+ model: s.model,
4646
+ references: s.referenceCount,
4647
+ cuts: s.cuts,
4648
+ runtime_seconds: s.runtimeSeconds,
4649
+ speaker: s.speaker,
4650
+ line: s.line,
4651
+ delivery: s.delivery,
4652
+ action: s.action,
4653
+ prop: s.prop,
4654
+ shot_size: s.shotSize,
4655
+ camera: s.camera,
4656
+ continues: s.continues,
4657
+ credits: q.credits,
4658
+ };
4659
+ });
4660
+ return ok({ shots, total_credits: total, unpriced, variety: r.variety }, `${shots.length} shot(s), at least ${fmtCredits(total)} to fire them all` +
4661
+ (unpriced > 0 ? ` (${unpriced} could not be priced — no model or no duration set).` : '.') +
4662
+ ' 🚨 That is a FLOOR, not the bill: a listing does not compose, so the two dimensions that' +
4663
+ ' depend on the reference set — Seedance reference-clip seconds and MiniMax reference images' +
4664
+ ' past the free five — are missing from it. slates_get_shot prices one exactly, and' +
4665
+ ' slates_generate_from_shots quotes the set exactly before it fires anything.' +
4666
+ (describeVarietyReport(r.variety) ? `
4667
+
4668
+ ${describeVarietyReport(r.variety)}` : ''));
4669
+ },
4670
+ };
4671
+ export const getShot = {
4672
+ id: 'slates_get_shot',
4673
+ description: 'Read one Shot in full — the COMPOSED prompt the request will actually carry, its numbered references, anything it points at that no longer exists, and its exact credit quote. Audit your own work here before firing.',
4674
+ input: z.object({
4675
+ shotId: z.string().describe('The Shot id, or its SHOT-A code as shown on the row.'),
4676
+ }),
4677
+ async run(input, ctx) {
4678
+ const desktop = ctx.desktop();
4679
+ await desktop.requireCapability('shots', 'saved Shots');
4680
+ const r = await desktop.get('/agent/shots/get', { id: input.shotId });
4681
+ const registry = await ctx.cloud().get('/api/agent/models');
4682
+ const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
4683
+ const q = shotQuote(r.shot, byKey);
4684
+ return ok({ ...r.shot, cost_key: q.key, credits: q.credits }, `"${r.shot.name || 'Untitled'}" — ${r.shot.model ?? 'no model set'}, ` +
4685
+ (q.key && !r.shot.blocked
4686
+ ? `${fmtCredits(q.credits)} (${q.key}).`
4687
+ : `CANNOT FIRE YET: ${r.shot.blocked ?? 'not priceable — set a model and a duration.'}`) +
4688
+ (r.shot.blocked ? '' : ` Fires with ${JSON.stringify(r.shot.firesWith)}.`) +
4689
+ `\nCOMPOSED PROMPT (what the model is told): ${r.shot.composedPrompt}`);
4690
+ },
4691
+ };
4692
+ /**
4693
+ * SPLIT and MERGE — the chop decision, and the only two cross-row operations on
4694
+ * this surface.
4695
+ *
4696
+ * 🔑 THIS IS THE EXECUTIVE CALL, AND IT IS WHY THEY ARE OPS. *"Sometimes you
4697
+ * might be using more dialogue in one single 30-second generation. Sometimes it
4698
+ * might just be a 4-second generation of one line."* Merging five lines into
4699
+ * one long take or splitting them into five short ones changes the rhythm AND
4700
+ * the price, and the agent has to be able to re-chop what it wrote.
4701
+ *
4702
+ * 🚨 NEITHER EDITS TEXT AS A SIDE EFFECT. Split moves the text after a caret
4703
+ * the caller placed; merge joins two texts at their boundary. Nothing else in
4704
+ * either row is rewritten. That property is what keeps a finished script
4705
+ * finished, and it is the acceptance test for anything added here later.
4706
+ */
4707
+ export const splitShot = {
4708
+ id: 'slates_split_shot',
4709
+ description: 'Split one Shot into two at a caret in its line. The text after the caret moves to the new Shot, which INHERITS the model, params and references and lands directly after it in the scene. A mid-sentence split marks the second row as continuing the first — one sentence, two cuts, which is the signature voiceover move. Takes stay with the first row. Omit the caret to add a sibling cut with the same visuals and no words moved.',
4710
+ input: z.object({
4711
+ shotId: z.string().describe('The Shot id, or its SHOT-A code.'),
4712
+ caret: z
4713
+ .number()
4714
+ .int()
4715
+ .min(0)
4716
+ .optional()
4717
+ .describe("Character offset into the Shot's `line` to cut at. Omitted = the end of the line."),
4718
+ }),
4719
+ async run(input, ctx) {
4720
+ const desktop = ctx.desktop();
4721
+ await desktop.requireCapability('shots', 'saved Shots');
4722
+ const r = await desktop.post('/agent/shots/split', { id: input.shotId, caret: input.caret });
4723
+ return ok(r, `Split into ${r.first?.code || 'the first Shot'} and ` +
4724
+ `${r.second?.code || 'a new Shot'}. Duration was INHERITED, not divided — ` +
4725
+ `set it on each if the chop changed how long they run.`);
4726
+ },
4727
+ };
4728
+ export const mergeShots = {
4729
+ id: 'slates_merge_shots',
4730
+ description: "Merge two adjacent Shots into one. Texts join, references union, the FIRST Shot's model and params win, and the durations SUM — which may exceed the model's window, in which case it is shown and never blocked. Lossy in one direction: the second Shot's model and params are discarded, and there is no undo (the takes are the history).",
4731
+ input: z.object({
4732
+ firstId: z.string().describe('The Shot that survives — its model and params win. Id or SHOT-A code.'),
4733
+ secondId: z.string().describe('The Shot folded into it. Id or SHOT-A code.'),
4734
+ }),
4735
+ async run(input, ctx) {
4736
+ const desktop = ctx.desktop();
4737
+ await desktop.requireCapability('shots', 'saved Shots');
4738
+ const r = await desktop.post('/agent/shots/merge', {
4739
+ firstId: input.firstId,
4740
+ secondId: input.secondId,
4741
+ });
4742
+ return ok(r.shot, `Merged into ${r.shot?.code || 'one Shot'} — ` +
4743
+ `${r.shot?.cuts ?? 1} cut(s), ${r.shot?.runtimeSeconds ?? 'no'} second(s).`);
4744
+ },
4745
+ };
4746
+ export const generateFromShots = {
4747
+ id: 'slates_generate_from_shots',
4748
+ billable: true,
4749
+ description: 'Generate from saved Shots, ONE AFTER ANOTHER, with a single quote and a single approval for the whole set. It blocks until the last one lands, so a set of video Shots can outlast the HTTP timeout while the run keeps going — if that happens, poll slates_get_shot for each Shot\'s generationIds instead of re-firing, which double-spends.',
4750
+ input: z.object({
4751
+ shotIds: z.array(z.string()).min(1).max(20).describe('The Shots to fire, in order — ids or SHOT-A codes.'),
4752
+ confirm: z.boolean().optional().describe('Set true after explicit user OK on the TOTAL below.'),
4753
+ }),
4754
+ async run(input, ctx) {
4755
+ const desktop = ctx.desktop();
4756
+ await desktop.requireCapability('shots', 'saved Shots');
4757
+ // Resolve and price every Shot BEFORE anything fires. A dead id found
4758
+ // halfway through a batch means a partially-fired, partially-BILLED run.
4759
+ const details = [];
4760
+ for (const id of input.shotIds) {
4761
+ const r = await desktop.get('/agent/shots/get', { id });
4762
+ details.push(r.shot);
4763
+ }
4764
+ const registry = await ctx.cloud().get('/api/agent/models');
4765
+ const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
4766
+ const quotes = details.map((d) => ({ detail: d, ...shotQuote(d, byKey) }));
4767
+ const total = quotes.reduce((n, q) => n + q.credits, 0);
4768
+ const largest = quotes.reduce((m, q) => (q.credits > m ? q.credits : m), 0);
4769
+ const unpriced = quotes.filter((q) => q.key == null || !byKey.has(q.key));
4770
+ const blockedShots = details.filter((d) => d.blocked);
4771
+ if (!input.confirm) {
4772
+ // ONE approval for the set, itemised. N approvals would re-introduce the
4773
+ // friction the batch exists to remove; the safety is the STATED TOTAL,
4774
+ // prominent — count, total, and the largest single Shot.
4775
+ const lines = quotes.map((q) => ` - ${q.detail.name || 'Untitled'} · ${q.detail.model ?? 'no model'} · ` +
4776
+ (q.key ? fmtCredits(q.credits) : 'NOT PRICEABLE'));
4777
+ const blockedLines = blockedShots.map((d) => ` ✖ ${d.name || 'Untitled'} WILL NOT FIRE: ${d.blocked}`);
4778
+ const warnings = details
4779
+ .filter((d) => d.missing || d.unresolvedTokens?.length)
4780
+ .map((d) => ` ! ${d.name || 'Untitled'}: ${d.missing ? 'references something that no longer exists' : ''}` +
4781
+ `${d.unresolvedTokens?.length ? ` ${d.unresolvedTokens.join(', ')} match nothing saved (sent as written, no reference attached)` : ''}`);
4782
+ return ok({
4783
+ requires_confirm: true,
4784
+ count: quotes.length,
4785
+ total_credits: total,
4786
+ largest_single_credits: largest,
4787
+ blocked_count: blockedShots.length,
4788
+ shots: quotes.map((q) => ({
4789
+ id: q.detail.id,
4790
+ name: q.detail.name,
4791
+ model: q.detail.model,
4792
+ cost_key: q.key,
4793
+ credits: q.credits,
4794
+ blocked: q.detail.blocked,
4795
+ })),
4796
+ }, `Firing ${quotes.length} Shot(s) SEQUENTIALLY.\n` +
4797
+ `TOTAL ${fmtCredits(total)} · largest single ${fmtCredits(largest)}\n` +
4798
+ lines.join('\n') +
4799
+ (blockedLines.length > 0 ? `\n${blockedLines.join('\n')}` : '') +
4800
+ (unpriced.length > 0
4801
+ ? `\n ! ${unpriced.length} shot(s) could not be priced — they will still be attempted and may fail.`
4802
+ : '') +
4803
+ (warnings.length > 0 ? `\n${warnings.join('\n')}` : '') +
4804
+ `\n\nRe-call with confirm: true after explicit user OK on that total.`);
4805
+ }
4806
+ const r = await desktop.post('/agent/shots/batch-generate', { shotIds: input.shotIds });
4807
+ const failedLines = (r.results ?? [])
4808
+ .filter((x) => x.status === 'failed')
4809
+ .map((x) => ` ✗ ${x.name || 'Untitled'}: ${x.error ?? 'failed'}`);
4810
+ return ok(r, `${r.succeeded} of ${r.total} generated for about ${fmtCredits(total)}.` +
4811
+ (failedLines.length > 0
4812
+ ? // Reported, never retried: an agent that treats a failed render as
4813
+ // something to try again spends credits before anyone notices.
4814
+ `\n${failedLines.join('\n')}\nThese were NOT retried. Read each error, fix the Shot, and re-fire only what you meant to.`
4815
+ : '') +
4816
+ ` ${BACKGROUND_REVIEW_POINTER}`);
4817
+ },
4818
+ };
3678
4819
  function resolveGuideTopic(topic) {
3679
4820
  const t = topic.trim().toLowerCase();
3680
4821
  if (SKILLS[t])
@@ -3750,17 +4891,29 @@ function resolveGuideTopic(topic) {
3750
4891
  return 'slates-prompting-kling-v3';
3751
4892
  if (t.startsWith('kling-v3'))
3752
4893
  return 'slates-prompting-kling-v3';
3753
- // Audio — seed-audio BEFORE the seedance check: "seed-audio" also starts
3754
- // with "seed", and falling through would hand the video guide to the audio
3755
- // model (the exact class of aliasing bug this comment block warns about).
3756
- // Speech, dialogue and scratch VO all live on Seed Audio now — the TTS
3757
- // surface is gone, so "tts"/"voiceover" must NOT land on the ElevenLabs
3758
- // guide, which is SFX-only.
4894
+ // Audio — the TTS seat FIRST, then seed-audio, then eleven-sfx.
4895
+ //
4896
+ // 🚨 "tts"/"voiceover" ROUTE HERE AGAIN (2026-09-05). They used to land on
4897
+ // seed-audio because there was no TTS surface at all; there is one now, so
4898
+ // leaving them there would hand a scene-renderer's guide to someone asking
4899
+ // about speech. They must still never reach the ElevenLabs guide, which is
4900
+ // SFX-only.
4901
+ if (t.startsWith('inworld') ||
4902
+ t === 'tts' ||
4903
+ t === 'text-to-speech' ||
4904
+ t === 'text to speech' ||
4905
+ t === 'voiceover' ||
4906
+ t === 'voice') {
4907
+ return 'slates-prompting-inworld-tts';
4908
+ }
4909
+ // seed-audio BEFORE the seedance check: "seed-audio" also starts with "seed",
4910
+ // and falling through would hand the video guide to the audio model (the
4911
+ // exact class of aliasing bug this comment block warns about). `dialogue`
4912
+ // stays here: dialogue inside a SCENE is what seed-audio is for, while a
4913
+ // single voice saying a single line is the TTS seat above.
3759
4914
  if (t.startsWith('seed-audio') ||
3760
4915
  t === 'seed audio' ||
3761
4916
  t === 'audio' ||
3762
- t === 'tts' ||
3763
- t === 'voiceover' ||
3764
4917
  t === 'dialogue') {
3765
4918
  return 'slates-prompting-seed-audio';
3766
4919
  }
@@ -3819,25 +4972,72 @@ function describeGuideTopics() {
3819
4972
  }
3820
4973
  export const getPromptingGuide = {
3821
4974
  id: 'slates_get_prompting_guide',
3822
- description: "Return the full markdown of a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id (e.g. 'veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2') which maps to the right guide. ALWAYS read 'slates-cost-discipline' plus the relevant model guide before your first generation in a session.",
4975
+ description: 'For app help and exact UI instructions use topic "app-manual" with a query such as "voice recording". This returns the canonical product manual, shared by every agent surface. ' +
4976
+ // 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
4977
+ // MEASURED at 13% compliance before and after the enforcement work — pointer
4978
+ // prose is the shape that does not move the agent. What replaced it is
4979
+ // structural: the never-use list rides the generate ops' descriptions and
4980
+ // the craft card rides the estimate result, so the facts arrive whether or
4981
+ // not this op is ever called.
4982
+ "Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
3823
4983
  input: z.object({
4984
+ query: z.string().max(200).optional().describe('For app-manual: keywords to retrieve relevant UI sections. Omit for the entire manual.'),
3824
4985
  topic: z
3825
4986
  .string()
3826
4987
  .min(1)
3827
4988
  .describe(`Guide name, model id, or style name. ${describeGuideTopics()}`),
4989
+ depth: z.enum(['card', 'full']).optional().describe('"card" returns just the levers block (a few hundred words — the same card slates_estimate_generation_cost already attached, so usually redundant). "full" (default) returns the whole guide, up to several thousand words.'),
3828
4990
  }),
3829
4991
  async run(input) {
4992
+ if (input.topic.trim().toLowerCase() === 'app-manual') {
4993
+ const content = appManualSections(input.query);
4994
+ return { text: content, data: { topic: 'app-manual', bytes: Buffer.byteLength(content, 'utf8') } };
4995
+ }
3830
4996
  const resolved = resolveGuideTopic(input.topic);
3831
4997
  const content = resolved ? SKILLS[resolved] : undefined;
3832
4998
  if (!resolved || content === undefined) {
3833
4999
  throw new Error(`Unknown guide topic: ${input.topic}. Valid topics: ${Object.keys(SKILLS).sort().join(', ')}`);
3834
5000
  }
5001
+ if (input.depth === 'card') {
5002
+ const card = describeCraftCard(resolved);
5003
+ if (card) {
5004
+ return { text: card, data: { topic: resolved, depth: 'card', bytes: Buffer.byteLength(card, 'utf8') } };
5005
+ }
5006
+ // No card on this guide — returning nothing would read as "no guidance",
5007
+ // which is worse than a fall-through the result names.
5008
+ }
3835
5009
  return {
3836
5010
  text: content,
3837
- data: { topic: resolved, bytes: Buffer.byteLength(content, 'utf8') },
5011
+ data: { topic: resolved, depth: 'full', bytes: Buffer.byteLength(content, 'utf8') },
3838
5012
  };
3839
5013
  },
3840
5014
  };
5015
+ /**
5016
+ * The one op that changes what OTHER ops are visible.
5017
+ *
5018
+ * 🚨 IT EXISTS BECAUSE THE SURFACE IS 112 KB AND EVERY TURN PAYS FOR ALL OF IT.
5019
+ * The desktop Studio Agent sends `core` plus this; a group arrives when the
5020
+ * work needs it and stays for the rest of the run. On the MCP surface every op
5021
+ * is registered up front (a stdio server has no run to append to), so this
5022
+ * returns the same definitions as a plain listing — useful either way, since
5023
+ * it is also how an agent asks "what else can you do".
5024
+ */
5025
+ export const loadTools = {
5026
+ id: 'slates_load_tools',
5027
+ description: 'Load a deferred group of tools for the rest of this session. The core surface is always present; these four groups are held back so every turn does not pay for the whole registry. ' +
5028
+ Object.entries(GROUP_SUMMARY)
5029
+ .map(([g, s]) => `"${g}": ${s}`)
5030
+ .join('. ') +
5031
+ '. Call it the moment the work needs one of those — the tools arrive in the same turn\'s result and stay loaded. On MCP clients every tool is already registered and this just lists the group.',
5032
+ input: z.object({
5033
+ group: z.enum(['library', 'timeline', 'admin', 'blender']).describe('Which group to load.'),
5034
+ }),
5035
+ async run(input) {
5036
+ const defs = toolDefinitions(ALL_OPERATIONS.filter((op) => groupFor(op.id) === input.group), { surface: 'mcp' });
5037
+ return ok({ group: input.group, tools: defs }, `Loaded the "${input.group}" group — ${defs.length} tool(s) now available:\n` +
5038
+ defs.map((d) => `${d.name}: ${d.description.split(/(?<=\.)\s/)[0]}`).join('\n'));
5039
+ },
5040
+ };
3841
5041
  // ── Blender previs ──────────────────────────────────────────────
3842
5042
  //
3843
5043
  // The only ops that talk to a third transport: a localhost socket into a
@@ -4027,6 +5227,7 @@ export const ALL_OPERATIONS = [
4027
5227
  generateImage,
4028
5228
  generateVideo,
4029
5229
  generateAudio,
5230
+ listVoices,
4030
5231
  generateLipSync,
4031
5232
  generateMotionTransfer,
4032
5233
  editVideo,
@@ -4068,14 +5269,31 @@ export const ALL_OPERATIONS = [
4068
5269
  updateFrame,
4069
5270
  batchUpdateFrames,
4070
5271
  deleteFrame,
5272
+ // ── Shots: the prompt bar, serialized ────────────────────────────────
5273
+ // Beside the storyboard ops because that is the neighbourhood they belong to
5274
+ // — structure, not spend. The one that spends sits last in the group.
5275
+ createShot,
5276
+ updateShot,
5277
+ duplicateShot,
5278
+ // Split and merge are the CHOP decision — where the rhythm and the price are
5279
+ // actually decided. They sit beside the writers, not with the spender.
5280
+ splitShot,
5281
+ mergeShots,
5282
+ listShots,
5283
+ getShot,
5284
+ generateFromShots,
4071
5285
  getPromptingGuide,
5286
+ loadTools,
4072
5287
  // ── Blender previs, LAST and deliberately ────────────────────────────
4073
- // This order is not cosmetic: `slate/src/main/studio-agent/ops.ts` maps this
4074
- // array straight into the Anthropic `tools` array, and that block sits inside
4075
- // the desktop Studio Agent's PROMPT-CACHED PREFIX. These six landed at the
4076
- // TOP, which put a third transport nobody without Blender can reach ahead of
4077
- // `slates_get_workspace_state` in every conversation the app has. They are a
4078
- // niche lane off the end of the surface, and the list should read that way.
5288
+ // These six landed at the TOP once, which put a third transport nobody
5289
+ // without Blender can reach ahead of `slates_get_workspace_state` in every
5290
+ // conversation the app has. They are a niche lane off the end of the surface,
5291
+ // and the list should read that way.
5292
+ //
5293
+ // ⚠️ POSITION NO LONGER CONTROLS COST. They are the `blender` tier group (see
5294
+ // surface.ts), so the desktop does not send them at all until
5295
+ // `slates_load_tools` asks for them. Order here is now the READING order on
5296
+ // the MCP surface, which sends everything — still last, same reason.
4079
5297
  blenderStatus,
4080
5298
  blenderExecute,
4081
5299
  blenderScene,
@@ -4083,4 +5301,32 @@ export const ALL_OPERATIONS = [
4083
5301
  blenderSearchDocs,
4084
5302
  blenderRenderBlocking,
4085
5303
  ];
5304
+ // ── Surface metadata, stamped once at load ──────────────────────
5305
+ //
5306
+ // 🚨 DERIVED, NEVER HAND-SET PER OP. Every op gets all four MCP annotations and
5307
+ // its tier here, from the rules in `surface.ts`, so a new op cannot ship
5308
+ // un-annotated (which a host reads as "not destructive") or accidentally
5309
+ // deferred. The lockstep check re-derives the hints INDEPENDENTLY from each
5310
+ // op's own transport verbs, so a read-only claim on an op that posts is caught
5311
+ // rather than trusted.
5312
+ for (const op of ALL_OPERATIONS) {
5313
+ const mutable = op;
5314
+ mutable.annotations = annotate(mutable.id, mutable.billable);
5315
+ mutable.tier = tierFor(mutable.id);
5316
+ mutable.group = groupFor(mutable.id);
5317
+ }
5318
+ // A group naming an op that does not exist would silently defer nothing, and
5319
+ // the tool it meant to hold back would keep costing prefix bytes forever.
5320
+ {
5321
+ const ids = new Set(ALL_OPERATIONS.map((o) => o.id));
5322
+ for (const [group, members] of Object.entries(OPERATION_GROUPS)) {
5323
+ for (const id of members) {
5324
+ if (!ids.has(id)) {
5325
+ throw new Error(`[operations] OPERATION_GROUPS.${group} names "${id}", which is not in ALL_OPERATIONS. ` +
5326
+ `Fix the id or drop the entry — a phantom member defers nothing.`);
5327
+ }
5328
+ }
5329
+ }
5330
+ }
5331
+ export { toolDefinitions, toolDefinition, groupFor, tierFor, OPERATION_GROUPS, GROUP_SUMMARY } from './surface.js';
4086
5332
  //# sourceMappingURL=index.js.map