@slatesvideo/shared 0.6.1 → 0.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/dist/clients/blender.d.ts +50 -0
  2. package/dist/clients/blender.js +195 -0
  3. package/dist/index.d.ts +3 -1
  4. package/dist/index.js +26 -0
  5. package/dist/operations/index.d.ts +25 -1
  6. package/dist/operations/index.js +454 -24
  7. package/dist/prompts/agent-doctrine.d.ts +36 -0
  8. package/dist/prompts/agent-doctrine.js +194 -0
  9. package/dist/prompts/banned-tokens.d.ts +28 -0
  10. package/dist/prompts/banned-tokens.js +152 -0
  11. package/dist/prompts/model-capabilities.d.ts +13 -1
  12. package/dist/prompts/model-capabilities.js +97 -2
  13. package/dist/prompts/model-facts.d.ts +20 -0
  14. package/dist/prompts/model-facts.js +87 -23
  15. package/dist/prompts/prompting-tips.d.ts +1 -1
  16. package/dist/prompts/prompting-tips.js +65 -0
  17. package/dist/prompts/reference-composer.d.ts +57 -0
  18. package/dist/prompts/reference-composer.js +70 -1
  19. package/dist/skills/content.js +8 -2
  20. package/exports/slates-prompt-builder/generated/SKILL.md +3 -3
  21. package/exports/slates-prompt-builder/generated/reference-seedance.md +2 -1
  22. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +8 -8
  23. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  24. package/package.json +1 -1
  25. package/skills/slates-blocking-to-prompt.md +250 -0
  26. package/skills/slates-camera-language.md +196 -0
  27. package/skills/slates-dialogue-blocking.md +134 -0
  28. package/skills/slates-previs-blocking.md +153 -0
  29. package/skills/slates-prompting-ltx-2-5.md +180 -0
  30. package/skills/slates-prompting-nano-banana-2.md +10 -0
  31. package/skills/slates-prompting-seedance.md +10 -1
  32. package/skills/slates-restyle-from-blocking.md +121 -0
@@ -12,10 +12,16 @@
12
12
  import { z } from 'zod';
13
13
  import { SlatesCloudClient } from '../clients/cloud.js';
14
14
  import { SlatesDesktopClient } from '../clients/desktop.js';
15
+ import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
15
16
  import { SKILLS } from '../skills/content.js';
16
17
  // Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
17
18
  // "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
18
- import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } from '../prompts/model-facts.js';
19
+ import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
20
+ // 🚨 THE routing renderer. Routing prose is GENERATED here, never typed:
21
+ // slates-mcp/CLAUDE.md forbids restating it in an op description, and this
22
+ // file did it anyway for 1,282 characters that repeated MODEL_FACTS phrase
23
+ // for phrase. Edit model-facts.ts; both surfaces follow.
24
+ describeRouting, } from '../prompts/model-facts.js';
19
25
  // 🚨 MODEL CAPABILITY SSOT. Aspect ratios, video resolutions and durations are
20
26
  // VALIDATED against and DESCRIBED from `MODEL_CAPABILITIES` — this file must
21
27
  // never re-state one of those constraints as a literal enum or a sentence
@@ -26,6 +32,13 @@ import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } fr
26
32
  // 1080p. A customer burned a round trip on `4:5` + Seedance: accepted here,
27
33
  // queued, credits reserved, rejected by the provider asynchronously.
28
34
  import { AGENT_ROUTE_PROVIDER, aspectRatioUnion, videoResolutionUnion, durationBounds, aspectRatiosFor, getModelCapability, defaultVideoResolutionFor, checkAspectRatio, checkVideoResolution, checkDuration, describeAspectRatios, describeVideoResolutions, describeDurations, describeReferenceImageCaps, } from '../prompts/model-capabilities.js';
35
+ // 🚨 "LOAD THE GUIDE" MADE STRUCTURAL. The never-use token lists are EXTRACTED
36
+ // from the skill files (between `@banned` markers) and inlined into the two
37
+ // generate ops' descriptions, which are always in context on both surfaces —
38
+ // no call to skip, no discretion. `bannedTokenWarning` then reports what the
39
+ // submitted prompt actually contained, in the result, without blocking it.
40
+ // Never hand-type one of these tokens here; edit the skill.
41
+ import { describeBannedTokens, bannedTokenWarning } from '../prompts/banned-tokens.js';
29
42
  export function defaultContext() {
30
43
  return {
31
44
  cloud: () => new SlatesCloudClient(),
@@ -79,6 +92,30 @@ function creditsFromDollars(dollars) {
79
92
  // Shared describe-text for the background flag on every generate_* op.
80
93
  const BACKGROUND_DESCRIBE = 'Submit and return immediately with generationId(s) instead of blocking until the file is saved. ' +
81
94
  'Poll with slates_get_generation_status. Recommended for video (1-5 min renders).';
95
+ // ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
96
+ //
97
+ // QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
98
+ // pre-condition can. What it CAN have is a result the agent cannot avoid
99
+ // reading, naming the exact op. Deliberately NOT an auto-fetch: that would
100
+ // spend vision tokens on every generation whether review was wanted or not,
101
+ // and the sandbox doctrine says the tool is available, not mandatory.
102
+ //
103
+ // ⚠️ These three differ because what the agent already HAS differs, and telling
104
+ // it to re-fetch something already in front of it burns a turn for nothing:
105
+ // a blocking image generation returns the pixels inline, a video generation
106
+ // returns none, and a background submission has no asset yet.
107
+ const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it against the brief before you describe it. ' +
108
+ 'Any claim about how it LOOKS must come from pixels you actually received (slates-vision-feedback-loop).';
109
+ const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
110
+ 'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
111
+ const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
112
+ 'slates_get_asset_video_frames for video.';
113
+ // The image saved, but reading it back off disk failed (best-effort fetch). The
114
+ // agent has an asset and NO pixels, which is the one state where a quality
115
+ // claim would be pure invention — so this branch has to say so rather than
116
+ // fall through to no pointer at all.
117
+ const IMAGE_FETCH_POINTER = 'The pixels could not be attached to this result: call slates_get_asset_image on the asset id ' +
118
+ 'above before describing how it looks.';
82
119
  // Early-return shape when a generation route accepted the job in background
83
120
  // mode ({ background: true } in the response). No inline-image fetch — the
84
121
  // asset doesn't exist yet; the poller delivers it on completion.
@@ -87,7 +124,8 @@ function backgroundSubmitted(kind, ids, extra, note) {
87
124
  return {
88
125
  text: `Submitted ${kind} in the background — generationId(s): ${idText}. ` +
89
126
  `Call slates_get_generation_status with waitSeconds: 45 (it long-polls and returns on completion — ` +
90
- `never a rapid loop; video renders commonly take 1-5 minutes). Generations survive app restarts.` +
127
+ `never a rapid loop; video renders commonly take 1-5 minutes). Generations survive app restarts. ` +
128
+ BACKGROUND_REVIEW_POINTER +
91
129
  (note ? ` ${note}` : ''),
92
130
  data: { generationIds: ids, status: 'processing', ...extra },
93
131
  };
@@ -190,6 +228,21 @@ export const VIDEO_MODELS = [
190
228
  // the Max row at base rates and offers it 2K/4K it cannot render.
191
229
  'minimax-h3',
192
230
  'minimax-h3-max',
231
+ // LTX-2.5, two seats in one family (2026-08-29). Base is the VOLUME seat —
232
+ // cheapest native 1080p second we sell, free native audio at every tier, the
233
+ // only row reaching 1440p, and the longest clips in the catalogue (20s).
234
+ // Pro is the fidelity seat and is NOT a superset: shorter ladder (no
235
+ // 1440p/4K), shorter clips (6/8/10 only) and ~1/3 dearer.
236
+ //
237
+ // NEVER PREFIX-MATCH: 'ltx-2-5-pro' starts with 'ltx-2-5'. A prefix test
238
+ // bills Pro at base rates AND offers it 1440p/4K and 12-20s durations it
239
+ // cannot render — the same trap as the MiniMax pair, one row worse.
240
+ //
241
+ // Durations are DISCRETE AND EVEN (6,8,10,12,14,16,18,20); the union bounds
242
+ // below stay 3-30 because other rows are wider, so `assertVideoCapabilities`
243
+ // is what refuses an odd second. It reads `values`, not just min/max.
244
+ 'ltx-2-5',
245
+ 'ltx-2-5-pro',
193
246
  ];
194
247
  // ── Capability-derived param vocabulary + guard ─────────────────
195
248
  //
@@ -230,6 +283,17 @@ const VIDEO_RESOLUTION_VOCAB = VIDEO_RESOLUTIONS;
230
283
  const MINIMAX_MODELS = new Set(['minimax-h3', 'minimax-h3-max']);
231
284
  /** Reference images fal does not charge for. */
232
285
  const MINIMAX_FREE_REF_IMAGES = 5;
286
+ /**
287
+ * The LTX-2.5 pair. A SET, not a prefix test — `ltx-2-5-pro` starts with
288
+ * `ltx-2-5`, and the two rows differ on ladder, duration list AND price.
289
+ *
290
+ * Their cost key is the plainest shape in the file — `{model}-{res}-{N}s` with
291
+ * no suffix ever, because LTX has no paid option: native audio is included at
292
+ * every tier (so no `-audio` variant like Kling) and there is no reference
293
+ * endpoint at all (so no `-ref{K}` variant like H3). Mirrors `ltxCreditKey()`
294
+ * in slate/src/shared/pricing.ts.
295
+ */
296
+ const LTX_MODELS = new Set(['ltx-2-5', 'ltx-2-5-pro']);
233
297
  /** K for the `-ref{K}` suffix: images past the free five, capped by the model's
234
298
  * own declared ceiling. 0 for h3-max (no reference transport) and for anything
235
299
  * that is not a MiniMax row. Mirrors refImageSurchargeCount() in
@@ -1076,7 +1140,17 @@ function imageCostKey(model, resolution, quality = 'medium') {
1076
1140
  }
1077
1141
  export const generateImage = {
1078
1142
  id: 'slates_generate_image',
1079
- description: 'Generate an image via Slates credits. Models: nano-banana-2 (default), nano-banana-2-lite (fast/cheap drafts, 1K only), nano-banana-pro (hero-frame/typography premium), gpt-image-2 (sharp text / character sheets / grids; quality medium|high), flux-2-max (photoreal, less censored), seedream-5-lite (cheapest flat, less censored). Which model for which job: read the slates-model-selection skill. Pass projectId to save into a Slates project (recommended — asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use.',
1143
+ description: 'Generate an image via Slates credits.\n' +
1144
+ // GENERATED from MODEL_FACTS — the hand-typed model list that stood here
1145
+ // was a third copy of the routing doctrine, and it had already gone stale
1146
+ // (it still described nano-banana-2-lite by a capability the param owns).
1147
+ `${describeRouting('image')}\n` +
1148
+ 'Full table: the slates-model-selection skill. ' +
1149
+ 'Pass projectId to save into a Slates project (recommended — asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use. ' +
1150
+ // GENERATED from the skill file's own never-use list -- the one piece of
1151
+ // prompting doctrine that is ALWAYS in context, because the agent has
1152
+ // demonstrably skipped the call that would have taught it.
1153
+ describeBannedTokens('image'),
1080
1154
  input: z.object({
1081
1155
  prompt: z.string().min(1).max(4000),
1082
1156
  model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
@@ -1095,6 +1169,10 @@ export const generateImage = {
1095
1169
  // Mirrors the cost confirm gate — defaults silently wasted credits
1096
1170
  // (1:1 when user wanted 16:9, 4k when 1k would have done). The LLM
1097
1171
  // is forced to ask the user or read the skill instead of guessing.
1172
+ // Non-blocking prompt hygiene. Computed once, reported on every exit path
1173
+ // that echoes a prompt -- the clarification and confirm gates are PRE-spend,
1174
+ // which is where a rewrite is still free.
1175
+ const promptWarning = bannedTokenWarning(input.prompt, 'image');
1098
1176
  if (!input.aspectRatio || !input.resolution) {
1099
1177
  const missing = [];
1100
1178
  if (!input.aspectRatio)
@@ -1104,7 +1182,9 @@ export const generateImage = {
1104
1182
  return ok({
1105
1183
  requires_clarification: true,
1106
1184
  missing,
1107
- message: `Missing required field(s): ${missing.join(', ')}. ` +
1185
+ ...(promptWarning ? { prompt_warning: promptWarning } : {}),
1186
+ message: (promptWarning ? `${promptWarning}\n\n` : '') +
1187
+ `Missing required field(s): ${missing.join(', ')}. ` +
1108
1188
  `Read the slates-prompting-nano-banana-2 + slates-cost-discipline skills, ` +
1109
1189
  // Generated from MODEL_CAPABILITIES — never retype a ratio list.
1110
1190
  `or ask the user. Aspect ratio by model: ${describeAspectRatios(IMAGE_MODELS)}. ` +
@@ -1188,7 +1268,9 @@ export const generateImage = {
1188
1268
  model: costKey,
1189
1269
  estimated_cents: totalCents,
1190
1270
  estimated_credits: totalCents,
1191
- message: `Cost exceeds ${CONFIRM_CREDITS} credits. Re-call with confirm=true to proceed, or pick a smaller resolution / count.`,
1271
+ ...(promptWarning ? { prompt_warning: promptWarning } : {}),
1272
+ message: (promptWarning ? `${promptWarning}\n\n` : '') +
1273
+ `Cost exceeds ${CONFIRM_CREDITS} credits. Re-call with confirm=true to proceed, or pick a smaller resolution / count.`,
1192
1274
  });
1193
1275
  }
1194
1276
  const previews = await previewAssets(ctx, referenceAssetIds.map((id) => ({ id, type: 'image', role: 'reference' })));
@@ -1200,7 +1282,8 @@ export const generateImage = {
1200
1282
  `Review them against your prompt — every reference's role must be labeled in the prompt text. ` +
1201
1283
  `If the references suggest a different composition / style than the current prompt captures, REVISE the prompt before confirming. ` +
1202
1284
  `When you talk to the user about this gen, refer to each reference by its code (e.g. "${previews[0]?.ref ?? 'IMG-A?'}") — they'll see the matching badge in the Slates gallery.` +
1203
- `\n\nWhen ready, re-call slates_generate_image with confirm=true and the (possibly revised) prompt.`,
1285
+ `\n\nWhen ready, re-call slates_generate_image with confirm=true and the (possibly revised) prompt.` +
1286
+ (promptWarning ? `\n\n${promptWarning}` : ''),
1204
1287
  images: previews.flatMap((p) => p.images),
1205
1288
  data: {
1206
1289
  requires_confirm: true,
@@ -1275,15 +1358,21 @@ export const generateImage = {
1275
1358
  }
1276
1359
  const requestedCount = input.count ?? 1;
1277
1360
  return {
1278
- text: partialFailure
1279
- ? `Partial result: ${assetList.length} of ${requestedCount} image(s) saved into project ${input.projectId} ` +
1280
- `(error on the rest: ${result.error ?? 'unknown error'}). ` +
1281
- `The saved assets are in data.assets — re-generate only the missing count, don't redo the whole batch. ` +
1282
- `Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"`
1283
- : `Generated ${assetList.length} image(s) into project ${input.projectId} ` +
1284
- `for ${fmtCredits(totalCents)}. ` +
1285
- `Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
1286
- (refEcho ? ` ${refEcho}` : ''),
1361
+ // The pixels are ALREADY here when the disk read worked, so the
1362
+ // review pointer says "look at what you have", not "call another op".
1363
+ // Telling the agent to re-fetch an image already in its context would
1364
+ // buy a wasted turn and teach the wrong habit.
1365
+ text: `${images.length > 0 ? IMAGE_INLINE_REVIEW : IMAGE_FETCH_POINTER} ` +
1366
+ (promptWarning ? `${promptWarning} ` : '') +
1367
+ (partialFailure
1368
+ ? `Partial result: ${assetList.length} of ${requestedCount} image(s) saved into project ${input.projectId} ` +
1369
+ `(error on the rest: ${result.error ?? 'unknown error'}). ` +
1370
+ `The saved assets are in data.assets — re-generate only the missing count, don't redo the whole batch. ` +
1371
+ `Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"`
1372
+ : `Generated ${assetList.length} image(s) into project ${input.projectId} ` +
1373
+ `for ${fmtCredits(totalCents)}. ` +
1374
+ `Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
1375
+ (refEcho ? ` ${refEcho}` : '')),
1287
1376
  images,
1288
1377
  data: {
1289
1378
  model: imageModel,
@@ -1351,7 +1440,9 @@ export const generateImage = {
1351
1440
  images.push({ data: buf.toString('base64'), mimeType: mt });
1352
1441
  }
1353
1442
  return {
1354
- text: `Generated ${urls.length} image(s) for ${fmtCredits(totalCents)}. ` +
1443
+ text: `${IMAGE_INLINE_REVIEW} ` +
1444
+ (promptWarning ? `${promptWarning} ` : '') +
1445
+ `Generated ${urls.length} image(s) for ${fmtCredits(totalCents)}. ` +
1355
1446
  `Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"`,
1356
1447
  images,
1357
1448
  data: {
@@ -1568,6 +1659,16 @@ export function videoCostKey(input) {
1568
1659
  const k = minimaxRefSurchargeCount(input.model, input.referenceImages);
1569
1660
  return `${input.model}-${res}-${input.duration}s${k > 0 ? `-ref${k}` : ''}`;
1570
1661
  }
1662
+ // LTX-2.5, both seats. EXACT-ID SET, NEVER A PREFIX — `ltx-2-5-pro` starts
1663
+ // with `ltx-2-5`, and a prefix match would quote base rates for the dearer
1664
+ // row. No suffix dimension exists: audio is free and there are no references.
1665
+ // The resolution default is read PER ROW (both are 1080p today, but the two
1666
+ // ladders differ, so a shared literal would be a latent bug the day one
1667
+ // moves). Mirrors ltxCreditKey() in slate/src/shared/pricing.ts.
1668
+ if (LTX_MODELS.has(input.model)) {
1669
+ const res = input.videoResolution ?? defaultVideoResolutionFor(input.model);
1670
+ return `${input.model}-${res}-${input.duration}s`;
1671
+ }
1571
1672
  if (input.model.startsWith('seedance')) {
1572
1673
  // Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
1573
1674
  // × vref × res × duration). AI-face route bills the `-face-` key (~45% over
@@ -1813,6 +1914,33 @@ function resolveVideoModel(raw) {
1813
1914
  }
1814
1915
  return null;
1815
1916
  }
1917
+ /**
1918
+ * Pair each reference-audio asset id with the words spoken in it.
1919
+ *
1920
+ * 🚨 THE MODEL RE-TRANSCRIBES A SUPPLIED TAKE. Seedance does not consume
1921
+ * reference audio verbatim — it re-synthesises something close to it, and a
1922
+ * 2026-08-28 field test heard "an app called Slates" come back as "a map called
1923
+ * Slates". The clip carries the voice, the accent and the timing; only text
1924
+ * carries the words. This is the text, and without it every generation with a
1925
+ * voice take has its line guessed.
1926
+ *
1927
+ * Positional in, KEYED out: the desktop route merges its deprecated singular
1928
+ * `audioReferenceAssetId` onto the END of the plural list, so an index would
1929
+ * address different clips depending on which shape the caller used. Blank
1930
+ * entries are dropped rather than sent as empty strings — a clip with no
1931
+ * speech has no line, which is not the same as a line that is empty.
1932
+ */
1933
+ function spokenTextByAssetId(assetIds, spoken) {
1934
+ if (!assetIds?.length || !spoken?.length)
1935
+ return undefined;
1936
+ const out = {};
1937
+ assetIds.forEach((id, i) => {
1938
+ const text = spoken[i]?.trim();
1939
+ if (id && text)
1940
+ out[id] = text;
1941
+ });
1942
+ return Object.keys(out).length > 0 ? out : undefined;
1943
+ }
1816
1944
  // Maps a video model id to its bundled prompting skill (frontmatter `name:`),
1817
1945
  // so guidance text points at a skill that actually exists. Deriving the name
1818
1946
  // via model.split('-')[0] produced 'slates-prompting-kling' / '...-veo', which
@@ -1839,7 +1967,10 @@ function promptingSkillFor(model) {
1839
1967
  }
1840
1968
  export const generateVideo = {
1841
1969
  id: 'slates_generate_video',
1842
- description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3 / slates-prompting-minimax-h3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
1970
+ description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3 / slates-prompting-minimax-h3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8"). ' +
1971
+ // GENERATED from the skill's own slop-token list. Always in context on both
1972
+ // surfaces, so it survives an agent that skips slates_get_prompting_guide.
1973
+ describeBannedTokens('video'),
1843
1974
  input: z.object({
1844
1975
  prompt: z.string().min(1).max(4000),
1845
1976
  // ROUTING doctrine only. Every capability number was stripped on 2026-08-16
@@ -1848,7 +1979,14 @@ export const generateVideo = {
1848
1979
  // "seedance-2.5 480p/720p" were both stated here AND there, and the two
1849
1980
  // copies disagreed — and the second of those went stale on 2026-08-24 when
1850
1981
  // 2.5 gained 1080p, which is exactly the failure mode generating it fixes.
1851
- model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (longer takes, far more references, audio-only refs — but no 4K, and dearer than seedance-2 at every shared resolution, so stay on seedance-2 unless length or reference count is the point), Veo = native-synced-audio niche only, never the default, omni-flash = cheap tier with audio included (t2v, single-start-frame i2v, or reference images; no last frame, no video/audio refs), minimax-h3 = the AUTHORED-AUDIO seat (dialogue, scene sound and score directed as three separate layers in one pass, plus declared reference relationships; reference images past the fifth are a PAID key dimension — pass referenceImages when quoting), minimax-h3-max = the same model post-trained by fal for SPEED, capped at 768p, and DEARER than minimax-h3 at the tier they share — a deliberate pick, never a default and never the cheap H3; it still takes firstFrameAssetId/lastFrameAssetId, but has no reference endpoint, so the reference set is minimax-h3 only. All are VIDEO-only. Each model's legal aspect ratios, durations and resolutions are in those params' own descriptions — read them there, not from memory. For per-call cost, call slates_estimate_generation_cost.`),
1982
+ model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate ` +
1983
+ `params (registry cost keys like "kling-v3-standard-8s" auto-resolve). All are VIDEO-only.\n` +
1984
+ // GENERATED from MODEL_FACTS. The paragraph that stood here restated it by
1985
+ // hand and had already drifted a phrase at a time.
1986
+ `${describeRouting('video', 'generate')}\n` +
1987
+ `Full table: the slates-model-selection skill. Each model's legal aspect ratios, durations and ` +
1988
+ `resolutions are in those params' own descriptions — read them there, not from memory. ` +
1989
+ `For per-call cost, call slates_estimate_generation_cost.`),
1852
1990
  projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
1853
1991
  // 🚨 THESE THREE DESCRIPTIONS ARE GENERATED FROM `MODEL_CAPABILITIES`.
1854
1992
  // Never hand-write a ratio, resolution or duration into them again — every
@@ -1877,6 +2015,7 @@ export const generateVideo = {
1877
2015
  videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS (UUIDs or badge codes) read alongside the images and audio in the same generation — own-footage restyle, MOTION TRANSFER ("the character from image 1 performs the motion from video 1"), or dialogue conditioning. Cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; ignored elsewhere. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to combined input+output seconds (the vref key) — pass videoReferenceSecondsEach so the quote is right. If any clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Over the cap is REFUSED, never trimmed: a dropped clip would already have been priced in.`),
1878
2016
  videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — the bill is Σceil(each) + output seconds. The server re-derives this by probing every uploaded clip, so an understated value just gets corrected upward.'),
1879
2017
  audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO clips (UUIDs or badge codes) read alongside the images and video — e.g. lip-sync a character to a line ("the character in image 1 speaks the dialogue from audio 1"). Cited as "audio 1", "audio 2"… in the order given. No billing surcharge (Seedance audio is included). ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
2018
+ audioReferenceSpokenText: z.array(z.string()).optional().describe('STRONGLY RECOMMENDED whenever a reference clip contains SPEECH. Same order and length as audioReferenceAssetIds; use "" for a clip with no words (music, ambience, room tone). The model RE-TRANSCRIBES a supplied take rather than using it verbatim — a field test heard "an app called Slates" come back as "a map called Slates" — so the audio decides the VOICE, the ACCENT and the TIMING while only text decides the WORDS. Give the exact line here and it is quoted into the prompt beside the citation. Omit it and the words are a guess. Pairs with the plural audioReferenceAssetIds; the deprecated singular audioReferenceAssetId carries no text.'),
1880
2019
  sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
1881
2020
  audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
1882
2021
  generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
@@ -1911,11 +2050,16 @@ export const generateVideo = {
1911
2050
  // no asset to reference later, and a failed gen leaves the user with
1912
2051
  // nothing. The MCP-only headless path that exists for image gen is
1913
2052
  // not reasonable for video given the cost.
2053
+ // Non-blocking prompt hygiene, computed once. Reported on the gates that
2054
+ // fire BEFORE any spend, where a rewrite is still free.
2055
+ const promptWarning = bannedTokenWarning(input.prompt, 'video');
1914
2056
  if (!input.projectId) {
1915
2057
  return ok({
1916
2058
  requires_clarification: true,
1917
2059
  missing: ['projectId'],
1918
- message: 'projectId is required for video generation. Use slates_list_projects to find one or slates_create_project to make a new one. Video gens cost tens to a few hundred credits per call — they need to land in a project so the user sees the progress card and the result.',
2060
+ ...(promptWarning ? { prompt_warning: promptWarning } : {}),
2061
+ message: (promptWarning ? `${promptWarning}\n\n` : '') +
2062
+ 'projectId is required for video generation. Use slates_list_projects to find one or slates_create_project to make a new one. Video gens cost tens to a few hundred credits per call — they need to land in a project so the user sees the progress card and the result.',
1919
2063
  });
1920
2064
  }
1921
2065
  if (!input.aspectRatio || !input.duration) {
@@ -1927,7 +2071,9 @@ export const generateVideo = {
1927
2071
  return ok({
1928
2072
  requires_clarification: true,
1929
2073
  missing,
1930
- message: `Missing required field(s): ${missing.join(', ')}. ` +
2074
+ ...(promptWarning ? { prompt_warning: promptWarning } : {}),
2075
+ message: (promptWarning ? `${promptWarning}\n\n` : '') +
2076
+ `Missing required field(s): ${missing.join(', ')}. ` +
1931
2077
  `Read the slates-cost-discipline + ${promptingSkillFor(input.model)} skills, ` +
1932
2078
  // Generated from MODEL_CAPABILITIES. The prose that stood here claimed
1933
2079
  // "Veo locks to 16:9", "Kling/Seedance support 1:1 16:9 9:16 4:3 3:4
@@ -1973,6 +2119,34 @@ export const generateVideo = {
1973
2119
  });
1974
2120
  }
1975
2121
  }
2122
+ // LTX-2.5's SHAPE constraint: FRAMES, NEVER REFERENCES. fal publishes
2123
+ // text-to-video and image-to-video for `lightricks/ltx-2.5` and nothing
2124
+ // else — there is no reference-to-video endpoint on either seat, so there
2125
+ // is no transport for an ingredient, character, environment, style,
2126
+ // reference-video or reference-audio input.
2127
+ //
2128
+ // This has to be an EXPLICIT refusal rather than a silent no-op: accepting
2129
+ // reference ids we cannot send is the exact "no error, no warning, no
2130
+ // images in the request" failure `seedance-2.5-edit` shipped with. Start
2131
+ // and end FRAMES are unaffected — image-to-video carries `image_url` plus
2132
+ // an optional `end_image_url`, which is what `features.lastFrame` records.
2133
+ if (LTX_MODELS.has(input.model)) {
2134
+ const refImages = (input.ingredientAssetIds?.length ?? 0) +
2135
+ (input.characterAssetIds?.length ?? 0) +
2136
+ (input.environmentAssetIds?.length ?? 0) +
2137
+ (input.styleAssetIds?.length ?? 0);
2138
+ const refMedia = (input.videoReferenceAssetIds?.length ?? 0) +
2139
+ (input.audioReferenceAssetIds?.length ?? 0) +
2140
+ (input.videoReferenceAssetId ? 1 : 0) +
2141
+ (input.audioReferenceAssetId ? 1 : 0);
2142
+ if (refImages > 0 || refMedia > 0) {
2143
+ return ok({
2144
+ requires_clarification: true,
2145
+ missing: [],
2146
+ message: `${input.model} takes a prompt and up to two frames (start and/or end) — it has no reference endpoint at all, so reference images, video and audio cannot be sent. Drop them, or switch to minimax-h3, which reads ${getModelCapability('minimax-h3')?.maxIngredientImages ?? 9} images plus reference video and audio.`,
2147
+ });
2148
+ }
2149
+ }
1976
2150
  // MiniMax H3's SHAPE constraints. Counts and caps are read from the
1977
2151
  // capability SSOT; only the endpoint SHAPE is stated here, because it is
1978
2152
  // not a number the registry models: fal publishes text-to-video,
@@ -2033,6 +2207,20 @@ export const generateVideo = {
2033
2207
  message: `${input.model} takes at most ${maxTotal} reference files across all modalities (you passed ${refImages + refMedia}).`,
2034
2208
  });
2035
2209
  }
2210
+ // A misaligned spoken-text array would attach one clip's line to another
2211
+ // and send the model the wrong words with nothing on screen to say so.
2212
+ // Refuse rather than truncate: the whole point of the field is that the
2213
+ // WORDS are exact.
2214
+ if (input.audioReferenceSpokenText &&
2215
+ input.audioReferenceSpokenText.length !== (input.audioReferenceAssetIds?.length ?? 0)) {
2216
+ return ok({
2217
+ requires_clarification: true,
2218
+ missing: [],
2219
+ message: `audioReferenceSpokenText must be the same length as audioReferenceAssetIds ` +
2220
+ `(${input.audioReferenceSpokenText.length} vs ${input.audioReferenceAssetIds?.length ?? 0}) ` +
2221
+ `— it pairs by position. Use "" for a clip with no speech.`,
2222
+ });
2223
+ }
2036
2224
  // fal: "Audio cannot be the only reference input; provide at least one
2037
2225
  // reference image or video with it."
2038
2226
  const audioRefs = (input.audioReferenceAssetIds?.length ?? 0) + (input.audioReferenceAssetId ? 1 : 0);
@@ -2225,6 +2413,7 @@ export const generateVideo = {
2225
2413
  text: `Pre-flight for ${input.duration}s ${input.model} (${costKey}): ` +
2226
2414
  `${fmtCredits(totalCents)}.` +
2227
2415
  refSummary +
2416
+ (promptWarning ? `\n\n${promptWarning}` : '') +
2228
2417
  `\n\nWhen ready, re-call slates_generate_video with confirm=true and the (possibly revised) prompt.`,
2229
2418
  images: refImages,
2230
2419
  data: {
@@ -2269,6 +2458,13 @@ export const generateVideo = {
2269
2458
  audioReferenceAssetId: input.audioReferenceAssetId,
2270
2459
  videoReferenceAssetIds: input.videoReferenceAssetIds,
2271
2460
  audioReferenceAssetIds: input.audioReferenceAssetIds,
2461
+ // The route keys this by ASSET ID, not by position, because its own
2462
+ // singular/plural merge appends to the END of the list — an index would
2463
+ // mean different clips depending on which shape the caller used. The op
2464
+ // takes the friendlier positional array and re-keys it here, AFTER
2465
+ // `rids()` has resolved badge codes, so the keys are the ids the route
2466
+ // will resolve. A blank entry is dropped rather than sent as "".
2467
+ audioReferenceSpokenText: spokenTextByAssetId(input.audioReferenceAssetIds, input.audioReferenceSpokenText),
2272
2468
  sound: input.sound,
2273
2469
  audioLanguage: input.audioLanguage,
2274
2470
  generateMusic: input.generateMusic,
@@ -2296,7 +2492,12 @@ export const generateVideo = {
2296
2492
  }, refEcho);
2297
2493
  }
2298
2494
  return {
2299
- text: `Generated ${input.duration}s ${input.model} video into project ${input.projectId} ` +
2495
+ text:
2496
+ // No frames come back with a video generation, so unlike the image op
2497
+ // this one has to name the op that fetches them.
2498
+ `${VIDEO_REVIEW_POINTER} ` +
2499
+ (promptWarning ? `${promptWarning} ` : '') +
2500
+ `Generated ${input.duration}s ${input.model} video into project ${input.projectId} ` +
2300
2501
  `for ${fmtCredits(totalCents)}. ` +
2301
2502
  `Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
2302
2503
  (refEcho ? ` ${refEcho}` : ''),
@@ -2325,7 +2526,10 @@ export const generateAudio = {
2325
2526
  projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
2326
2527
  model: z
2327
2528
  .enum(AUDIO_MODELS)
2328
- .describe('Audio surface. seed-audio = scene/ambience/beds/dialogue (default choice), eleven-sfx = one precise effect or a seamless loop. Routing doctrine: slates-model-selection skill.'),
2529
+ .describe(
2530
+ // GENERATED from MODEL_FACTS — the audio lane's routing lived only in the
2531
+ // system prompt and in a one-line paraphrase here.
2532
+ `Audio surface.\n${describeRouting('audio')}\nFull table: the slates-model-selection skill.`),
2329
2533
  prompt: z
2330
2534
  .string()
2331
2535
  .min(1)
@@ -2693,7 +2897,13 @@ export const editVideo = {
2693
2897
  sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
2694
2898
  prompt: z.string().min(1).max(2500).describe('The change, not the whole scene — e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
2695
2899
  // Clip windows and resolutions GENERATED from MODEL_CAPABILITIES.
2696
- model: zEnum(EDIT_VIDEO_MODELS).optional().describe(`Default kling-v3.0-omni-edit. Pro (~25¢/s vs ~19¢/s) only for hero shots where fidelity matters. omni-flash-edit (~19¢/s) for prompt-only edits — it takes NO character/style refs. seedance-2.5-edit is the ONLY engine that accepts a clip longer than 15s; it also takes NO character/style refs on this op. Source-clip length per engine: ${describeDurations(EDIT_VIDEO_MODELS)}. Output resolution: ${describeVideoResolutions(EDIT_VIDEO_MODELS)}`),
2900
+ model: zEnum(EDIT_VIDEO_MODELS).optional().describe(
2901
+ // Routing GENERATED from MODEL_FACTS; the paragraph that stood here was
2902
+ // hand-written and carried per-second prices, which drift and which the
2903
+ // agent's own REAL NUMBERS ONLY rule forbids it repeating.
2904
+ `Default kling-v3.0-omni-edit. Neither omni-flash-edit nor seedance-2.5-edit takes character/style refs on this op.\n` +
2905
+ `${describeRouting('video', 'edit')}\n` +
2906
+ `Source-clip length per engine: ${describeDurations(EDIT_VIDEO_MODELS)}. Output resolution: ${describeVideoResolutions(EDIT_VIDEO_MODELS)}`),
2697
2907
  characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY — rejected on omni-flash-edit.'),
2698
2908
  styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
2699
2909
  keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
@@ -3472,6 +3682,33 @@ function resolveGuideTopic(topic) {
3472
3682
  if (t === 'slates-character-turnaround' || t === 'character-turnaround') {
3473
3683
  return 'slates-character-identity';
3474
3684
  }
3685
+ // ⚠️ Previs aliases sit HIGH, before the model-prefix rules below. `dialogue`
3686
+ // already resolves to the Seed Audio guide, and `style`/`camera` words are a
3687
+ // hair away from the style-prompting and model blocks — anchoring these here
3688
+ // keeps a previs ask out of an audio guide.
3689
+ if (t === 'previs' ||
3690
+ t === 'pre-vis' ||
3691
+ t === 'previz' ||
3692
+ t === 'blocking' ||
3693
+ t === 'blockout' ||
3694
+ t === 'greybox' ||
3695
+ t === 'grey-box' ||
3696
+ t === 'graybox' ||
3697
+ t === 'blender') {
3698
+ return 'slates-previs-blocking';
3699
+ }
3700
+ if (t === 'camera' || t === 'camera-moves' || t === 'camera moves' || t === 'shot-list' || t === 'shot list') {
3701
+ return 'slates-camera-language';
3702
+ }
3703
+ if (t === 'blocking-to-prompt' || t === 'previs-prompt' || t === 'reference-video' || t === 'video-to-video' || t === 'v2v') {
3704
+ return 'slates-blocking-to-prompt';
3705
+ }
3706
+ if (t === 'dialogue-blocking' || t === 'dialogue blocking' || t === '180-rule' || t === 'eyelines' || t === 'seating') {
3707
+ return 'slates-dialogue-blocking';
3708
+ }
3709
+ if (t === 'restyle' || t === 're-style' || t === 'style-swap' || t === 'style swap' || t === 'style-variants') {
3710
+ return 'slates-restyle-from-blocking';
3711
+ }
3475
3712
  if (t === 'model-selection' ||
3476
3713
  t === 'model selection' ||
3477
3714
  t === 'which-model' ||
@@ -3501,6 +3738,12 @@ function resolveGuideTopic(topic) {
3501
3738
  t.startsWith('h3-')) {
3502
3739
  return 'slates-prompting-minimax-h3';
3503
3740
  }
3741
+ // LTX-2.5 — both seats share one skill. A PREFIX is right HERE (unlike every
3742
+ // rate, key and endpoint lookup, which must be exact) precisely because both
3743
+ // rows resolve to the same guide: `ltx-2-5-pro` matching the `ltx` prefix is
3744
+ // the intended outcome, not a collision.
3745
+ if (t.startsWith('ltx') || t === 'lightricks')
3746
+ return 'slates-prompting-ltx-2-5';
3504
3747
  if (t.startsWith('kling-mc'))
3505
3748
  return 'slates-prompting-motion-transfer';
3506
3749
  if (t === 'edit-video' || t === 'video-edit' || t === 'edit video' || t === 'video edit')
@@ -3553,6 +3796,27 @@ function resolveGuideTopic(topic) {
3553
3796
  }
3554
3797
  return null;
3555
3798
  }
3799
+ /**
3800
+ * The guide index, GENERATED from SKILLS.
3801
+ *
3802
+ * The list here was hand-typed and had drifted to 25 of 32 names — the prompting
3803
+ * guides for GPT Image 2, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
3804
+ * all missing, so an agent reading this description could not learn they exist.
3805
+ * A hand-typed index of a generated corpus is a stale index; it is only a matter
3806
+ * of when.
3807
+ *
3808
+ * Per-model guides are listed as BARE NAMES: the name is the description, and
3809
+ * `resolveGuideTopic()` resolves a model id to the right one anyway.
3810
+ */
3811
+ function describeGuideTopics() {
3812
+ const names = Object.keys(SKILLS).sort();
3813
+ const perModel = names.filter((n) => n.startsWith('slates-prompting-'));
3814
+ const rest = names.filter((n) => !n.startsWith('slates-prompting-'));
3815
+ return (`Workflow and cross-cutting guides: ${rest.join(', ')}. ` +
3816
+ `Per-model prompting guides (or just pass the model id, which resolves to one of these): ` +
3817
+ `${perModel.join(', ')}. ` +
3818
+ `Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.`);
3819
+ }
3556
3820
  export const getPromptingGuide = {
3557
3821
  id: 'slates_get_prompting_guide',
3558
3822
  description: "Return the full markdown of a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id (e.g. 'veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2') which maps to the right guide. ALWAYS read 'slates-cost-discipline' plus the relevant model guide before your first generation in a session.",
@@ -3560,7 +3824,7 @@ export const getPromptingGuide = {
3560
3824
  topic: z
3561
3825
  .string()
3562
3826
  .min(1)
3563
- .describe('Guide name, model id, or style name. Guides: slates-model-selection (which model for which job — read before choosing any model), slates-cost-discipline, slates-content-policy, slates-style-prompting, slates-prompting-nano-banana-2, slates-prompting-veo-3, slates-prompting-kling-v3, slates-prompting-seedance, slates-prompting-seed-audio, slates-prompting-elevenlabs, slates-prompting-lip-sync, slates-prompting-motion-transfer, slates-prompting-flux-2-max, slates-prompting-seedream-5-lite, slates-edit-and-iterate, slates-vision-feedback-loop, slates-character-identity, slates-storyboard-from-script, slates-direct-response-ad, slates-one-prompt-film. Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.'),
3827
+ .describe(`Guide name, model id, or style name. ${describeGuideTopics()}`),
3564
3828
  }),
3565
3829
  async run(input) {
3566
3830
  const resolved = resolveGuideTopic(input.topic);
@@ -3574,6 +3838,159 @@ export const getPromptingGuide = {
3574
3838
  };
3575
3839
  },
3576
3840
  };
3841
+ // ── Blender previs ──────────────────────────────────────────────
3842
+ //
3843
+ // The only ops that talk to a third transport: a localhost socket into a
3844
+ // running Blender carrying the Slates add-on. They exist to produce ONE
3845
+ // artifact — a grey-box blocking clip whose asset id goes straight into
3846
+ // `slates_generate_video`'s `videoReferenceAssetIds`, so the model renders a
3847
+ // world around a camera path instead of inventing one.
3848
+ //
3849
+ // 🚨 There is deliberately no camera-move library here. Camera work is written
3850
+ // as `bpy` by the agent through `slates_blender_execute`, against the Blender
3851
+ // API reference the add-on ships. A fixed menu of moves would cap the workflow
3852
+ // at whatever we thought of; code execution plus real docs does not.
3853
+ // The setup sentence and the download URL live in `clients/blender.ts`, beside
3854
+ // the port range they belong to — one home, so a moved page is one edit.
3855
+ const BLENDER_UNAVAILABLE_HINT = `Blender previs needs the Slates Blender add-on running. ${BLENDER_SETUP_HINT}`;
3856
+ export const blenderStatus = {
3857
+ id: 'slates_blender_status',
3858
+ description: 'Check whether a Blender running the Slates add-on is reachable, and if so return its scene summary (timing, camera, collection tree). Call this FIRST in any previs workflow — every other Blender op fails with the same setup message when the bridge is down, and knowing the frame range and fps up front is what keeps the blocking and the prompt timings in agreement.',
3859
+ input: z.object({}),
3860
+ async run() {
3861
+ const client = new BlenderBridgeClient();
3862
+ if (!(await client.isReachable())) {
3863
+ return ok({ connected: false, hint: BLENDER_UNAVAILABLE_HINT });
3864
+ }
3865
+ return ok({ connected: true, scene: await client.call('result = _mod("scene").summary()') });
3866
+ },
3867
+ };
3868
+ export const blenderExecute = {
3869
+ id: 'slates_blender_execute',
3870
+ description: 'Run Python (`bpy`) inside the connected Blender and return whatever the code assigns to a dict named `result`. This is how blocking gets built: primitives, empties, constraints, camera rigs, keyframes, markers. Anything Blender can do, this can do. Assign a dict to `result` to get data back (e.g. `result = {"camera": cam.name}`); print() output comes back separately as stdout. On an exception you get the full traceback — read it, fix the code, retry. Before writing an unfamiliar call, look up its real signature with slates_blender_docs rather than guessing: the add-on ships the Blender 5.1 API reference precisely so you do not have to recall it.',
3871
+ input: z.object({
3872
+ code: z
3873
+ .string()
3874
+ .min(1)
3875
+ .describe('Python source. Assign a JSON-serialisable dict to `result` to return data.'),
3876
+ timeoutSeconds: z
3877
+ .number()
3878
+ .int()
3879
+ .min(5)
3880
+ .max(900)
3881
+ .optional()
3882
+ .describe('How long to wait for Blender (default 60). Raise it for heavy geometry.'),
3883
+ }),
3884
+ async run(input) {
3885
+ const client = new BlenderBridgeClient();
3886
+ const { result, stdout, stderr } = await client.execute(input.code, {
3887
+ timeoutMs: input.timeoutSeconds ? input.timeoutSeconds * 1000 : undefined,
3888
+ });
3889
+ return ok({ result, ...(stdout ? { stdout } : {}), ...(stderr ? { stderr } : {}) });
3890
+ },
3891
+ };
3892
+ export const blenderScene = {
3893
+ id: 'slates_blender_scene',
3894
+ description: 'Scene summary from the connected Blender: frame range, fps, duration, render resolution, the active camera with its keyframe times in both frames and seconds, and the full collection/object tree with transforms and constraints. Cheap — call it freely between edits. The camera keyframe times ARE the cut structure, so read them before writing any shot-by-shot prompt.',
3895
+ input: z.object({}),
3896
+ async run() {
3897
+ return ok(await new BlenderBridgeClient().call('result = _mod("scene").summary()'));
3898
+ },
3899
+ };
3900
+ export const blenderDocs = {
3901
+ id: 'slates_blender_docs',
3902
+ description: 'Look up a dotted Blender Python API identifier in the bundled 5.1 reference — e.g. "bpy.ops.object", "bpy.types.Camera", "bpy.types.FollowPathConstraint". Pass "*" for top-level modules or "bpy.ops.*" to list a namespace. Use this instead of recalling a signature from memory: invented operator names and wrong enum values are the most common way previs code fails, and they fail silently often enough to be worth the lookup.',
3903
+ input: z.object({
3904
+ identifier: z.string().min(1).describe('Dotted identifier, or a namespace wildcard like "bpy.ops.*".'),
3905
+ }),
3906
+ async run(input) {
3907
+ const code = `result = _mod("docs").lookup(${JSON.stringify(input.identifier)})`;
3908
+ return ok(await new BlenderBridgeClient().call(code));
3909
+ },
3910
+ };
3911
+ export const blenderSearchDocs = {
3912
+ id: 'slates_blender_search_docs',
3913
+ description: 'Full-text search of the bundled Blender documentation for when you do not know the identifier yet. scope "api" searches the Python reference; "manual" searches the user manual for concepts and workflow ("how does Follow Path work", "bezier interpolation handles"). Use slates_blender_docs when you know the name and this when you do not.',
3914
+ input: z.object({
3915
+ query: z.string().min(2),
3916
+ scope: z.enum(['api', 'manual']).optional().describe('Default "api".'),
3917
+ maxResults: z.number().int().min(1).max(20).optional().describe('Default 8.'),
3918
+ }),
3919
+ async run(input) {
3920
+ const args = [
3921
+ JSON.stringify(input.query),
3922
+ JSON.stringify(input.scope ?? 'api'),
3923
+ String(input.maxResults ?? 8),
3924
+ ].join(', ');
3925
+ return ok(await new BlenderBridgeClient().call(`result = _mod("docs").search(${args})`));
3926
+ },
3927
+ };
3928
+ export const blenderRenderBlocking = {
3929
+ id: 'slates_blender_render_blocking',
3930
+ description: "Render the connected Blender scene camera to a grey-box mp4 — the blocking clip that locks camera motion for generation — and, when projectId is given, import it into that Slates project as a video asset in the same call. The returned asset id goes into slates_generate_video's videoReferenceAssetIds and the returned durationSeconds into videoReferenceSecondsEach. Renders through the SCENE camera using scene render settings, never the user's viewport, so output does not depend on where they left their mouse. Untextured is correct: the clip supplies camera path and timing, the references supply the look. Colour in the blocking is NOTATION, not look — a distinct viewport colour per character is what binds a proxy to its reference image across cuts, and a marked face encodes which way a featureless proxy is facing. State every such mapping in the generation prompt AND state that the colours themselves are not inherited, or the model renders a literally red person. See the slates-previs-blocking and slates-blocking-to-prompt skills. Keep it at or under 30s — seedance-2.5 accepts reference videos up to 30s, the other reference-video models up to 15s.",
3931
+ input: z.object({
3932
+ projectId: z
3933
+ .string()
3934
+ .uuid()
3935
+ .optional()
3936
+ .describe('Import the clip into this project and return its asset. Omit to just get a file path.'),
3937
+ resolutionX: z.number().int().min(256).max(4096).optional().describe('Default 1920.'),
3938
+ resolutionY: z.number().int().min(256).max(4096).optional().describe('Default 1080.'),
3939
+ fps: z.number().int().min(1).max(120).optional().describe('Default 24. Match what the shot list assumes.'),
3940
+ frameStart: z.number().int().optional().describe("Default: the scene's own frame_start."),
3941
+ frameEnd: z.number().int().optional().describe("Default: the scene's own frame_end."),
3942
+ basename: z.string().min(1).max(64).optional().describe('Filename stem. Default "blocking".'),
3943
+ }),
3944
+ async run(input, ctx) {
3945
+ const kwargs = [];
3946
+ if (input.resolutionX !== undefined)
3947
+ kwargs.push(`resolution_x=${input.resolutionX}`);
3948
+ if (input.resolutionY !== undefined)
3949
+ kwargs.push(`resolution_y=${input.resolutionY}`);
3950
+ if (input.fps !== undefined)
3951
+ kwargs.push(`fps=${input.fps}`);
3952
+ if (input.frameStart !== undefined)
3953
+ kwargs.push(`frame_start=${input.frameStart}`);
3954
+ if (input.frameEnd !== undefined)
3955
+ kwargs.push(`frame_end=${input.frameEnd}`);
3956
+ if (input.basename !== undefined)
3957
+ kwargs.push(`basename=${JSON.stringify(input.basename)}`);
3958
+ // 🚨 THE RENDER IS DEFERRED, AND THAT IS WHY THIS IS NOT ONE LINE.
3959
+ // In an interactive Blender `render_blocking` INVOKES the render rather
3960
+ // than executing it — a synchronous animation render driven from the
3961
+ // bridge's own `bpy.app.timers` callback would re-enter the main loop
3962
+ // running it. So it hands back a `check_is_finished` callable instead of
3963
+ // the clip, and assigning that to `check_is_finished` is the bridge's
3964
+ // documented convention for "hold the socket open and answer when the job
3965
+ // lands" (see the add-on's `bridge/deferred.py`). The deferred path wraps
3966
+ // the eventual dict in the SAME `{status, result}` envelope, so everything
3967
+ // downstream of this call is identical either way. Headless Blender has no
3968
+ // job system to poll, renders synchronously, and takes the `else`.
3969
+ const render = (await new BlenderBridgeClient().call(`_previs_result = _mod("previs").render_blocking(${kwargs.join(', ')})\n` +
3970
+ 'if callable(_previs_result):\n' +
3971
+ ' check_is_finished = _previs_result\n' +
3972
+ 'else:\n' +
3973
+ ' result = _previs_result\n', { timeoutMs: RENDER_TIMEOUT_MS }));
3974
+ if (!input.projectId) {
3975
+ return ok({
3976
+ ...render,
3977
+ next: 'Pass projectId to import this into a Slates project, or call slates_upload_reference_image with type "video".',
3978
+ });
3979
+ }
3980
+ // The same desktop route slates_upload_reference_image uses — the file is
3981
+ // probed on ingest, so duration and dimensions are known immediately.
3982
+ const uploaded = await ctx.desktop().post('/agent/assets/upload', {
3983
+ projectId: input.projectId,
3984
+ filePath: render.filePath,
3985
+ type: 'video',
3986
+ });
3987
+ return ok({
3988
+ ...render,
3989
+ asset: uploaded.asset ?? uploaded,
3990
+ next: "Pass the asset's id in slates_generate_video videoReferenceAssetIds, with videoReferenceSecondsEach set to durationSeconds.",
3991
+ });
3992
+ },
3993
+ };
3577
3994
  // ── Aggregation ─────────────────────────────────────────────────
3578
3995
  export const ALL_OPERATIONS = [
3579
3996
  getWorkspaceState,
@@ -3652,5 +4069,18 @@ export const ALL_OPERATIONS = [
3652
4069
  batchUpdateFrames,
3653
4070
  deleteFrame,
3654
4071
  getPromptingGuide,
4072
+ // ── Blender previs, LAST and deliberately ────────────────────────────
4073
+ // This order is not cosmetic: `slate/src/main/studio-agent/ops.ts` maps this
4074
+ // array straight into the Anthropic `tools` array, and that block sits inside
4075
+ // the desktop Studio Agent's PROMPT-CACHED PREFIX. These six landed at the
4076
+ // TOP, which put a third transport nobody without Blender can reach ahead of
4077
+ // `slates_get_workspace_state` in every conversation the app has. They are a
4078
+ // niche lane off the end of the surface, and the list should read that way.
4079
+ blenderStatus,
4080
+ blenderExecute,
4081
+ blenderScene,
4082
+ blenderDocs,
4083
+ blenderSearchDocs,
4084
+ blenderRenderBlocking,
3655
4085
  ];
3656
4086
  //# sourceMappingURL=index.js.map