@slatesvideo/shared 0.6.1 → 0.6.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/clients/blender.d.ts +50 -0
- package/dist/clients/blender.js +195 -0
- package/dist/index.d.ts +3 -1
- package/dist/index.js +26 -0
- package/dist/operations/index.d.ts +25 -1
- package/dist/operations/index.js +454 -24
- package/dist/prompts/agent-doctrine.d.ts +36 -0
- package/dist/prompts/agent-doctrine.js +194 -0
- package/dist/prompts/banned-tokens.d.ts +28 -0
- package/dist/prompts/banned-tokens.js +152 -0
- package/dist/prompts/model-capabilities.d.ts +13 -1
- package/dist/prompts/model-capabilities.js +97 -2
- package/dist/prompts/model-facts.d.ts +20 -0
- package/dist/prompts/model-facts.js +87 -23
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +65 -0
- package/dist/prompts/reference-composer.d.ts +57 -0
- package/dist/prompts/reference-composer.js +70 -1
- package/dist/skills/content.js +8 -2
- package/exports/slates-prompt-builder/generated/SKILL.md +3 -3
- package/exports/slates-prompt-builder/generated/reference-seedance.md +2 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +8 -8
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +1 -1
- package/skills/slates-blocking-to-prompt.md +250 -0
- package/skills/slates-camera-language.md +196 -0
- package/skills/slates-dialogue-blocking.md +134 -0
- package/skills/slates-previs-blocking.md +153 -0
- package/skills/slates-prompting-ltx-2-5.md +180 -0
- package/skills/slates-prompting-nano-banana-2.md +10 -0
- package/skills/slates-prompting-seedance.md +10 -1
- package/skills/slates-restyle-from-blocking.md +121 -0
package/dist/operations/index.js
CHANGED
|
@@ -12,10 +12,16 @@
|
|
|
12
12
|
import { z } from 'zod';
|
|
13
13
|
import { SlatesCloudClient } from '../clients/cloud.js';
|
|
14
14
|
import { SlatesDesktopClient } from '../clients/desktop.js';
|
|
15
|
+
import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
|
|
15
16
|
import { SKILLS } from '../skills/content.js';
|
|
16
17
|
// Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
|
|
17
18
|
// "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
|
|
18
|
-
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords
|
|
19
|
+
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
|
|
20
|
+
// 🚨 THE routing renderer. Routing prose is GENERATED here, never typed:
|
|
21
|
+
// slates-mcp/CLAUDE.md forbids restating it in an op description, and this
|
|
22
|
+
// file did it anyway for 1,282 characters that repeated MODEL_FACTS phrase
|
|
23
|
+
// for phrase. Edit model-facts.ts; both surfaces follow.
|
|
24
|
+
describeRouting, } from '../prompts/model-facts.js';
|
|
19
25
|
// 🚨 MODEL CAPABILITY SSOT. Aspect ratios, video resolutions and durations are
|
|
20
26
|
// VALIDATED against and DESCRIBED from `MODEL_CAPABILITIES` — this file must
|
|
21
27
|
// never re-state one of those constraints as a literal enum or a sentence
|
|
@@ -26,6 +32,13 @@ import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } fr
|
|
|
26
32
|
// 1080p. A customer burned a round trip on `4:5` + Seedance: accepted here,
|
|
27
33
|
// queued, credits reserved, rejected by the provider asynchronously.
|
|
28
34
|
import { AGENT_ROUTE_PROVIDER, aspectRatioUnion, videoResolutionUnion, durationBounds, aspectRatiosFor, getModelCapability, defaultVideoResolutionFor, checkAspectRatio, checkVideoResolution, checkDuration, describeAspectRatios, describeVideoResolutions, describeDurations, describeReferenceImageCaps, } from '../prompts/model-capabilities.js';
|
|
35
|
+
// 🚨 "LOAD THE GUIDE" MADE STRUCTURAL. The never-use token lists are EXTRACTED
|
|
36
|
+
// from the skill files (between `@banned` markers) and inlined into the two
|
|
37
|
+
// generate ops' descriptions, which are always in context on both surfaces —
|
|
38
|
+
// no call to skip, no discretion. `bannedTokenWarning` then reports what the
|
|
39
|
+
// submitted prompt actually contained, in the result, without blocking it.
|
|
40
|
+
// Never hand-type one of these tokens here; edit the skill.
|
|
41
|
+
import { describeBannedTokens, bannedTokenWarning } from '../prompts/banned-tokens.js';
|
|
29
42
|
export function defaultContext() {
|
|
30
43
|
return {
|
|
31
44
|
cloud: () => new SlatesCloudClient(),
|
|
@@ -79,6 +92,30 @@ function creditsFromDollars(dollars) {
|
|
|
79
92
|
// Shared describe-text for the background flag on every generate_* op.
|
|
80
93
|
const BACKGROUND_DESCRIBE = 'Submit and return immediately with generationId(s) instead of blocking until the file is saved. ' +
|
|
81
94
|
'Poll with slates_get_generation_status. Recommended for video (1-5 min renders).';
|
|
95
|
+
// ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
|
|
96
|
+
//
|
|
97
|
+
// QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
|
|
98
|
+
// pre-condition can. What it CAN have is a result the agent cannot avoid
|
|
99
|
+
// reading, naming the exact op. Deliberately NOT an auto-fetch: that would
|
|
100
|
+
// spend vision tokens on every generation whether review was wanted or not,
|
|
101
|
+
// and the sandbox doctrine says the tool is available, not mandatory.
|
|
102
|
+
//
|
|
103
|
+
// ⚠️ These three differ because what the agent already HAS differs, and telling
|
|
104
|
+
// it to re-fetch something already in front of it burns a turn for nothing:
|
|
105
|
+
// a blocking image generation returns the pixels inline, a video generation
|
|
106
|
+
// returns none, and a background submission has no asset yet.
|
|
107
|
+
const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it against the brief before you describe it. ' +
|
|
108
|
+
'Any claim about how it LOOKS must come from pixels you actually received (slates-vision-feedback-loop).';
|
|
109
|
+
const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
|
|
110
|
+
'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
|
|
111
|
+
const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
|
|
112
|
+
'slates_get_asset_video_frames for video.';
|
|
113
|
+
// The image saved, but reading it back off disk failed (best-effort fetch). The
|
|
114
|
+
// agent has an asset and NO pixels, which is the one state where a quality
|
|
115
|
+
// claim would be pure invention — so this branch has to say so rather than
|
|
116
|
+
// fall through to no pointer at all.
|
|
117
|
+
const IMAGE_FETCH_POINTER = 'The pixels could not be attached to this result: call slates_get_asset_image on the asset id ' +
|
|
118
|
+
'above before describing how it looks.';
|
|
82
119
|
// Early-return shape when a generation route accepted the job in background
|
|
83
120
|
// mode ({ background: true } in the response). No inline-image fetch — the
|
|
84
121
|
// asset doesn't exist yet; the poller delivers it on completion.
|
|
@@ -87,7 +124,8 @@ function backgroundSubmitted(kind, ids, extra, note) {
|
|
|
87
124
|
return {
|
|
88
125
|
text: `Submitted ${kind} in the background — generationId(s): ${idText}. ` +
|
|
89
126
|
`Call slates_get_generation_status with waitSeconds: 45 (it long-polls and returns on completion — ` +
|
|
90
|
-
`never a rapid loop; video renders commonly take 1-5 minutes). Generations survive app restarts
|
|
127
|
+
`never a rapid loop; video renders commonly take 1-5 minutes). Generations survive app restarts. ` +
|
|
128
|
+
BACKGROUND_REVIEW_POINTER +
|
|
91
129
|
(note ? ` ${note}` : ''),
|
|
92
130
|
data: { generationIds: ids, status: 'processing', ...extra },
|
|
93
131
|
};
|
|
@@ -190,6 +228,21 @@ export const VIDEO_MODELS = [
|
|
|
190
228
|
// the Max row at base rates and offers it 2K/4K it cannot render.
|
|
191
229
|
'minimax-h3',
|
|
192
230
|
'minimax-h3-max',
|
|
231
|
+
// LTX-2.5, two seats in one family (2026-08-29). Base is the VOLUME seat —
|
|
232
|
+
// cheapest native 1080p second we sell, free native audio at every tier, the
|
|
233
|
+
// only row reaching 1440p, and the longest clips in the catalogue (20s).
|
|
234
|
+
// Pro is the fidelity seat and is NOT a superset: shorter ladder (no
|
|
235
|
+
// 1440p/4K), shorter clips (6/8/10 only) and ~1/3 dearer.
|
|
236
|
+
//
|
|
237
|
+
// NEVER PREFIX-MATCH: 'ltx-2-5-pro' starts with 'ltx-2-5'. A prefix test
|
|
238
|
+
// bills Pro at base rates AND offers it 1440p/4K and 12-20s durations it
|
|
239
|
+
// cannot render — the same trap as the MiniMax pair, one row worse.
|
|
240
|
+
//
|
|
241
|
+
// Durations are DISCRETE AND EVEN (6,8,10,12,14,16,18,20); the union bounds
|
|
242
|
+
// below stay 3-30 because other rows are wider, so `assertVideoCapabilities`
|
|
243
|
+
// is what refuses an odd second. It reads `values`, not just min/max.
|
|
244
|
+
'ltx-2-5',
|
|
245
|
+
'ltx-2-5-pro',
|
|
193
246
|
];
|
|
194
247
|
// ── Capability-derived param vocabulary + guard ─────────────────
|
|
195
248
|
//
|
|
@@ -230,6 +283,17 @@ const VIDEO_RESOLUTION_VOCAB = VIDEO_RESOLUTIONS;
|
|
|
230
283
|
const MINIMAX_MODELS = new Set(['minimax-h3', 'minimax-h3-max']);
|
|
231
284
|
/** Reference images fal does not charge for. */
|
|
232
285
|
const MINIMAX_FREE_REF_IMAGES = 5;
|
|
286
|
+
/**
|
|
287
|
+
* The LTX-2.5 pair. A SET, not a prefix test — `ltx-2-5-pro` starts with
|
|
288
|
+
* `ltx-2-5`, and the two rows differ on ladder, duration list AND price.
|
|
289
|
+
*
|
|
290
|
+
* Their cost key is the plainest shape in the file — `{model}-{res}-{N}s` with
|
|
291
|
+
* no suffix ever, because LTX has no paid option: native audio is included at
|
|
292
|
+
* every tier (so no `-audio` variant like Kling) and there is no reference
|
|
293
|
+
* endpoint at all (so no `-ref{K}` variant like H3). Mirrors `ltxCreditKey()`
|
|
294
|
+
* in slate/src/shared/pricing.ts.
|
|
295
|
+
*/
|
|
296
|
+
const LTX_MODELS = new Set(['ltx-2-5', 'ltx-2-5-pro']);
|
|
233
297
|
/** K for the `-ref{K}` suffix: images past the free five, capped by the model's
|
|
234
298
|
* own declared ceiling. 0 for h3-max (no reference transport) and for anything
|
|
235
299
|
* that is not a MiniMax row. Mirrors refImageSurchargeCount() in
|
|
@@ -1076,7 +1140,17 @@ function imageCostKey(model, resolution, quality = 'medium') {
|
|
|
1076
1140
|
}
|
|
1077
1141
|
export const generateImage = {
|
|
1078
1142
|
id: 'slates_generate_image',
|
|
1079
|
-
description: 'Generate an image via Slates credits
|
|
1143
|
+
description: 'Generate an image via Slates credits.\n' +
|
|
1144
|
+
// GENERATED from MODEL_FACTS — the hand-typed model list that stood here
|
|
1145
|
+
// was a third copy of the routing doctrine, and it had already gone stale
|
|
1146
|
+
// (it still described nano-banana-2-lite by a capability the param owns).
|
|
1147
|
+
`${describeRouting('image')}\n` +
|
|
1148
|
+
'Full table: the slates-model-selection skill. ' +
|
|
1149
|
+
'Pass projectId to save into a Slates project (recommended — asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use. ' +
|
|
1150
|
+
// GENERATED from the skill file's own never-use list -- the one piece of
|
|
1151
|
+
// prompting doctrine that is ALWAYS in context, because the agent has
|
|
1152
|
+
// demonstrably skipped the call that would have taught it.
|
|
1153
|
+
describeBannedTokens('image'),
|
|
1080
1154
|
input: z.object({
|
|
1081
1155
|
prompt: z.string().min(1).max(4000),
|
|
1082
1156
|
model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
|
|
@@ -1095,6 +1169,10 @@ export const generateImage = {
|
|
|
1095
1169
|
// Mirrors the cost confirm gate — defaults silently wasted credits
|
|
1096
1170
|
// (1:1 when user wanted 16:9, 4k when 1k would have done). The LLM
|
|
1097
1171
|
// is forced to ask the user or read the skill instead of guessing.
|
|
1172
|
+
// Non-blocking prompt hygiene. Computed once, reported on every exit path
|
|
1173
|
+
// that echoes a prompt -- the clarification and confirm gates are PRE-spend,
|
|
1174
|
+
// which is where a rewrite is still free.
|
|
1175
|
+
const promptWarning = bannedTokenWarning(input.prompt, 'image');
|
|
1098
1176
|
if (!input.aspectRatio || !input.resolution) {
|
|
1099
1177
|
const missing = [];
|
|
1100
1178
|
if (!input.aspectRatio)
|
|
@@ -1104,7 +1182,9 @@ export const generateImage = {
|
|
|
1104
1182
|
return ok({
|
|
1105
1183
|
requires_clarification: true,
|
|
1106
1184
|
missing,
|
|
1107
|
-
|
|
1185
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
1186
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
1187
|
+
`Missing required field(s): ${missing.join(', ')}. ` +
|
|
1108
1188
|
`Read the slates-prompting-nano-banana-2 + slates-cost-discipline skills, ` +
|
|
1109
1189
|
// Generated from MODEL_CAPABILITIES — never retype a ratio list.
|
|
1110
1190
|
`or ask the user. Aspect ratio by model: ${describeAspectRatios(IMAGE_MODELS)}. ` +
|
|
@@ -1188,7 +1268,9 @@ export const generateImage = {
|
|
|
1188
1268
|
model: costKey,
|
|
1189
1269
|
estimated_cents: totalCents,
|
|
1190
1270
|
estimated_credits: totalCents,
|
|
1191
|
-
|
|
1271
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
1272
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
1273
|
+
`Cost exceeds ${CONFIRM_CREDITS} credits. Re-call with confirm=true to proceed, or pick a smaller resolution / count.`,
|
|
1192
1274
|
});
|
|
1193
1275
|
}
|
|
1194
1276
|
const previews = await previewAssets(ctx, referenceAssetIds.map((id) => ({ id, type: 'image', role: 'reference' })));
|
|
@@ -1200,7 +1282,8 @@ export const generateImage = {
|
|
|
1200
1282
|
`Review them against your prompt — every reference's role must be labeled in the prompt text. ` +
|
|
1201
1283
|
`If the references suggest a different composition / style than the current prompt captures, REVISE the prompt before confirming. ` +
|
|
1202
1284
|
`When you talk to the user about this gen, refer to each reference by its code (e.g. "${previews[0]?.ref ?? 'IMG-A?'}") — they'll see the matching badge in the Slates gallery.` +
|
|
1203
|
-
`\n\nWhen ready, re-call slates_generate_image with confirm=true and the (possibly revised) prompt
|
|
1285
|
+
`\n\nWhen ready, re-call slates_generate_image with confirm=true and the (possibly revised) prompt.` +
|
|
1286
|
+
(promptWarning ? `\n\n${promptWarning}` : ''),
|
|
1204
1287
|
images: previews.flatMap((p) => p.images),
|
|
1205
1288
|
data: {
|
|
1206
1289
|
requires_confirm: true,
|
|
@@ -1275,15 +1358,21 @@ export const generateImage = {
|
|
|
1275
1358
|
}
|
|
1276
1359
|
const requestedCount = input.count ?? 1;
|
|
1277
1360
|
return {
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
`
|
|
1286
|
-
|
|
1361
|
+
// The pixels are ALREADY here when the disk read worked, so the
|
|
1362
|
+
// review pointer says "look at what you have", not "call another op".
|
|
1363
|
+
// Telling the agent to re-fetch an image already in its context would
|
|
1364
|
+
// buy a wasted turn and teach the wrong habit.
|
|
1365
|
+
text: `${images.length > 0 ? IMAGE_INLINE_REVIEW : IMAGE_FETCH_POINTER} ` +
|
|
1366
|
+
(promptWarning ? `${promptWarning} ` : '') +
|
|
1367
|
+
(partialFailure
|
|
1368
|
+
? `Partial result: ${assetList.length} of ${requestedCount} image(s) saved into project ${input.projectId} ` +
|
|
1369
|
+
`(error on the rest: ${result.error ?? 'unknown error'}). ` +
|
|
1370
|
+
`The saved assets are in data.assets — re-generate only the missing count, don't redo the whole batch. ` +
|
|
1371
|
+
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"`
|
|
1372
|
+
: `Generated ${assetList.length} image(s) into project ${input.projectId} ` +
|
|
1373
|
+
`for ${fmtCredits(totalCents)}. ` +
|
|
1374
|
+
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
|
|
1375
|
+
(refEcho ? ` ${refEcho}` : '')),
|
|
1287
1376
|
images,
|
|
1288
1377
|
data: {
|
|
1289
1378
|
model: imageModel,
|
|
@@ -1351,7 +1440,9 @@ export const generateImage = {
|
|
|
1351
1440
|
images.push({ data: buf.toString('base64'), mimeType: mt });
|
|
1352
1441
|
}
|
|
1353
1442
|
return {
|
|
1354
|
-
text:
|
|
1443
|
+
text: `${IMAGE_INLINE_REVIEW} ` +
|
|
1444
|
+
(promptWarning ? `${promptWarning} ` : '') +
|
|
1445
|
+
`Generated ${urls.length} image(s) for ${fmtCredits(totalCents)}. ` +
|
|
1355
1446
|
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"`,
|
|
1356
1447
|
images,
|
|
1357
1448
|
data: {
|
|
@@ -1568,6 +1659,16 @@ export function videoCostKey(input) {
|
|
|
1568
1659
|
const k = minimaxRefSurchargeCount(input.model, input.referenceImages);
|
|
1569
1660
|
return `${input.model}-${res}-${input.duration}s${k > 0 ? `-ref${k}` : ''}`;
|
|
1570
1661
|
}
|
|
1662
|
+
// LTX-2.5, both seats. EXACT-ID SET, NEVER A PREFIX — `ltx-2-5-pro` starts
|
|
1663
|
+
// with `ltx-2-5`, and a prefix match would quote base rates for the dearer
|
|
1664
|
+
// row. No suffix dimension exists: audio is free and there are no references.
|
|
1665
|
+
// The resolution default is read PER ROW (both are 1080p today, but the two
|
|
1666
|
+
// ladders differ, so a shared literal would be a latent bug the day one
|
|
1667
|
+
// moves). Mirrors ltxCreditKey() in slate/src/shared/pricing.ts.
|
|
1668
|
+
if (LTX_MODELS.has(input.model)) {
|
|
1669
|
+
const res = input.videoResolution ?? defaultVideoResolutionFor(input.model);
|
|
1670
|
+
return `${input.model}-${res}-${input.duration}s`;
|
|
1671
|
+
}
|
|
1571
1672
|
if (input.model.startsWith('seedance')) {
|
|
1572
1673
|
// Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
|
|
1573
1674
|
// × vref × res × duration). AI-face route bills the `-face-` key (~45% over
|
|
@@ -1813,6 +1914,33 @@ function resolveVideoModel(raw) {
|
|
|
1813
1914
|
}
|
|
1814
1915
|
return null;
|
|
1815
1916
|
}
|
|
1917
|
+
/**
|
|
1918
|
+
* Pair each reference-audio asset id with the words spoken in it.
|
|
1919
|
+
*
|
|
1920
|
+
* 🚨 THE MODEL RE-TRANSCRIBES A SUPPLIED TAKE. Seedance does not consume
|
|
1921
|
+
* reference audio verbatim — it re-synthesises something close to it, and a
|
|
1922
|
+
* 2026-08-28 field test heard "an app called Slates" come back as "a map called
|
|
1923
|
+
* Slates". The clip carries the voice, the accent and the timing; only text
|
|
1924
|
+
* carries the words. This is the text, and without it every generation with a
|
|
1925
|
+
* voice take has its line guessed.
|
|
1926
|
+
*
|
|
1927
|
+
* Positional in, KEYED out: the desktop route merges its deprecated singular
|
|
1928
|
+
* `audioReferenceAssetId` onto the END of the plural list, so an index would
|
|
1929
|
+
* address different clips depending on which shape the caller used. Blank
|
|
1930
|
+
* entries are dropped rather than sent as empty strings — a clip with no
|
|
1931
|
+
* speech has no line, which is not the same as a line that is empty.
|
|
1932
|
+
*/
|
|
1933
|
+
function spokenTextByAssetId(assetIds, spoken) {
|
|
1934
|
+
if (!assetIds?.length || !spoken?.length)
|
|
1935
|
+
return undefined;
|
|
1936
|
+
const out = {};
|
|
1937
|
+
assetIds.forEach((id, i) => {
|
|
1938
|
+
const text = spoken[i]?.trim();
|
|
1939
|
+
if (id && text)
|
|
1940
|
+
out[id] = text;
|
|
1941
|
+
});
|
|
1942
|
+
return Object.keys(out).length > 0 ? out : undefined;
|
|
1943
|
+
}
|
|
1816
1944
|
// Maps a video model id to its bundled prompting skill (frontmatter `name:`),
|
|
1817
1945
|
// so guidance text points at a skill that actually exists. Deriving the name
|
|
1818
1946
|
// via model.split('-')[0] produced 'slates-prompting-kling' / '...-veo', which
|
|
@@ -1839,7 +1967,10 @@ function promptingSkillFor(model) {
|
|
|
1839
1967
|
}
|
|
1840
1968
|
export const generateVideo = {
|
|
1841
1969
|
id: 'slates_generate_video',
|
|
1842
|
-
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3 / slates-prompting-minimax-h3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").'
|
|
1970
|
+
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3 / slates-prompting-minimax-h3) — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm — pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8"). ' +
|
|
1971
|
+
// GENERATED from the skill's own slop-token list. Always in context on both
|
|
1972
|
+
// surfaces, so it survives an agent that skips slates_get_prompting_guide.
|
|
1973
|
+
describeBannedTokens('video'),
|
|
1843
1974
|
input: z.object({
|
|
1844
1975
|
prompt: z.string().min(1).max(4000),
|
|
1845
1976
|
// ROUTING doctrine only. Every capability number was stripped on 2026-08-16
|
|
@@ -1848,7 +1979,14 @@ export const generateVideo = {
|
|
|
1848
1979
|
// "seedance-2.5 480p/720p" were both stated here AND there, and the two
|
|
1849
1980
|
// copies disagreed — and the second of those went stale on 2026-08-24 when
|
|
1850
1981
|
// 2.5 gained 1080p, which is exactly the failure mode generating it fixes.
|
|
1851
|
-
model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate
|
|
1982
|
+
model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate ` +
|
|
1983
|
+
`params (registry cost keys like "kling-v3-standard-8s" auto-resolve). All are VIDEO-only.\n` +
|
|
1984
|
+
// GENERATED from MODEL_FACTS. The paragraph that stood here restated it by
|
|
1985
|
+
// hand and had already drifted a phrase at a time.
|
|
1986
|
+
`${describeRouting('video', 'generate')}\n` +
|
|
1987
|
+
`Full table: the slates-model-selection skill. Each model's legal aspect ratios, durations and ` +
|
|
1988
|
+
`resolutions are in those params' own descriptions — read them there, not from memory. ` +
|
|
1989
|
+
`For per-call cost, call slates_estimate_generation_cost.`),
|
|
1852
1990
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
|
|
1853
1991
|
// 🚨 THESE THREE DESCRIPTIONS ARE GENERATED FROM `MODEL_CAPABILITIES`.
|
|
1854
1992
|
// Never hand-write a ratio, resolution or duration into them again — every
|
|
@@ -1877,6 +2015,7 @@ export const generateVideo = {
|
|
|
1877
2015
|
videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS (UUIDs or badge codes) read alongside the images and audio in the same generation — own-footage restyle, MOTION TRANSFER ("the character from image 1 performs the motion from video 1"), or dialogue conditioning. Cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; ignored elsewhere. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to combined input+output seconds (the vref key) — pass videoReferenceSecondsEach so the quote is right. If any clip contains a human/AI character, pair with seedanceFace=true (the default Seedance route blocks people). Over the cap is REFUSED, never trimmed: a dropped clip would already have been priced in.`),
|
|
1878
2016
|
videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each reference clip\'s duration in seconds (from the asset listing). Feeds the vref cost key — the bill is Σceil(each) + output seconds. The server re-derives this by probing every uploaded clip, so an understated value just gets corrected upward.'),
|
|
1879
2017
|
audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO clips (UUIDs or badge codes) read alongside the images and video — e.g. lip-sync a character to a line ("the character in image 1 speaks the dialogue from audio 1"). Cited as "audio 1", "audio 2"… in the order given. No billing surcharge (Seedance audio is included). ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
|
|
2018
|
+
audioReferenceSpokenText: z.array(z.string()).optional().describe('STRONGLY RECOMMENDED whenever a reference clip contains SPEECH. Same order and length as audioReferenceAssetIds; use "" for a clip with no words (music, ambience, room tone). The model RE-TRANSCRIBES a supplied take rather than using it verbatim — a field test heard "an app called Slates" come back as "a map called Slates" — so the audio decides the VOICE, the ACCENT and the TIMING while only text decides the WORDS. Give the exact line here and it is quoted into the prompt beside the citation. Omit it and the words are a guess. Pairs with the plural audioReferenceAssetIds; the deprecated singular audioReferenceAssetId carries no text.'),
|
|
1880
2019
|
sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
|
|
1881
2020
|
audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
|
|
1882
2021
|
generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
|
|
@@ -1911,11 +2050,16 @@ export const generateVideo = {
|
|
|
1911
2050
|
// no asset to reference later, and a failed gen leaves the user with
|
|
1912
2051
|
// nothing. The MCP-only headless path that exists for image gen is
|
|
1913
2052
|
// not reasonable for video given the cost.
|
|
2053
|
+
// Non-blocking prompt hygiene, computed once. Reported on the gates that
|
|
2054
|
+
// fire BEFORE any spend, where a rewrite is still free.
|
|
2055
|
+
const promptWarning = bannedTokenWarning(input.prompt, 'video');
|
|
1914
2056
|
if (!input.projectId) {
|
|
1915
2057
|
return ok({
|
|
1916
2058
|
requires_clarification: true,
|
|
1917
2059
|
missing: ['projectId'],
|
|
1918
|
-
|
|
2060
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
2061
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
2062
|
+
'projectId is required for video generation. Use slates_list_projects to find one or slates_create_project to make a new one. Video gens cost tens to a few hundred credits per call — they need to land in a project so the user sees the progress card and the result.',
|
|
1919
2063
|
});
|
|
1920
2064
|
}
|
|
1921
2065
|
if (!input.aspectRatio || !input.duration) {
|
|
@@ -1927,7 +2071,9 @@ export const generateVideo = {
|
|
|
1927
2071
|
return ok({
|
|
1928
2072
|
requires_clarification: true,
|
|
1929
2073
|
missing,
|
|
1930
|
-
|
|
2074
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
2075
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
2076
|
+
`Missing required field(s): ${missing.join(', ')}. ` +
|
|
1931
2077
|
`Read the slates-cost-discipline + ${promptingSkillFor(input.model)} skills, ` +
|
|
1932
2078
|
// Generated from MODEL_CAPABILITIES. The prose that stood here claimed
|
|
1933
2079
|
// "Veo locks to 16:9", "Kling/Seedance support 1:1 16:9 9:16 4:3 3:4
|
|
@@ -1973,6 +2119,34 @@ export const generateVideo = {
|
|
|
1973
2119
|
});
|
|
1974
2120
|
}
|
|
1975
2121
|
}
|
|
2122
|
+
// LTX-2.5's SHAPE constraint: FRAMES, NEVER REFERENCES. fal publishes
|
|
2123
|
+
// text-to-video and image-to-video for `lightricks/ltx-2.5` and nothing
|
|
2124
|
+
// else — there is no reference-to-video endpoint on either seat, so there
|
|
2125
|
+
// is no transport for an ingredient, character, environment, style,
|
|
2126
|
+
// reference-video or reference-audio input.
|
|
2127
|
+
//
|
|
2128
|
+
// This has to be an EXPLICIT refusal rather than a silent no-op: accepting
|
|
2129
|
+
// reference ids we cannot send is the exact "no error, no warning, no
|
|
2130
|
+
// images in the request" failure `seedance-2.5-edit` shipped with. Start
|
|
2131
|
+
// and end FRAMES are unaffected — image-to-video carries `image_url` plus
|
|
2132
|
+
// an optional `end_image_url`, which is what `features.lastFrame` records.
|
|
2133
|
+
if (LTX_MODELS.has(input.model)) {
|
|
2134
|
+
const refImages = (input.ingredientAssetIds?.length ?? 0) +
|
|
2135
|
+
(input.characterAssetIds?.length ?? 0) +
|
|
2136
|
+
(input.environmentAssetIds?.length ?? 0) +
|
|
2137
|
+
(input.styleAssetIds?.length ?? 0);
|
|
2138
|
+
const refMedia = (input.videoReferenceAssetIds?.length ?? 0) +
|
|
2139
|
+
(input.audioReferenceAssetIds?.length ?? 0) +
|
|
2140
|
+
(input.videoReferenceAssetId ? 1 : 0) +
|
|
2141
|
+
(input.audioReferenceAssetId ? 1 : 0);
|
|
2142
|
+
if (refImages > 0 || refMedia > 0) {
|
|
2143
|
+
return ok({
|
|
2144
|
+
requires_clarification: true,
|
|
2145
|
+
missing: [],
|
|
2146
|
+
message: `${input.model} takes a prompt and up to two frames (start and/or end) — it has no reference endpoint at all, so reference images, video and audio cannot be sent. Drop them, or switch to minimax-h3, which reads ${getModelCapability('minimax-h3')?.maxIngredientImages ?? 9} images plus reference video and audio.`,
|
|
2147
|
+
});
|
|
2148
|
+
}
|
|
2149
|
+
}
|
|
1976
2150
|
// MiniMax H3's SHAPE constraints. Counts and caps are read from the
|
|
1977
2151
|
// capability SSOT; only the endpoint SHAPE is stated here, because it is
|
|
1978
2152
|
// not a number the registry models: fal publishes text-to-video,
|
|
@@ -2033,6 +2207,20 @@ export const generateVideo = {
|
|
|
2033
2207
|
message: `${input.model} takes at most ${maxTotal} reference files across all modalities (you passed ${refImages + refMedia}).`,
|
|
2034
2208
|
});
|
|
2035
2209
|
}
|
|
2210
|
+
// A misaligned spoken-text array would attach one clip's line to another
|
|
2211
|
+
// and send the model the wrong words with nothing on screen to say so.
|
|
2212
|
+
// Refuse rather than truncate: the whole point of the field is that the
|
|
2213
|
+
// WORDS are exact.
|
|
2214
|
+
if (input.audioReferenceSpokenText &&
|
|
2215
|
+
input.audioReferenceSpokenText.length !== (input.audioReferenceAssetIds?.length ?? 0)) {
|
|
2216
|
+
return ok({
|
|
2217
|
+
requires_clarification: true,
|
|
2218
|
+
missing: [],
|
|
2219
|
+
message: `audioReferenceSpokenText must be the same length as audioReferenceAssetIds ` +
|
|
2220
|
+
`(${input.audioReferenceSpokenText.length} vs ${input.audioReferenceAssetIds?.length ?? 0}) ` +
|
|
2221
|
+
`— it pairs by position. Use "" for a clip with no speech.`,
|
|
2222
|
+
});
|
|
2223
|
+
}
|
|
2036
2224
|
// fal: "Audio cannot be the only reference input; provide at least one
|
|
2037
2225
|
// reference image or video with it."
|
|
2038
2226
|
const audioRefs = (input.audioReferenceAssetIds?.length ?? 0) + (input.audioReferenceAssetId ? 1 : 0);
|
|
@@ -2225,6 +2413,7 @@ export const generateVideo = {
|
|
|
2225
2413
|
text: `Pre-flight for ${input.duration}s ${input.model} (${costKey}): ` +
|
|
2226
2414
|
`${fmtCredits(totalCents)}.` +
|
|
2227
2415
|
refSummary +
|
|
2416
|
+
(promptWarning ? `\n\n${promptWarning}` : '') +
|
|
2228
2417
|
`\n\nWhen ready, re-call slates_generate_video with confirm=true and the (possibly revised) prompt.`,
|
|
2229
2418
|
images: refImages,
|
|
2230
2419
|
data: {
|
|
@@ -2269,6 +2458,13 @@ export const generateVideo = {
|
|
|
2269
2458
|
audioReferenceAssetId: input.audioReferenceAssetId,
|
|
2270
2459
|
videoReferenceAssetIds: input.videoReferenceAssetIds,
|
|
2271
2460
|
audioReferenceAssetIds: input.audioReferenceAssetIds,
|
|
2461
|
+
// The route keys this by ASSET ID, not by position, because its own
|
|
2462
|
+
// singular/plural merge appends to the END of the list — an index would
|
|
2463
|
+
// mean different clips depending on which shape the caller used. The op
|
|
2464
|
+
// takes the friendlier positional array and re-keys it here, AFTER
|
|
2465
|
+
// `rids()` has resolved badge codes, so the keys are the ids the route
|
|
2466
|
+
// will resolve. A blank entry is dropped rather than sent as "".
|
|
2467
|
+
audioReferenceSpokenText: spokenTextByAssetId(input.audioReferenceAssetIds, input.audioReferenceSpokenText),
|
|
2272
2468
|
sound: input.sound,
|
|
2273
2469
|
audioLanguage: input.audioLanguage,
|
|
2274
2470
|
generateMusic: input.generateMusic,
|
|
@@ -2296,7 +2492,12 @@ export const generateVideo = {
|
|
|
2296
2492
|
}, refEcho);
|
|
2297
2493
|
}
|
|
2298
2494
|
return {
|
|
2299
|
-
text:
|
|
2495
|
+
text:
|
|
2496
|
+
// No frames come back with a video generation, so unlike the image op
|
|
2497
|
+
// this one has to name the op that fetches them.
|
|
2498
|
+
`${VIDEO_REVIEW_POINTER} ` +
|
|
2499
|
+
(promptWarning ? `${promptWarning} ` : '') +
|
|
2500
|
+
`Generated ${input.duration}s ${input.model} video into project ${input.projectId} ` +
|
|
2300
2501
|
`for ${fmtCredits(totalCents)}. ` +
|
|
2301
2502
|
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
|
|
2302
2503
|
(refEcho ? ` ${refEcho}` : ''),
|
|
@@ -2325,7 +2526,10 @@ export const generateAudio = {
|
|
|
2325
2526
|
projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
|
|
2326
2527
|
model: z
|
|
2327
2528
|
.enum(AUDIO_MODELS)
|
|
2328
|
-
.describe(
|
|
2529
|
+
.describe(
|
|
2530
|
+
// GENERATED from MODEL_FACTS — the audio lane's routing lived only in the
|
|
2531
|
+
// system prompt and in a one-line paraphrase here.
|
|
2532
|
+
`Audio surface.\n${describeRouting('audio')}\nFull table: the slates-model-selection skill.`),
|
|
2329
2533
|
prompt: z
|
|
2330
2534
|
.string()
|
|
2331
2535
|
.min(1)
|
|
@@ -2693,7 +2897,13 @@ export const editVideo = {
|
|
|
2693
2897
|
sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
|
|
2694
2898
|
prompt: z.string().min(1).max(2500).describe('The change, not the whole scene — e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
|
|
2695
2899
|
// Clip windows and resolutions GENERATED from MODEL_CAPABILITIES.
|
|
2696
|
-
model: zEnum(EDIT_VIDEO_MODELS).optional().describe(
|
|
2900
|
+
model: zEnum(EDIT_VIDEO_MODELS).optional().describe(
|
|
2901
|
+
// Routing GENERATED from MODEL_FACTS; the paragraph that stood here was
|
|
2902
|
+
// hand-written and carried per-second prices, which drift and which the
|
|
2903
|
+
// agent's own REAL NUMBERS ONLY rule forbids it repeating.
|
|
2904
|
+
`Default kling-v3.0-omni-edit. Neither omni-flash-edit nor seedance-2.5-edit takes character/style refs on this op.\n` +
|
|
2905
|
+
`${describeRouting('video', 'edit')}\n` +
|
|
2906
|
+
`Source-clip length per engine: ${describeDurations(EDIT_VIDEO_MODELS)}. Output resolution: ${describeVideoResolutions(EDIT_VIDEO_MODELS)}`),
|
|
2697
2907
|
characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2698
2908
|
styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2699
2909
|
keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
|
|
@@ -3472,6 +3682,33 @@ function resolveGuideTopic(topic) {
|
|
|
3472
3682
|
if (t === 'slates-character-turnaround' || t === 'character-turnaround') {
|
|
3473
3683
|
return 'slates-character-identity';
|
|
3474
3684
|
}
|
|
3685
|
+
// ⚠️ Previs aliases sit HIGH, before the model-prefix rules below. `dialogue`
|
|
3686
|
+
// already resolves to the Seed Audio guide, and `style`/`camera` words are a
|
|
3687
|
+
// hair away from the style-prompting and model blocks — anchoring these here
|
|
3688
|
+
// keeps a previs ask out of an audio guide.
|
|
3689
|
+
if (t === 'previs' ||
|
|
3690
|
+
t === 'pre-vis' ||
|
|
3691
|
+
t === 'previz' ||
|
|
3692
|
+
t === 'blocking' ||
|
|
3693
|
+
t === 'blockout' ||
|
|
3694
|
+
t === 'greybox' ||
|
|
3695
|
+
t === 'grey-box' ||
|
|
3696
|
+
t === 'graybox' ||
|
|
3697
|
+
t === 'blender') {
|
|
3698
|
+
return 'slates-previs-blocking';
|
|
3699
|
+
}
|
|
3700
|
+
if (t === 'camera' || t === 'camera-moves' || t === 'camera moves' || t === 'shot-list' || t === 'shot list') {
|
|
3701
|
+
return 'slates-camera-language';
|
|
3702
|
+
}
|
|
3703
|
+
if (t === 'blocking-to-prompt' || t === 'previs-prompt' || t === 'reference-video' || t === 'video-to-video' || t === 'v2v') {
|
|
3704
|
+
return 'slates-blocking-to-prompt';
|
|
3705
|
+
}
|
|
3706
|
+
if (t === 'dialogue-blocking' || t === 'dialogue blocking' || t === '180-rule' || t === 'eyelines' || t === 'seating') {
|
|
3707
|
+
return 'slates-dialogue-blocking';
|
|
3708
|
+
}
|
|
3709
|
+
if (t === 'restyle' || t === 're-style' || t === 'style-swap' || t === 'style swap' || t === 'style-variants') {
|
|
3710
|
+
return 'slates-restyle-from-blocking';
|
|
3711
|
+
}
|
|
3475
3712
|
if (t === 'model-selection' ||
|
|
3476
3713
|
t === 'model selection' ||
|
|
3477
3714
|
t === 'which-model' ||
|
|
@@ -3501,6 +3738,12 @@ function resolveGuideTopic(topic) {
|
|
|
3501
3738
|
t.startsWith('h3-')) {
|
|
3502
3739
|
return 'slates-prompting-minimax-h3';
|
|
3503
3740
|
}
|
|
3741
|
+
// LTX-2.5 — both seats share one skill. A PREFIX is right HERE (unlike every
|
|
3742
|
+
// rate, key and endpoint lookup, which must be exact) precisely because both
|
|
3743
|
+
// rows resolve to the same guide: `ltx-2-5-pro` matching the `ltx` prefix is
|
|
3744
|
+
// the intended outcome, not a collision.
|
|
3745
|
+
if (t.startsWith('ltx') || t === 'lightricks')
|
|
3746
|
+
return 'slates-prompting-ltx-2-5';
|
|
3504
3747
|
if (t.startsWith('kling-mc'))
|
|
3505
3748
|
return 'slates-prompting-motion-transfer';
|
|
3506
3749
|
if (t === 'edit-video' || t === 'video-edit' || t === 'edit video' || t === 'video edit')
|
|
@@ -3553,6 +3796,27 @@ function resolveGuideTopic(topic) {
|
|
|
3553
3796
|
}
|
|
3554
3797
|
return null;
|
|
3555
3798
|
}
|
|
3799
|
+
/**
|
|
3800
|
+
* The guide index, GENERATED from SKILLS.
|
|
3801
|
+
*
|
|
3802
|
+
* The list here was hand-typed and had drifted to 25 of 32 names — the prompting
|
|
3803
|
+
* guides for GPT Image 2, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
|
|
3804
|
+
* all missing, so an agent reading this description could not learn they exist.
|
|
3805
|
+
* A hand-typed index of a generated corpus is a stale index; it is only a matter
|
|
3806
|
+
* of when.
|
|
3807
|
+
*
|
|
3808
|
+
* Per-model guides are listed as BARE NAMES: the name is the description, and
|
|
3809
|
+
* `resolveGuideTopic()` resolves a model id to the right one anyway.
|
|
3810
|
+
*/
|
|
3811
|
+
function describeGuideTopics() {
|
|
3812
|
+
const names = Object.keys(SKILLS).sort();
|
|
3813
|
+
const perModel = names.filter((n) => n.startsWith('slates-prompting-'));
|
|
3814
|
+
const rest = names.filter((n) => !n.startsWith('slates-prompting-'));
|
|
3815
|
+
return (`Workflow and cross-cutting guides: ${rest.join(', ')}. ` +
|
|
3816
|
+
`Per-model prompting guides (or just pass the model id, which resolves to one of these): ` +
|
|
3817
|
+
`${perModel.join(', ')}. ` +
|
|
3818
|
+
`Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.`);
|
|
3819
|
+
}
|
|
3556
3820
|
export const getPromptingGuide = {
|
|
3557
3821
|
id: 'slates_get_prompting_guide',
|
|
3558
3822
|
description: "Return the full markdown of a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id (e.g. 'veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2') which maps to the right guide. ALWAYS read 'slates-cost-discipline' plus the relevant model guide before your first generation in a session.",
|
|
@@ -3560,7 +3824,7 @@ export const getPromptingGuide = {
|
|
|
3560
3824
|
topic: z
|
|
3561
3825
|
.string()
|
|
3562
3826
|
.min(1)
|
|
3563
|
-
.describe(
|
|
3827
|
+
.describe(`Guide name, model id, or style name. ${describeGuideTopics()}`),
|
|
3564
3828
|
}),
|
|
3565
3829
|
async run(input) {
|
|
3566
3830
|
const resolved = resolveGuideTopic(input.topic);
|
|
@@ -3574,6 +3838,159 @@ export const getPromptingGuide = {
|
|
|
3574
3838
|
};
|
|
3575
3839
|
},
|
|
3576
3840
|
};
|
|
3841
|
+
// ── Blender previs ──────────────────────────────────────────────
|
|
3842
|
+
//
|
|
3843
|
+
// The only ops that talk to a third transport: a localhost socket into a
|
|
3844
|
+
// running Blender carrying the Slates add-on. They exist to produce ONE
|
|
3845
|
+
// artifact — a grey-box blocking clip whose asset id goes straight into
|
|
3846
|
+
// `slates_generate_video`'s `videoReferenceAssetIds`, so the model renders a
|
|
3847
|
+
// world around a camera path instead of inventing one.
|
|
3848
|
+
//
|
|
3849
|
+
// 🚨 There is deliberately no camera-move library here. Camera work is written
|
|
3850
|
+
// as `bpy` by the agent through `slates_blender_execute`, against the Blender
|
|
3851
|
+
// API reference the add-on ships. A fixed menu of moves would cap the workflow
|
|
3852
|
+
// at whatever we thought of; code execution plus real docs does not.
|
|
3853
|
+
// The setup sentence and the download URL live in `clients/blender.ts`, beside
|
|
3854
|
+
// the port range they belong to — one home, so a moved page is one edit.
|
|
3855
|
+
const BLENDER_UNAVAILABLE_HINT = `Blender previs needs the Slates Blender add-on running. ${BLENDER_SETUP_HINT}`;
|
|
3856
|
+
export const blenderStatus = {
|
|
3857
|
+
id: 'slates_blender_status',
|
|
3858
|
+
description: 'Check whether a Blender running the Slates add-on is reachable, and if so return its scene summary (timing, camera, collection tree). Call this FIRST in any previs workflow — every other Blender op fails with the same setup message when the bridge is down, and knowing the frame range and fps up front is what keeps the blocking and the prompt timings in agreement.',
|
|
3859
|
+
input: z.object({}),
|
|
3860
|
+
async run() {
|
|
3861
|
+
const client = new BlenderBridgeClient();
|
|
3862
|
+
if (!(await client.isReachable())) {
|
|
3863
|
+
return ok({ connected: false, hint: BLENDER_UNAVAILABLE_HINT });
|
|
3864
|
+
}
|
|
3865
|
+
return ok({ connected: true, scene: await client.call('result = _mod("scene").summary()') });
|
|
3866
|
+
},
|
|
3867
|
+
};
|
|
3868
|
+
export const blenderExecute = {
|
|
3869
|
+
id: 'slates_blender_execute',
|
|
3870
|
+
description: 'Run Python (`bpy`) inside the connected Blender and return whatever the code assigns to a dict named `result`. This is how blocking gets built: primitives, empties, constraints, camera rigs, keyframes, markers. Anything Blender can do, this can do. Assign a dict to `result` to get data back (e.g. `result = {"camera": cam.name}`); print() output comes back separately as stdout. On an exception you get the full traceback — read it, fix the code, retry. Before writing an unfamiliar call, look up its real signature with slates_blender_docs rather than guessing: the add-on ships the Blender 5.1 API reference precisely so you do not have to recall it.',
|
|
3871
|
+
input: z.object({
|
|
3872
|
+
code: z
|
|
3873
|
+
.string()
|
|
3874
|
+
.min(1)
|
|
3875
|
+
.describe('Python source. Assign a JSON-serialisable dict to `result` to return data.'),
|
|
3876
|
+
timeoutSeconds: z
|
|
3877
|
+
.number()
|
|
3878
|
+
.int()
|
|
3879
|
+
.min(5)
|
|
3880
|
+
.max(900)
|
|
3881
|
+
.optional()
|
|
3882
|
+
.describe('How long to wait for Blender (default 60). Raise it for heavy geometry.'),
|
|
3883
|
+
}),
|
|
3884
|
+
async run(input) {
|
|
3885
|
+
const client = new BlenderBridgeClient();
|
|
3886
|
+
const { result, stdout, stderr } = await client.execute(input.code, {
|
|
3887
|
+
timeoutMs: input.timeoutSeconds ? input.timeoutSeconds * 1000 : undefined,
|
|
3888
|
+
});
|
|
3889
|
+
return ok({ result, ...(stdout ? { stdout } : {}), ...(stderr ? { stderr } : {}) });
|
|
3890
|
+
},
|
|
3891
|
+
};
|
|
3892
|
+
export const blenderScene = {
|
|
3893
|
+
id: 'slates_blender_scene',
|
|
3894
|
+
description: 'Scene summary from the connected Blender: frame range, fps, duration, render resolution, the active camera with its keyframe times in both frames and seconds, and the full collection/object tree with transforms and constraints. Cheap — call it freely between edits. The camera keyframe times ARE the cut structure, so read them before writing any shot-by-shot prompt.',
|
|
3895
|
+
input: z.object({}),
|
|
3896
|
+
async run() {
|
|
3897
|
+
return ok(await new BlenderBridgeClient().call('result = _mod("scene").summary()'));
|
|
3898
|
+
},
|
|
3899
|
+
};
|
|
3900
|
+
export const blenderDocs = {
|
|
3901
|
+
id: 'slates_blender_docs',
|
|
3902
|
+
description: 'Look up a dotted Blender Python API identifier in the bundled 5.1 reference — e.g. "bpy.ops.object", "bpy.types.Camera", "bpy.types.FollowPathConstraint". Pass "*" for top-level modules or "bpy.ops.*" to list a namespace. Use this instead of recalling a signature from memory: invented operator names and wrong enum values are the most common way previs code fails, and they fail silently often enough to be worth the lookup.',
|
|
3903
|
+
input: z.object({
|
|
3904
|
+
identifier: z.string().min(1).describe('Dotted identifier, or a namespace wildcard like "bpy.ops.*".'),
|
|
3905
|
+
}),
|
|
3906
|
+
async run(input) {
|
|
3907
|
+
const code = `result = _mod("docs").lookup(${JSON.stringify(input.identifier)})`;
|
|
3908
|
+
return ok(await new BlenderBridgeClient().call(code));
|
|
3909
|
+
},
|
|
3910
|
+
};
|
|
3911
|
+
export const blenderSearchDocs = {
|
|
3912
|
+
id: 'slates_blender_search_docs',
|
|
3913
|
+
description: 'Full-text search of the bundled Blender documentation for when you do not know the identifier yet. scope "api" searches the Python reference; "manual" searches the user manual for concepts and workflow ("how does Follow Path work", "bezier interpolation handles"). Use slates_blender_docs when you know the name and this when you do not.',
|
|
3914
|
+
input: z.object({
|
|
3915
|
+
query: z.string().min(2),
|
|
3916
|
+
scope: z.enum(['api', 'manual']).optional().describe('Default "api".'),
|
|
3917
|
+
maxResults: z.number().int().min(1).max(20).optional().describe('Default 8.'),
|
|
3918
|
+
}),
|
|
3919
|
+
async run(input) {
|
|
3920
|
+
const args = [
|
|
3921
|
+
JSON.stringify(input.query),
|
|
3922
|
+
JSON.stringify(input.scope ?? 'api'),
|
|
3923
|
+
String(input.maxResults ?? 8),
|
|
3924
|
+
].join(', ');
|
|
3925
|
+
return ok(await new BlenderBridgeClient().call(`result = _mod("docs").search(${args})`));
|
|
3926
|
+
},
|
|
3927
|
+
};
|
|
3928
|
+
export const blenderRenderBlocking = {
|
|
3929
|
+
id: 'slates_blender_render_blocking',
|
|
3930
|
+
description: "Render the connected Blender scene camera to a grey-box mp4 — the blocking clip that locks camera motion for generation — and, when projectId is given, import it into that Slates project as a video asset in the same call. The returned asset id goes into slates_generate_video's videoReferenceAssetIds and the returned durationSeconds into videoReferenceSecondsEach. Renders through the SCENE camera using scene render settings, never the user's viewport, so output does not depend on where they left their mouse. Untextured is correct: the clip supplies camera path and timing, the references supply the look. Colour in the blocking is NOTATION, not look — a distinct viewport colour per character is what binds a proxy to its reference image across cuts, and a marked face encodes which way a featureless proxy is facing. State every such mapping in the generation prompt AND state that the colours themselves are not inherited, or the model renders a literally red person. See the slates-previs-blocking and slates-blocking-to-prompt skills. Keep it at or under 30s — seedance-2.5 accepts reference videos up to 30s, the other reference-video models up to 15s.",
|
|
3931
|
+
input: z.object({
|
|
3932
|
+
projectId: z
|
|
3933
|
+
.string()
|
|
3934
|
+
.uuid()
|
|
3935
|
+
.optional()
|
|
3936
|
+
.describe('Import the clip into this project and return its asset. Omit to just get a file path.'),
|
|
3937
|
+
resolutionX: z.number().int().min(256).max(4096).optional().describe('Default 1920.'),
|
|
3938
|
+
resolutionY: z.number().int().min(256).max(4096).optional().describe('Default 1080.'),
|
|
3939
|
+
fps: z.number().int().min(1).max(120).optional().describe('Default 24. Match what the shot list assumes.'),
|
|
3940
|
+
frameStart: z.number().int().optional().describe("Default: the scene's own frame_start."),
|
|
3941
|
+
frameEnd: z.number().int().optional().describe("Default: the scene's own frame_end."),
|
|
3942
|
+
basename: z.string().min(1).max(64).optional().describe('Filename stem. Default "blocking".'),
|
|
3943
|
+
}),
|
|
3944
|
+
async run(input, ctx) {
|
|
3945
|
+
const kwargs = [];
|
|
3946
|
+
if (input.resolutionX !== undefined)
|
|
3947
|
+
kwargs.push(`resolution_x=${input.resolutionX}`);
|
|
3948
|
+
if (input.resolutionY !== undefined)
|
|
3949
|
+
kwargs.push(`resolution_y=${input.resolutionY}`);
|
|
3950
|
+
if (input.fps !== undefined)
|
|
3951
|
+
kwargs.push(`fps=${input.fps}`);
|
|
3952
|
+
if (input.frameStart !== undefined)
|
|
3953
|
+
kwargs.push(`frame_start=${input.frameStart}`);
|
|
3954
|
+
if (input.frameEnd !== undefined)
|
|
3955
|
+
kwargs.push(`frame_end=${input.frameEnd}`);
|
|
3956
|
+
if (input.basename !== undefined)
|
|
3957
|
+
kwargs.push(`basename=${JSON.stringify(input.basename)}`);
|
|
3958
|
+
// 🚨 THE RENDER IS DEFERRED, AND THAT IS WHY THIS IS NOT ONE LINE.
|
|
3959
|
+
// In an interactive Blender `render_blocking` INVOKES the render rather
|
|
3960
|
+
// than executing it — a synchronous animation render driven from the
|
|
3961
|
+
// bridge's own `bpy.app.timers` callback would re-enter the main loop
|
|
3962
|
+
// running it. So it hands back a `check_is_finished` callable instead of
|
|
3963
|
+
// the clip, and assigning that to `check_is_finished` is the bridge's
|
|
3964
|
+
// documented convention for "hold the socket open and answer when the job
|
|
3965
|
+
// lands" (see the add-on's `bridge/deferred.py`). The deferred path wraps
|
|
3966
|
+
// the eventual dict in the SAME `{status, result}` envelope, so everything
|
|
3967
|
+
// downstream of this call is identical either way. Headless Blender has no
|
|
3968
|
+
// job system to poll, renders synchronously, and takes the `else`.
|
|
3969
|
+
const render = (await new BlenderBridgeClient().call(`_previs_result = _mod("previs").render_blocking(${kwargs.join(', ')})\n` +
|
|
3970
|
+
'if callable(_previs_result):\n' +
|
|
3971
|
+
' check_is_finished = _previs_result\n' +
|
|
3972
|
+
'else:\n' +
|
|
3973
|
+
' result = _previs_result\n', { timeoutMs: RENDER_TIMEOUT_MS }));
|
|
3974
|
+
if (!input.projectId) {
|
|
3975
|
+
return ok({
|
|
3976
|
+
...render,
|
|
3977
|
+
next: 'Pass projectId to import this into a Slates project, or call slates_upload_reference_image with type "video".',
|
|
3978
|
+
});
|
|
3979
|
+
}
|
|
3980
|
+
// The same desktop route slates_upload_reference_image uses — the file is
|
|
3981
|
+
// probed on ingest, so duration and dimensions are known immediately.
|
|
3982
|
+
const uploaded = await ctx.desktop().post('/agent/assets/upload', {
|
|
3983
|
+
projectId: input.projectId,
|
|
3984
|
+
filePath: render.filePath,
|
|
3985
|
+
type: 'video',
|
|
3986
|
+
});
|
|
3987
|
+
return ok({
|
|
3988
|
+
...render,
|
|
3989
|
+
asset: uploaded.asset ?? uploaded,
|
|
3990
|
+
next: "Pass the asset's id in slates_generate_video videoReferenceAssetIds, with videoReferenceSecondsEach set to durationSeconds.",
|
|
3991
|
+
});
|
|
3992
|
+
},
|
|
3993
|
+
};
|
|
3577
3994
|
// ── Aggregation ─────────────────────────────────────────────────
|
|
3578
3995
|
export const ALL_OPERATIONS = [
|
|
3579
3996
|
getWorkspaceState,
|
|
@@ -3652,5 +4069,18 @@ export const ALL_OPERATIONS = [
|
|
|
3652
4069
|
batchUpdateFrames,
|
|
3653
4070
|
deleteFrame,
|
|
3654
4071
|
getPromptingGuide,
|
|
4072
|
+
// ── Blender previs, LAST and deliberately ────────────────────────────
|
|
4073
|
+
// This order is not cosmetic: `slate/src/main/studio-agent/ops.ts` maps this
|
|
4074
|
+
// array straight into the Anthropic `tools` array, and that block sits inside
|
|
4075
|
+
// the desktop Studio Agent's PROMPT-CACHED PREFIX. These six landed at the
|
|
4076
|
+
// TOP, which put a third transport nobody without Blender can reach ahead of
|
|
4077
|
+
// `slates_get_workspace_state` in every conversation the app has. They are a
|
|
4078
|
+
// niche lane off the end of the surface, and the list should read that way.
|
|
4079
|
+
blenderStatus,
|
|
4080
|
+
blenderExecute,
|
|
4081
|
+
blenderScene,
|
|
4082
|
+
blenderDocs,
|
|
4083
|
+
blenderSearchDocs,
|
|
4084
|
+
blenderRenderBlocking,
|
|
3655
4085
|
];
|
|
3656
4086
|
//# sourceMappingURL=index.js.map
|