@slatesvideo/shared 0.6.2 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api-url.d.ts +9 -0
- package/dist/api-url.js +9 -0
- package/dist/auth.d.ts +13 -1
- package/dist/auth.js +9 -5
- package/dist/clients/cloud.d.ts +3 -0
- package/dist/clients/cloud.js +34 -3
- package/dist/clients/desktop.js +3 -0
- package/dist/index.d.ts +8 -2
- package/dist/index.js +44 -1
- package/dist/operations/index.d.ts +243 -31
- package/dist/operations/index.js +1483 -154
- package/dist/operations/surface.d.ts +69 -0
- package/dist/operations/surface.js +227 -0
- package/dist/prompts/agent-doctrine.d.ts +36 -0
- package/dist/prompts/agent-doctrine.js +201 -0
- package/dist/prompts/asset-label.d.ts +23 -0
- package/dist/prompts/asset-label.js +70 -0
- package/dist/prompts/banned-tokens.d.ts +40 -0
- package/dist/prompts/banned-tokens.js +219 -0
- package/dist/prompts/character-sheet.d.ts +0 -2
- package/dist/prompts/character-sheet.js +0 -2
- package/dist/prompts/craft-cards.d.ts +20 -0
- package/dist/prompts/craft-cards.js +82 -0
- package/dist/prompts/environment-sheet.js +16 -0
- package/dist/prompts/index.d.ts +1 -0
- package/dist/prompts/index.js +4 -0
- package/dist/prompts/model-capabilities.d.ts +65 -1
- package/dist/prompts/model-capabilities.js +139 -2
- package/dist/prompts/model-facts.d.ts +20 -4
- package/dist/prompts/model-facts.js +95 -27
- package/dist/prompts/partials.generated.js +2 -1
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +123 -0
- package/dist/prompts/reference-composer.d.ts +36 -7
- package/dist/prompts/reference-composer.js +75 -20
- package/dist/prompts/reference-rules.d.ts +15 -26
- package/dist/prompts/reference-rules.js +15 -93
- package/dist/prompts/shot-grammar.d.ts +154 -0
- package/dist/prompts/shot-grammar.js +184 -0
- package/dist/prompts/shot-spec.d.ts +265 -0
- package/dist/prompts/shot-spec.js +303 -0
- package/dist/skills/content.js +25 -22
- package/exports/slates-prompt-builder/generated/SKILL.md +3 -3
- package/exports/slates-prompt-builder/generated/reference-content-policy.md +6 -0
- package/exports/slates-prompt-builder/generated/reference-kling.md +22 -0
- package/exports/slates-prompt-builder/generated/reference-nano-banana.md +17 -0
- package/exports/slates-prompt-builder/generated/reference-seedance.md +19 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +17 -17
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +83 -73
- package/skills/_partials/decision-log.md +5 -4
- package/skills/_partials/thresholds.md +19 -0
- package/skills/slates-content-policy.md +15 -1
- package/skills/slates-cost-discipline.md +26 -4
- package/skills/slates-model-selection.md +2 -2
- package/skills/slates-one-prompt-film.md +20 -12
- package/skills/slates-project-organization.md +1 -1
- package/skills/slates-prompting-elevenlabs.md +61 -2
- package/skills/slates-prompting-flux-2-max.md +39 -0
- package/skills/slates-prompting-gpt-image-2.md +109 -70
- package/skills/slates-prompting-inworld-tts.md +166 -0
- package/skills/slates-prompting-kling-v3.md +39 -0
- package/skills/slates-prompting-lip-sync.md +38 -0
- package/skills/slates-prompting-ltx-2-5.md +218 -0
- package/skills/slates-prompting-minimax-h3.md +39 -0
- package/skills/slates-prompting-motion-transfer.md +38 -0
- package/skills/slates-prompting-nano-banana-2.md +36 -0
- package/skills/slates-prompting-omni-flash.md +41 -0
- package/skills/slates-prompting-seed-audio.md +38 -0
- package/skills/slates-prompting-seedance-2-5.md +38 -0
- package/skills/slates-prompting-seedance.md +36 -1
- package/skills/slates-prompting-seedream-5-lite.md +38 -0
- package/skills/slates-prompting-veo-3.md +39 -0
- package/skills/slates-shot-variety.md +53 -0
- package/skills/slates-storyboard-from-script.md +31 -15
- package/skills/slates-style-prompting.md +1 -1
- package/skills/slates-vision-feedback-loop.md +1 -1
package/dist/operations/index.js
CHANGED
|
@@ -16,7 +16,12 @@ import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../c
|
|
|
16
16
|
import { SKILLS } from '../skills/content.js';
|
|
17
17
|
// Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
|
|
18
18
|
// "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
|
|
19
|
-
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords
|
|
19
|
+
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
|
|
20
|
+
// 🚨 THE routing renderer. Routing prose is GENERATED here, never typed:
|
|
21
|
+
// slates-mcp/CLAUDE.md forbids restating it in an op description, and this
|
|
22
|
+
// file did it anyway for 1,282 characters that repeated MODEL_FACTS phrase
|
|
23
|
+
// for phrase. Edit model-facts.ts; both surfaces follow.
|
|
24
|
+
describeRouting, } from '../prompts/model-facts.js';
|
|
20
25
|
// 🚨 MODEL CAPABILITY SSOT. Aspect ratios, video resolutions and durations are
|
|
21
26
|
// VALIDATED against and DESCRIBED from `MODEL_CAPABILITIES` — this file must
|
|
22
27
|
// never re-state one of those constraints as a literal enum or a sentence
|
|
@@ -27,12 +32,48 @@ import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } fr
|
|
|
27
32
|
// 1080p. A customer burned a round trip on `4:5` + Seedance: accepted here,
|
|
28
33
|
// queued, credits reserved, rejected by the provider asynchronously.
|
|
29
34
|
import { AGENT_ROUTE_PROVIDER, aspectRatioUnion, videoResolutionUnion, durationBounds, aspectRatiosFor, getModelCapability, defaultVideoResolutionFor, checkAspectRatio, checkVideoResolution, checkDuration, describeAspectRatios, describeVideoResolutions, describeDurations, describeReferenceImageCaps, } from '../prompts/model-capabilities.js';
|
|
35
|
+
// 🚨 "LOAD THE GUIDE" MADE STRUCTURAL. The never-use token lists are EXTRACTED
|
|
36
|
+
// from the skill files (between `@banned` markers) and inlined into the two
|
|
37
|
+
// generate ops' descriptions, which are always in context on both surfaces —
|
|
38
|
+
// no call to skip, no discretion. `bannedTokenWarning` then reports what the
|
|
39
|
+
// submitted prompt actually contained, in the result, without blocking it.
|
|
40
|
+
// Never hand-type one of these tokens here; edit the skill.
|
|
41
|
+
import { describeBannedTokens, bannedTokenWarning, describeBannedTokensForSkill, } from '../prompts/banned-tokens.js';
|
|
42
|
+
// 🚨 THE OTHER HALF OF THE SAME LESSON. The banned list is the NEGATIVE half and
|
|
43
|
+
// it rides an op description. The CRAFT CARD is the POSITIVE half — the levers
|
|
44
|
+
// that make a shot good rather than merely un-bad — and it rides the estimate
|
|
45
|
+
// RESULT, which the doctrine already makes the agent call before generating.
|
|
46
|
+
// Zero prefix bytes, present at the moment the model has just been named.
|
|
47
|
+
import { describeCraftCard } from '../prompts/craft-cards.js';
|
|
48
|
+
// ONE captioning rule for a generated asset, shared with the desktop gallery.
|
|
49
|
+
import { assetCaption } from '../prompts/asset-label.js';
|
|
50
|
+
// 🚨 THE ONE ROLE LIST + the Shot shape, mirrored into the desktop. The op
|
|
51
|
+
// surface DERIVES its per-role params from these rather than naming roles by
|
|
52
|
+
// hand — a hand-typed bucket name is what `attachmentRoles` exists to kill.
|
|
53
|
+
import { ORDERED_ATTACHMENT_ROLES, ATTACHMENT_ROLE_DESCRIPTION, SCRIPT_FIELD_DESCRIPTION, SCRIPT_TEXT_FIELDS, } from '../prompts/shot-spec.js';
|
|
54
|
+
import { SHOT_SIZE_BUCKETS, CAMERA_MOVE_BUCKETS, SPEECH_RATE, } from '../prompts/shot-grammar.js';
|
|
55
|
+
// Annotations, tiers and the ONE schema renderer. Declared next door so this
|
|
56
|
+
// module never hand-sets a hint or a tier per op: `annotate()` derives all four
|
|
57
|
+
// from the id and the lockstep check re-derives them from the transport verbs.
|
|
58
|
+
import { annotate, groupFor, tierFor, toolDefinitions, GROUP_SUMMARY, OPERATION_GROUPS, } from './surface.js';
|
|
30
59
|
export function defaultContext() {
|
|
31
60
|
return {
|
|
32
61
|
cloud: () => new SlatesCloudClient(),
|
|
33
62
|
desktop: () => new SlatesDesktopClient(),
|
|
34
63
|
};
|
|
35
64
|
}
|
|
65
|
+
/** Thrown by an op that noticed `ctx.signal` had aborted mid-work. */
|
|
66
|
+
export class OperationCancelledError extends Error {
|
|
67
|
+
code = 'OPERATION_CANCELLED';
|
|
68
|
+
constructor(detail) {
|
|
69
|
+
super(`Cancelled by the user${detail ? ` — ${detail}` : ''}.`);
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
/** Throw if the caller has cancelled. Call between billable items, never inside one. */
|
|
73
|
+
function throwIfCancelled(ctx, detail) {
|
|
74
|
+
if (ctx.signal?.aborted)
|
|
75
|
+
throw new OperationCancelledError(detail);
|
|
76
|
+
}
|
|
36
77
|
// ── Helpers ─────────────────────────────────────────────────────
|
|
37
78
|
/**
|
|
38
79
|
* `z.enum` over a GENERATED list. Zod's signature wants a non-empty tuple
|
|
@@ -60,8 +101,88 @@ function ok(data, text) {
|
|
|
60
101
|
// the same credit value); balances the same. Read via creditCost(); display
|
|
61
102
|
// via fmtCredits(). The confirm gate fires above CONFIRM_CREDITS (≈ the old
|
|
62
103
|
// $0.50 gate at the 3¢/credit peg).
|
|
63
|
-
const CONFIRM_CREDITS = 17;
|
|
104
|
+
export const CONFIRM_CREDITS = 17;
|
|
64
105
|
const CENTS_PER_CREDIT = 3; // peg: 1 credit = 3¢ billed = 2¢ COGS (mirror of slates-api)
|
|
106
|
+
/**
|
|
107
|
+
* The desktop deviation guard's ceiling multiplier: the Studio Agent pauses and
|
|
108
|
+
* re-asks when projected generation spend exceeds the approved ledger by more
|
|
109
|
+
* than this factor.
|
|
110
|
+
*
|
|
111
|
+
* 🚨 EXPORTED BECAUSE THE NUMBER IS QUOTED DOWNSTREAM. `slates-cost-discipline`
|
|
112
|
+
* told the agent the threshold was 25% while `loop.ts` paused at 20% — a skill
|
|
113
|
+
* contradicting the code on the one number that decides whether a run stops.
|
|
114
|
+
* The skill now renders it from here through `_partials/thresholds.md`; the
|
|
115
|
+
* desktop loop imports it instead of declaring its own.
|
|
116
|
+
*/
|
|
117
|
+
export const DEVIATION_FACTOR = 1.2;
|
|
118
|
+
/**
|
|
119
|
+
* The confirm-gate sentence, rendered ONCE from CONFIRM_CREDITS.
|
|
120
|
+
*
|
|
121
|
+
* It was hand-typed three different ways for the same constant — "$0.50" on the
|
|
122
|
+
* image and video ops, "17 credits" on audio — so a rate change would have had
|
|
123
|
+
* to find three wordings. Byte-stable (a template over a literal), so the
|
|
124
|
+
* desktop's prompt-cached prefix is unaffected.
|
|
125
|
+
*/
|
|
126
|
+
const CONFIRM_GATE_SENTENCE = `Cost above ${CONFIRM_CREDITS} credits returns requires_confirm — pass confirm=true after explicit user OK.`;
|
|
127
|
+
// Declared HERE, above every op, because `slates_estimate_generation_cost`
|
|
128
|
+
// renders them into its `duration` description at MODULE LOAD — a const
|
|
129
|
+
// declared below the first schema that reads it is a temporal-dead-zone
|
|
130
|
+
// crash, not a lint nit (the same rule the VIDEO_MODELS block states).
|
|
131
|
+
/**
|
|
132
|
+
* Per-surface bounds and defaults.
|
|
133
|
+
*
|
|
134
|
+
* 🚨 THESE FOUR NUMBERS PER SURFACE LIVE IN THREE REPOS. A change is a
|
|
135
|
+
* three-site edit, every time:
|
|
136
|
+
* 1. HERE (`audioCostKey`, the agent's pre-flight quote)
|
|
137
|
+
* 2. `slate/src/shared/pricing.ts` → MODEL_REGISTRY `audio.durationSeconds`
|
|
138
|
+
* (min/max/default), read by `clampAudioDuration` + `audioCreditKey`
|
|
139
|
+
* 3. `slates-api/src/lib/audio-keys.ts` → the server's fail-closed bounds
|
|
140
|
+
* `slates-api/scripts/pricing-consistency-check.mjs` §4 asserts 1 and 2 agree
|
|
141
|
+
* at EVERY value including out-of-range ones; the gate check covers 3.
|
|
142
|
+
*
|
|
143
|
+
* The MINs used to be missing here, and the clamp floor was a hardcoded 1. That
|
|
144
|
+
* made `slates_estimate_generation_cost({model:'seed-audio', duration:2})`
|
|
145
|
+
* quote a real `seed-audio-2s` price for a generation the desktop would bill as
|
|
146
|
+
* 3s and the proxy would REJECT outright. Same for an omitted duration, which
|
|
147
|
+
* quoted `seed-audio-1s` against the desktop's `seed-audio-15s`.
|
|
148
|
+
*/
|
|
149
|
+
export const SEED_AUDIO_MIN_SECONDS = 3;
|
|
150
|
+
export const SEED_AUDIO_MAX_SECONDS = 120;
|
|
151
|
+
export const SEED_AUDIO_DEFAULT_SECONDS = 15;
|
|
152
|
+
export const ELEVEN_SFX_MIN_SECONDS = 1;
|
|
153
|
+
export const ELEVEN_SFX_MAX_SECONDS = 22;
|
|
154
|
+
export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
|
|
155
|
+
// ── The TTS character bucket ────────────────────────────────────────────────
|
|
156
|
+
//
|
|
157
|
+
// TTS is the ONE audio surface that does not bill per second: speech length
|
|
158
|
+
// falls out of the text, so there is no duration to charge for. It bills on a
|
|
159
|
+
// bucket of CHARACTERS instead, and the bucket exists because of the minimum
|
|
160
|
+
// billable floor rather than for tidiness.
|
|
161
|
+
//
|
|
162
|
+
// At $20.8/M characters and a 1.5× markup, a 200-character line is $0.006 of
|
|
163
|
+
// basis — under `MIN_AUDIO_BILLABLE_DOLLARS` ($0.01), so it bills the floor.
|
|
164
|
+
// The floor stops biting at 321 characters. A bucket SMALLER than that would be
|
|
165
|
+
// entirely floor-bound (every bucket the same price, so the displayed rate stops
|
|
166
|
+
// tracking cost and becomes a lie); a much LARGER one over-bills the short lines
|
|
167
|
+
// this feature is mostly for. 250 splits that difference and divides 2,000
|
|
168
|
+
// exactly, giving eight buckets and no ragged last one.
|
|
169
|
+
//
|
|
170
|
+
// 🚨 THE CAP IS READ FROM THE CAPABILITY SSOT, NEVER TYPED. 2,000 is the
|
|
171
|
+
// vendor's MEASURED limit (the API rejects 2,001 by name), and this module
|
|
172
|
+
// already imports MODEL_CAPABILITIES — so typing the number here would be a
|
|
173
|
+
// second copy of a fact one file away, which is exactly what that SSOT exists
|
|
174
|
+
// to delete. `getModelCapability` throws nothing on a miss, so the `??` would
|
|
175
|
+
// hide a renamed model: assert instead, at module load, where it fails the
|
|
176
|
+
// build rather than shipping a bucket ceiling of `undefined`.
|
|
177
|
+
export const TTS_MODEL = 'inworld-tts-2';
|
|
178
|
+
export const TTS_BUCKET_CHARS = 250;
|
|
179
|
+
export const TTS_MAX_CHARACTERS = (() => {
|
|
180
|
+
const max = getModelCapability(TTS_MODEL)?.maxCharacters;
|
|
181
|
+
if (!max)
|
|
182
|
+
throw new Error(`MODEL_CAPABILITIES['${TTS_MODEL}'] must declare maxCharacters`);
|
|
183
|
+
return max;
|
|
184
|
+
})();
|
|
185
|
+
export const TTS_BUCKET_COUNT = TTS_MAX_CHARACTERS / TTS_BUCKET_CHARS; // 8
|
|
65
186
|
function creditCost(m) {
|
|
66
187
|
if (!m)
|
|
67
188
|
return 0;
|
|
@@ -77,9 +198,34 @@ function creditsFromDollars(dollars) {
|
|
|
77
198
|
const cents = Math.round(dollars * 100);
|
|
78
199
|
return cents <= 0 ? 0 : Math.max(1, Math.ceil(cents / CENTS_PER_CREDIT));
|
|
79
200
|
}
|
|
80
|
-
// Shared describe-text for the background flag on every generate_* op.
|
|
81
|
-
|
|
82
|
-
|
|
201
|
+
// Shared describe-text for the background flag on every generate_* op. ONE
|
|
202
|
+
// sentence: it is repeated verbatim on seven ops, so every word costs seven
|
|
203
|
+
// times, and `slates_get_generation_status` explains the polling itself.
|
|
204
|
+
const BACKGROUND_DESCRIBE = 'Return generationId(s) immediately instead of blocking; poll slates_get_generation_status. Recommended for video.';
|
|
205
|
+
// ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
|
|
206
|
+
//
|
|
207
|
+
// QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
|
|
208
|
+
// pre-condition can. What it CAN have is a result the agent cannot avoid
|
|
209
|
+
// reading, naming the exact op. Deliberately NOT an auto-fetch: that would
|
|
210
|
+
// spend vision tokens on every generation whether review was wanted or not,
|
|
211
|
+
// and the sandbox doctrine says the tool is available, not mandatory.
|
|
212
|
+
//
|
|
213
|
+
// ⚠️ These three differ because what the agent already HAS differs, and telling
|
|
214
|
+
// it to re-fetch something already in front of it burns a turn for nothing:
|
|
215
|
+
// a blocking image generation returns the pixels inline, a video generation
|
|
216
|
+
// returns none, and a background submission has no asset yet.
|
|
217
|
+
const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it against the brief before you describe it. ' +
|
|
218
|
+
'Any claim about how it LOOKS must come from pixels you actually received (slates-vision-feedback-loop).';
|
|
219
|
+
const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
|
|
220
|
+
'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
|
|
221
|
+
const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
|
|
222
|
+
'slates_get_asset_video_frames for video.';
|
|
223
|
+
// The image saved, but reading it back off disk failed (best-effort fetch). The
|
|
224
|
+
// agent has an asset and NO pixels, which is the one state where a quality
|
|
225
|
+
// claim would be pure invention — so this branch has to say so rather than
|
|
226
|
+
// fall through to no pointer at all.
|
|
227
|
+
const IMAGE_FETCH_POINTER = 'The pixels could not be attached to this result: call slates_get_asset_image on the asset id ' +
|
|
228
|
+
'above before describing how it looks.';
|
|
83
229
|
// Early-return shape when a generation route accepted the job in background
|
|
84
230
|
// mode ({ background: true } in the response). No inline-image fetch — the
|
|
85
231
|
// asset doesn't exist yet; the poller delivers it on completion.
|
|
@@ -88,7 +234,8 @@ function backgroundSubmitted(kind, ids, extra, note) {
|
|
|
88
234
|
return {
|
|
89
235
|
text: `Submitted ${kind} in the background — generationId(s): ${idText}. ` +
|
|
90
236
|
`Call slates_get_generation_status with waitSeconds: 45 (it long-polls and returns on completion — ` +
|
|
91
|
-
`never a rapid loop; video renders commonly take 1-5 minutes). Generations survive app restarts
|
|
237
|
+
`never a rapid loop; video renders commonly take 1-5 minutes). Generations survive app restarts. ` +
|
|
238
|
+
BACKGROUND_REVIEW_POINTER +
|
|
92
239
|
(note ? ` ${note}` : ''),
|
|
93
240
|
data: { generationIds: ids, status: 'processing', ...extra },
|
|
94
241
|
};
|
|
@@ -96,17 +243,40 @@ function backgroundSubmitted(kind, ids, extra, note) {
|
|
|
96
243
|
// ── Workspace + identity ────────────────────────────────────────
|
|
97
244
|
export const getWorkspaceState = {
|
|
98
245
|
id: 'slates_get_workspace_state',
|
|
99
|
-
description: 'Snapshot of the user\'s Slates workspace:
|
|
100
|
-
input: z.object({
|
|
246
|
+
description: 'Snapshot of the user\'s Slates workspace: the project list (most recent first) plus the active project in full when you name one. Call once at the start of a workflow to seed your understanding.',
|
|
247
|
+
input: z.object({
|
|
248
|
+
projectId: z.string().optional(),
|
|
249
|
+
limit: z.number().int().min(1).max(200).optional().describe('How many projects to list, newest first. Default 40.'),
|
|
250
|
+
}),
|
|
101
251
|
async run(input, ctx) {
|
|
102
252
|
const desktop = ctx.desktop();
|
|
103
253
|
const { projects } = await desktop.get('/agent/projects');
|
|
254
|
+
// 🚨 COMPACT ROWS, AND A CAP. This returned every project's FULL row — fine
|
|
255
|
+
// at fourteen projects and unbounded by construction, on the one op the
|
|
256
|
+
// doctrine tells the agent to call first in every session. The fields kept
|
|
257
|
+
// are the ones a routing decision actually uses; the full row is one
|
|
258
|
+
// slates_get_project away.
|
|
259
|
+
const limit = input.limit ?? 40;
|
|
260
|
+
const rows = (projects ?? []);
|
|
261
|
+
const compact = rows.slice(0, limit).map((p) => ({
|
|
262
|
+
id: p.id,
|
|
263
|
+
name: p.name,
|
|
264
|
+
asset_count: p.assetCount ?? p.asset_count ?? undefined,
|
|
265
|
+
updated_at: p.updatedAt ?? p.updated_at ?? undefined,
|
|
266
|
+
}));
|
|
104
267
|
let activeProject = undefined;
|
|
105
268
|
if (input.projectId) {
|
|
106
269
|
const r = await desktop.get('/agent/projects/get', { id: input.projectId });
|
|
107
270
|
activeProject = r.project;
|
|
108
271
|
}
|
|
109
|
-
return ok({
|
|
272
|
+
return ok({
|
|
273
|
+
projects: compact,
|
|
274
|
+
project_count: rows.length,
|
|
275
|
+
...(rows.length > compact.length
|
|
276
|
+
? { truncated: `${rows.length - compact.length} more — raise limit or call slates_list_projects.` }
|
|
277
|
+
: {}),
|
|
278
|
+
activeProject,
|
|
279
|
+
});
|
|
110
280
|
},
|
|
111
281
|
};
|
|
112
282
|
export const getMe = {
|
|
@@ -191,6 +361,21 @@ export const VIDEO_MODELS = [
|
|
|
191
361
|
// the Max row at base rates and offers it 2K/4K it cannot render.
|
|
192
362
|
'minimax-h3',
|
|
193
363
|
'minimax-h3-max',
|
|
364
|
+
// LTX-2.5, two seats in one family (2026-08-29). Base is the VOLUME seat —
|
|
365
|
+
// cheapest native 1080p second we sell, free native audio at every tier, the
|
|
366
|
+
// only row reaching 1440p, and the longest clips in the catalogue (20s).
|
|
367
|
+
// Pro is the fidelity seat and is NOT a superset: shorter ladder (no
|
|
368
|
+
// 1440p/4K), shorter clips (6/8/10 only) and ~1/3 dearer.
|
|
369
|
+
//
|
|
370
|
+
// NEVER PREFIX-MATCH: 'ltx-2-5-pro' starts with 'ltx-2-5'. A prefix test
|
|
371
|
+
// bills Pro at base rates AND offers it 1440p/4K and 12-20s durations it
|
|
372
|
+
// cannot render — the same trap as the MiniMax pair, one row worse.
|
|
373
|
+
//
|
|
374
|
+
// Durations are DISCRETE AND EVEN (6,8,10,12,14,16,18,20); the union bounds
|
|
375
|
+
// below stay 3-30 because other rows are wider, so `assertVideoCapabilities`
|
|
376
|
+
// is what refuses an odd second. It reads `values`, not just min/max.
|
|
377
|
+
'ltx-2-5',
|
|
378
|
+
'ltx-2-5-pro',
|
|
194
379
|
];
|
|
195
380
|
// ── Capability-derived param vocabulary + guard ─────────────────
|
|
196
381
|
//
|
|
@@ -231,6 +416,17 @@ const VIDEO_RESOLUTION_VOCAB = VIDEO_RESOLUTIONS;
|
|
|
231
416
|
const MINIMAX_MODELS = new Set(['minimax-h3', 'minimax-h3-max']);
|
|
232
417
|
/** Reference images fal does not charge for. */
|
|
233
418
|
const MINIMAX_FREE_REF_IMAGES = 5;
|
|
419
|
+
/**
|
|
420
|
+
* The LTX-2.5 pair. A SET, not a prefix test — `ltx-2-5-pro` starts with
|
|
421
|
+
* `ltx-2-5`, and the two rows differ on ladder, duration list AND price.
|
|
422
|
+
*
|
|
423
|
+
* Their cost key is the plainest shape in the file — `{model}-{res}-{N}s` with
|
|
424
|
+
* no suffix ever, because LTX has no paid option: native audio is included at
|
|
425
|
+
* every tier (so no `-audio` variant like Kling) and there is no reference
|
|
426
|
+
* endpoint at all (so no `-ref{K}` variant like H3). Mirrors `ltxCreditKey()`
|
|
427
|
+
* in slate/src/shared/pricing.ts.
|
|
428
|
+
*/
|
|
429
|
+
const LTX_MODELS = new Set(['ltx-2-5', 'ltx-2-5-pro']);
|
|
234
430
|
/** K for the `-ref{K}` suffix: images past the free five, capped by the model's
|
|
235
431
|
* own declared ceiling. 0 for h3-max (no reference transport) and for anything
|
|
236
432
|
* that is not a MiniMax row. Mirrors refImageSurchargeCount() in
|
|
@@ -288,8 +484,14 @@ export const estimateGenerationCost = {
|
|
|
288
484
|
// hand-typed "Video 3-15" here was wrong the day seedance-2.5 (4-30s)
|
|
289
485
|
// shipped, and "Seedance defaults to 1080p" was wrong for 2.5, which has no
|
|
290
486
|
// 1080p at all.
|
|
291
|
-
|
|
292
|
-
|
|
487
|
+
// 🚨 THE PER-MODEL TABLES ARE NOT REPEATED HERE. `slates_generate_video`
|
|
488
|
+
// carries `describeDurations` and `describeVideoResolutions` and is always
|
|
489
|
+
// in context beside this op; a second copy is 1.2 KB of the same generated
|
|
490
|
+
// text on every turn, and it would go stale in exactly one direction — the
|
|
491
|
+
// one where someone edits a table and forgets there were two.
|
|
492
|
+
duration: z.number().int().min(1).max(360).optional().describe(`Seconds; cost scales linearly. Required with a video or PER-SECOND audio base id. Per-model windows: see slates_generate_video's duration. Audio: seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (⚠️ the requested duration IS the bill), eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. ⛔ NOT for ${TTS_MODEL} — pass \`characters\`.`),
|
|
493
|
+
characters: z.number().int().min(1).max(TTS_MAX_CHARACTERS).optional().describe(`${TTS_MODEL} only — the LENGTH OF THE TEXT to speak (${TTS_BUCKET_CHARS}-char buckets).`),
|
|
494
|
+
videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe('Video only. Omitted, each model quotes at its own default. Per-model ladders: see slates_generate_video\'s videoResolution.'),
|
|
293
495
|
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
|
|
294
496
|
quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
|
|
295
497
|
sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
|
|
@@ -304,7 +506,11 @@ export const estimateGenerationCost = {
|
|
|
304
506
|
let key = byKey.has(input.model) ? input.model : null;
|
|
305
507
|
// 2) image base id + resolution (+ quality for gpt-image-2)
|
|
306
508
|
if (!key) {
|
|
307
|
-
|
|
509
|
+
// IMAGE_MODELS, never a second hand-typed copy: this list is declared
|
|
510
|
+
// below (a runtime read, so no temporal-dead-zone hazard) and is the same
|
|
511
|
+
// enum `slates_generate_image` accepts. Two copies is how the estimate op
|
|
512
|
+
// would quietly stop pricing the seventh image model.
|
|
513
|
+
const img = IMAGE_MODELS.find((m) => m === input.model);
|
|
308
514
|
if (img)
|
|
309
515
|
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
|
|
310
516
|
}
|
|
@@ -314,7 +520,27 @@ export const estimateGenerationCost = {
|
|
|
314
520
|
// a seedance spelling.
|
|
315
521
|
if (!key && AUDIO_MODELS.includes(input.model)) {
|
|
316
522
|
const m = input.model;
|
|
317
|
-
|
|
523
|
+
// The TTS seat prices on CHARACTERS, so it leaves this per-second lane
|
|
524
|
+
// entirely. Asking for a duration here would quote a number that does not
|
|
525
|
+
// exist for it.
|
|
526
|
+
if (m === TTS_MODEL) {
|
|
527
|
+
if (!input.characters) {
|
|
528
|
+
return ok({
|
|
529
|
+
requires_clarification: true,
|
|
530
|
+
missing: ['characters'],
|
|
531
|
+
message: `${TTS_MODEL} bills per character, not per second — there is no duration to pass. Send the LENGTH OF THE TEXT to be spoken as \`characters\` (1-${TTS_MAX_CHARACTERS}).`,
|
|
532
|
+
});
|
|
533
|
+
}
|
|
534
|
+
if (input.characters > TTS_MAX_CHARACTERS) {
|
|
535
|
+
return ok({
|
|
536
|
+
requires_clarification: true,
|
|
537
|
+
missing: ['characters'],
|
|
538
|
+
message: `${TTS_MODEL} accepts up to ${TTS_MAX_CHARACTERS} characters in one take — ${input.characters} would be refused at generation time. Split the text and quote each line.`,
|
|
539
|
+
});
|
|
540
|
+
}
|
|
541
|
+
key = audioCostKey({ model: m, characters: input.characters });
|
|
542
|
+
}
|
|
543
|
+
if (!key && !input.duration) {
|
|
318
544
|
return ok({
|
|
319
545
|
requires_clarification: true,
|
|
320
546
|
missing: ['duration'],
|
|
@@ -329,20 +555,26 @@ export const estimateGenerationCost = {
|
|
|
329
555
|
// real 3s price and no hint that 2s is not a thing it can order. The
|
|
330
556
|
// generate op gates the same way; the two must agree or the quote is a
|
|
331
557
|
// promise the generation refuses to keep.
|
|
558
|
+
// Only the PER-SECOND surfaces have duration bounds. Keyed by those two
|
|
559
|
+
// ids rather than by AudioModel so adding a third surface on a different
|
|
560
|
+
// unit is a compile error here instead of a silent `undefined` bounds
|
|
561
|
+
// lookup that would throw at quote time.
|
|
332
562
|
const audioBounds = {
|
|
333
563
|
'seed-audio': { min: SEED_AUDIO_MIN_SECONDS, max: SEED_AUDIO_MAX_SECONDS },
|
|
334
564
|
'eleven-sfx': { min: ELEVEN_SFX_MIN_SECONDS, max: ELEVEN_SFX_MAX_SECONDS },
|
|
335
565
|
};
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
566
|
+
if (!key && m !== TTS_MODEL && input.duration) {
|
|
567
|
+
const bounds = audioBounds[m];
|
|
568
|
+
if (input.duration < bounds.min || input.duration > bounds.max) {
|
|
569
|
+
return ok({
|
|
570
|
+
requires_clarification: true,
|
|
571
|
+
missing: ['duration'],
|
|
572
|
+
message: `${m} accepts ${bounds.min}-${bounds.max} seconds — ${input.duration}s is outside that range and would be refused at generation time. ` +
|
|
573
|
+
'Re-ask with a duration in range.',
|
|
574
|
+
});
|
|
575
|
+
}
|
|
576
|
+
key = audioCostKey({ model: m, durationSeconds: input.duration });
|
|
344
577
|
}
|
|
345
|
-
key = audioCostKey({ model: m, durationSeconds: input.duration });
|
|
346
578
|
}
|
|
347
579
|
// 2b) Kling O3 edit base id + duration (ceiled source-clip length)
|
|
348
580
|
if (!key && (input.model === 'kling-v3.0-omni-edit' || input.model === 'kling-v3.0-omni-pro-edit')) {
|
|
@@ -398,10 +630,23 @@ export const estimateGenerationCost = {
|
|
|
398
630
|
}
|
|
399
631
|
const perCredits = key != null ? byKey.get(key) : undefined;
|
|
400
632
|
if (key == null || perCredits == null) {
|
|
401
|
-
|
|
633
|
+
// Every id in the error comes from the SSOT arrays. The image half was
|
|
634
|
+
// hand-typed and named three of six, so an agent that mis-spelled
|
|
635
|
+
// `gpt-image-2` was told the model did not exist.
|
|
636
|
+
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | ${IMAGE_MODELS.join(' | ')}) plus duration/resolution params, or use slates_list_available_models with a filter.`);
|
|
402
637
|
}
|
|
403
638
|
const qty = input.quantity ?? 1;
|
|
404
639
|
const totalCredits = perCredits * qty;
|
|
640
|
+
// 🚨 THE CRAFT CARD RIDES THIS RESULT. Measured 2026-08-30: the never-use
|
|
641
|
+
// list inlined where the agent could not skip it moved compliance 0/8 →
|
|
642
|
+
// 30/32, while the same guidance behind `slates_get_prompting_guide` sat at
|
|
643
|
+
// 13% before AND after. This is that placement applied to the POSITIVE half.
|
|
644
|
+
// It is here rather than in a param description because a description is
|
|
645
|
+
// paid for on every turn of every session and read once — this is paid for
|
|
646
|
+
// only by the call that is about to use the model.
|
|
647
|
+
const skill = promptingSkillFor(input.model);
|
|
648
|
+
const card = describeCraftCard(skill);
|
|
649
|
+
const banned = describeBannedTokensForSkill(skill);
|
|
405
650
|
return ok({
|
|
406
651
|
model: input.model,
|
|
407
652
|
cost_key: key,
|
|
@@ -409,7 +654,11 @@ export const estimateGenerationCost = {
|
|
|
409
654
|
cost_per_credits: perCredits,
|
|
410
655
|
total_credits: totalCredits,
|
|
411
656
|
requires_confirm: totalCredits > CONFIRM_CREDITS,
|
|
412
|
-
|
|
657
|
+
...(card ? { craft_card: card, craft_card_skill: skill } : {}),
|
|
658
|
+
...(banned ? { banned_tokens: banned } : {}),
|
|
659
|
+
}, `${input.model}${qty > 1 ? ` ×${qty}` : ''}: ${fmtCredits(totalCredits)} (${key}).` +
|
|
660
|
+
(card ? `\n\n--- HOW TO PROMPT ${input.model} ---\n${card}` : '') +
|
|
661
|
+
(banned ? `\n\n${banned}` : ''));
|
|
413
662
|
},
|
|
414
663
|
};
|
|
415
664
|
// ── Projects ────────────────────────────────────────────────────
|
|
@@ -447,11 +696,17 @@ export const getProject = {
|
|
|
447
696
|
* tool result). Every op that embeds asset lists uses this. */
|
|
448
697
|
function compactAsset(a) {
|
|
449
698
|
const r = a;
|
|
450
|
-
const prompt = typeof r.prompt === 'string' ? r.prompt : '';
|
|
451
699
|
return {
|
|
452
700
|
id: r.id,
|
|
453
701
|
code: r.code ?? null,
|
|
454
|
-
|
|
702
|
+
// ONE captioning rule, shared with the desktop gallery — see
|
|
703
|
+
// prompts/asset-label.ts for why the first 40 characters of a prompt is not
|
|
704
|
+
// a label on reference-driven work.
|
|
705
|
+
label: assetCaption({
|
|
706
|
+
label: typeof r.label === 'string' ? r.label : null,
|
|
707
|
+
shotName: typeof r.shotName === 'string' ? r.shotName : null,
|
|
708
|
+
prompt: typeof r.prompt === 'string' ? r.prompt : null,
|
|
709
|
+
}),
|
|
455
710
|
type: r.type,
|
|
456
711
|
created_at: r.createdAt ?? r.created_at ?? undefined,
|
|
457
712
|
};
|
|
@@ -850,6 +1105,7 @@ export const createEnvironment = {
|
|
|
850
1105
|
};
|
|
851
1106
|
export const generateCharacterIdentity = {
|
|
852
1107
|
id: 'slates_generate_character_identity',
|
|
1108
|
+
billable: true,
|
|
853
1109
|
description: "Generate one character identity sheet from a base portrait asset and bind it as the character's canonical reference. Call after slates_create_character. Read slates-character-identity before calling and quote the cost from slates_estimate_generation_cost.",
|
|
854
1110
|
input: z.object({
|
|
855
1111
|
characterId: z.string().uuid(),
|
|
@@ -873,6 +1129,7 @@ export const generateCharacterIdentity = {
|
|
|
873
1129
|
};
|
|
874
1130
|
export const generateEnvironmentPlate = {
|
|
875
1131
|
id: 'slates_generate_environment_plate',
|
|
1132
|
+
billable: true,
|
|
876
1133
|
description: "Generate one clean establishing image from an optional base image and bind it as the environment's canonical reference. Call after slates_create_environment and quote the cost from slates_estimate_generation_cost.",
|
|
877
1134
|
input: z.object({
|
|
878
1135
|
environmentId: z.string().uuid(),
|
|
@@ -916,7 +1173,11 @@ export const createStoryboard = {
|
|
|
916
1173
|
};
|
|
917
1174
|
export const getStoryboardWithFrames = {
|
|
918
1175
|
id: 'slates_get_storyboard_with_frames',
|
|
919
|
-
|
|
1176
|
+
// 🚨 ONE BOARD, TWO READERS. Every frame carries its Shots — the same rows,
|
|
1177
|
+
// in the same order, that the user is looking at. Before this the agent got
|
|
1178
|
+
// scenes and frames with no Shots, so the user read an arranged board while
|
|
1179
|
+
// the agent read an unordered pile.
|
|
1180
|
+
description: "Deep-fetch a storyboard: every scene, every slot, and the SHOT in each slot — its script line, references, model, params and takes — plus the piece's variety distribution. This is the same board, in the same order, that the user is reading.",
|
|
920
1181
|
input: z.object({ storyboardId: z.string().uuid() }),
|
|
921
1182
|
async run(input, ctx) {
|
|
922
1183
|
const desktop = ctx.desktop();
|
|
@@ -924,7 +1185,7 @@ export const getStoryboardWithFrames = {
|
|
|
924
1185
|
desktop.get('/agent/storyboards/get', { id: input.storyboardId }),
|
|
925
1186
|
desktop.get('/agent/scenes/full', { storyboardId: input.storyboardId }),
|
|
926
1187
|
]);
|
|
927
|
-
return ok({ ...storyboard, scenes: scenes.scenes });
|
|
1188
|
+
return ok({ ...storyboard, scenes: scenes.scenes, variety: scenes.variety }, describeVarietyReport(scenes.variety));
|
|
928
1189
|
},
|
|
929
1190
|
};
|
|
930
1191
|
export const addScene = {
|
|
@@ -941,7 +1202,11 @@ export const addScene = {
|
|
|
941
1202
|
};
|
|
942
1203
|
export const addFrame = {
|
|
943
1204
|
id: 'slates_add_frame',
|
|
944
|
-
|
|
1205
|
+
// ⚠️ THIS OP AND `slates_create_shot({ frameId })` NOW OVERLAP: two doors to
|
|
1206
|
+
// "put something in a scene". Naming the wart is cheaper than merging them
|
|
1207
|
+
// mid-release. Use THIS one only to park an existing picture; use
|
|
1208
|
+
// slates_create_shot to write a beat, which is almost always what you want.
|
|
1209
|
+
description: 'Park an EXISTING image in a scene as a new slot. It creates a Shot for that picture automatically, because a scene is a list of Shots. To write a beat — line, references, model, prompt — use slates_create_shot instead; it needs no image and files itself.',
|
|
945
1210
|
input: z.object({
|
|
946
1211
|
projectId: z.string().uuid(),
|
|
947
1212
|
storyboardId: z.string().uuid(),
|
|
@@ -1077,7 +1342,20 @@ function imageCostKey(model, resolution, quality = 'medium') {
|
|
|
1077
1342
|
}
|
|
1078
1343
|
export const generateImage = {
|
|
1079
1344
|
id: 'slates_generate_image',
|
|
1080
|
-
|
|
1345
|
+
billable: true,
|
|
1346
|
+
description: 'Generate an image via Slates credits.\n' +
|
|
1347
|
+
// GENERATED from MODEL_FACTS — the hand-typed model list that stood here
|
|
1348
|
+
// was a third copy of the routing doctrine, and it had already gone stale
|
|
1349
|
+
// (it still described nano-banana-2-lite by a capability the param owns).
|
|
1350
|
+
`${describeRouting('image')}\n` +
|
|
1351
|
+
'Full table: the slates-model-selection skill. ' +
|
|
1352
|
+
'Pass projectId to save into a Slates project (recommended — asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). ' +
|
|
1353
|
+
CONFIRM_GATE_SENTENCE +
|
|
1354
|
+
' MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use. ' +
|
|
1355
|
+
// GENERATED from the skill file's own never-use list -- the one piece of
|
|
1356
|
+
// prompting doctrine that is ALWAYS in context, because the agent has
|
|
1357
|
+
// demonstrably skipped the call that would have taught it.
|
|
1358
|
+
describeBannedTokens('image'),
|
|
1081
1359
|
input: z.object({
|
|
1082
1360
|
prompt: z.string().min(1).max(4000),
|
|
1083
1361
|
model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
|
|
@@ -1096,6 +1374,10 @@ export const generateImage = {
|
|
|
1096
1374
|
// Mirrors the cost confirm gate — defaults silently wasted credits
|
|
1097
1375
|
// (1:1 when user wanted 16:9, 4k when 1k would have done). The LLM
|
|
1098
1376
|
// is forced to ask the user or read the skill instead of guessing.
|
|
1377
|
+
// Non-blocking prompt hygiene. Computed once, reported on every exit path
|
|
1378
|
+
// that echoes a prompt -- the clarification and confirm gates are PRE-spend,
|
|
1379
|
+
// which is where a rewrite is still free.
|
|
1380
|
+
const promptWarning = bannedTokenWarning(input.prompt, 'image');
|
|
1099
1381
|
if (!input.aspectRatio || !input.resolution) {
|
|
1100
1382
|
const missing = [];
|
|
1101
1383
|
if (!input.aspectRatio)
|
|
@@ -1105,7 +1387,9 @@ export const generateImage = {
|
|
|
1105
1387
|
return ok({
|
|
1106
1388
|
requires_clarification: true,
|
|
1107
1389
|
missing,
|
|
1108
|
-
|
|
1390
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
1391
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
1392
|
+
`Missing required field(s): ${missing.join(', ')}. ` +
|
|
1109
1393
|
`Read the slates-prompting-nano-banana-2 + slates-cost-discipline skills, ` +
|
|
1110
1394
|
// Generated from MODEL_CAPABILITIES — never retype a ratio list.
|
|
1111
1395
|
`or ask the user. Aspect ratio by model: ${describeAspectRatios(IMAGE_MODELS)}. ` +
|
|
@@ -1178,7 +1462,7 @@ export const generateImage = {
|
|
|
1178
1462
|
if (!entry)
|
|
1179
1463
|
throw new Error(`Model not in registry: ${costKey}`);
|
|
1180
1464
|
const totalCents = creditCost(entry) * (input.count ?? 1);
|
|
1181
|
-
// Confirm gate. Fires
|
|
1465
|
+
// Confirm gate. Fires above CONFIRM_CREDITS, AND (look-first, mirroring
|
|
1182
1466
|
// slates_generate_video) whenever reference assets are involved
|
|
1183
1467
|
// regardless of cost — the LLM must see what it's referencing before
|
|
1184
1468
|
// committing spend.
|
|
@@ -1189,7 +1473,9 @@ export const generateImage = {
|
|
|
1189
1473
|
model: costKey,
|
|
1190
1474
|
estimated_cents: totalCents,
|
|
1191
1475
|
estimated_credits: totalCents,
|
|
1192
|
-
|
|
1476
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
1477
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
1478
|
+
`Cost exceeds ${CONFIRM_CREDITS} credits. Re-call with confirm=true to proceed, or pick a smaller resolution / count.`,
|
|
1193
1479
|
});
|
|
1194
1480
|
}
|
|
1195
1481
|
const previews = await previewAssets(ctx, referenceAssetIds.map((id) => ({ id, type: 'image', role: 'reference' })));
|
|
@@ -1201,7 +1487,8 @@ export const generateImage = {
|
|
|
1201
1487
|
`Review them against your prompt — every reference's role must be labeled in the prompt text. ` +
|
|
1202
1488
|
`If the references suggest a different composition / style than the current prompt captures, REVISE the prompt before confirming. ` +
|
|
1203
1489
|
`When you talk to the user about this gen, refer to each reference by its code (e.g. "${previews[0]?.ref ?? 'IMG-A?'}") — they'll see the matching badge in the Slates gallery.` +
|
|
1204
|
-
`\n\nWhen ready, re-call slates_generate_image with confirm=true and the (possibly revised) prompt
|
|
1490
|
+
`\n\nWhen ready, re-call slates_generate_image with confirm=true and the (possibly revised) prompt.` +
|
|
1491
|
+
(promptWarning ? `\n\n${promptWarning}` : ''),
|
|
1205
1492
|
images: previews.flatMap((p) => p.images),
|
|
1206
1493
|
data: {
|
|
1207
1494
|
requires_confirm: true,
|
|
@@ -1276,15 +1563,21 @@ export const generateImage = {
|
|
|
1276
1563
|
}
|
|
1277
1564
|
const requestedCount = input.count ?? 1;
|
|
1278
1565
|
return {
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
`
|
|
1287
|
-
|
|
1566
|
+
// The pixels are ALREADY here when the disk read worked, so the
|
|
1567
|
+
// review pointer says "look at what you have", not "call another op".
|
|
1568
|
+
// Telling the agent to re-fetch an image already in its context would
|
|
1569
|
+
// buy a wasted turn and teach the wrong habit.
|
|
1570
|
+
text: `${images.length > 0 ? IMAGE_INLINE_REVIEW : IMAGE_FETCH_POINTER} ` +
|
|
1571
|
+
(promptWarning ? `${promptWarning} ` : '') +
|
|
1572
|
+
(partialFailure
|
|
1573
|
+
? `Partial result: ${assetList.length} of ${requestedCount} image(s) saved into project ${input.projectId} ` +
|
|
1574
|
+
`(error on the rest: ${result.error ?? 'unknown error'}). ` +
|
|
1575
|
+
`The saved assets are in data.assets — re-generate only the missing count, don't redo the whole batch. ` +
|
|
1576
|
+
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"`
|
|
1577
|
+
: `Generated ${assetList.length} image(s) into project ${input.projectId} ` +
|
|
1578
|
+
`for ${fmtCredits(totalCents)}. ` +
|
|
1579
|
+
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
|
|
1580
|
+
(refEcho ? ` ${refEcho}` : '')),
|
|
1288
1581
|
images,
|
|
1289
1582
|
data: {
|
|
1290
1583
|
model: imageModel,
|
|
@@ -1352,7 +1645,9 @@ export const generateImage = {
|
|
|
1352
1645
|
images.push({ data: buf.toString('base64'), mimeType: mt });
|
|
1353
1646
|
}
|
|
1354
1647
|
return {
|
|
1355
|
-
text:
|
|
1648
|
+
text: `${IMAGE_INLINE_REVIEW} ` +
|
|
1649
|
+
(promptWarning ? `${promptWarning} ` : '') +
|
|
1650
|
+
`Generated ${urls.length} image(s) for ${fmtCredits(totalCents)}. ` +
|
|
1356
1651
|
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"`,
|
|
1357
1652
|
images,
|
|
1358
1653
|
data: {
|
|
@@ -1391,6 +1686,7 @@ async function pollProxyJob(cloud, jobId, options = {}) {
|
|
|
1391
1686
|
// ── Edit image ──────────────────────────────────────────────────
|
|
1392
1687
|
export const editImage = {
|
|
1393
1688
|
id: 'slates_edit_image',
|
|
1689
|
+
billable: true,
|
|
1394
1690
|
description: 'Surgically edit an existing image asset with a text instruction (e.g. \'remove the lamppost\', \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The edited result is saved as a NEW asset in the project (prompt prefixed \'[Edit]\'); the source is untouched. Default model nano-banana-2 (only model that also accepts referenceAssetIds); flux-2-max / seedream-5-lite use their own edit endpoints and ignore references. Before first use call slates_get_prompting_guide with topic \'slates-edit-and-iterate\'.',
|
|
1395
1691
|
input: z.object({
|
|
1396
1692
|
projectId: z.string().uuid(),
|
|
@@ -1569,6 +1865,16 @@ export function videoCostKey(input) {
|
|
|
1569
1865
|
const k = minimaxRefSurchargeCount(input.model, input.referenceImages);
|
|
1570
1866
|
return `${input.model}-${res}-${input.duration}s${k > 0 ? `-ref${k}` : ''}`;
|
|
1571
1867
|
}
|
|
1868
|
+
// LTX-2.5, both seats. EXACT-ID SET, NEVER A PREFIX — `ltx-2-5-pro` starts
|
|
1869
|
+
// with `ltx-2-5`, and a prefix match would quote base rates for the dearer
|
|
1870
|
+
// row. No suffix dimension exists: audio is free and there are no references.
|
|
1871
|
+
// The resolution default is read PER ROW (both are 1080p today, but the two
|
|
1872
|
+
// ladders differ, so a shared literal would be a latent bug the day one
|
|
1873
|
+
// moves). Mirrors ltxCreditKey() in slate/src/shared/pricing.ts.
|
|
1874
|
+
if (LTX_MODELS.has(input.model)) {
|
|
1875
|
+
const res = input.videoResolution ?? defaultVideoResolutionFor(input.model);
|
|
1876
|
+
return `${input.model}-${res}-${input.duration}s`;
|
|
1877
|
+
}
|
|
1572
1878
|
if (input.model.startsWith('seedance')) {
|
|
1573
1879
|
// Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version × face
|
|
1574
1880
|
// × vref × res × duration). AI-face route bills the `-face-` key (~45% over
|
|
@@ -1663,31 +1969,7 @@ export function seedanceEditCostKey(input) {
|
|
|
1663
1969
|
// Exported: the exact `model` ids slates_generate_audio accepts — consumed
|
|
1664
1970
|
// by the desktop Studio Agent system prompt (SSOT; never restate these ids
|
|
1665
1971
|
// in prose that can drift). Mirrors VIDEO_MODELS for the third media type.
|
|
1666
|
-
export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx'];
|
|
1667
|
-
/**
|
|
1668
|
-
* Per-surface bounds and defaults.
|
|
1669
|
-
*
|
|
1670
|
-
* 🚨 THESE FOUR NUMBERS PER SURFACE LIVE IN THREE REPOS. A change is a
|
|
1671
|
-
* three-site edit, every time:
|
|
1672
|
-
* 1. HERE (`audioCostKey`, the agent's pre-flight quote)
|
|
1673
|
-
* 2. `slate/src/shared/pricing.ts` → MODEL_REGISTRY `audio.durationSeconds`
|
|
1674
|
-
* (min/max/default), read by `clampAudioDuration` + `audioCreditKey`
|
|
1675
|
-
* 3. `slates-api/src/lib/audio-keys.ts` → the server's fail-closed bounds
|
|
1676
|
-
* `slates-api/scripts/pricing-consistency-check.mjs` §4 asserts 1 and 2 agree
|
|
1677
|
-
* at EVERY value including out-of-range ones; the gate check covers 3.
|
|
1678
|
-
*
|
|
1679
|
-
* The MINs used to be missing here, and the clamp floor was a hardcoded 1. That
|
|
1680
|
-
* made `slates_estimate_generation_cost({model:'seed-audio', duration:2})`
|
|
1681
|
-
* quote a real `seed-audio-2s` price for a generation the desktop would bill as
|
|
1682
|
-
* 3s and the proxy would REJECT outright. Same for an omitted duration, which
|
|
1683
|
-
* quoted `seed-audio-1s` against the desktop's `seed-audio-15s`.
|
|
1684
|
-
*/
|
|
1685
|
-
export const SEED_AUDIO_MIN_SECONDS = 3;
|
|
1686
|
-
export const SEED_AUDIO_MAX_SECONDS = 120;
|
|
1687
|
-
export const SEED_AUDIO_DEFAULT_SECONDS = 15;
|
|
1688
|
-
export const ELEVEN_SFX_MIN_SECONDS = 1;
|
|
1689
|
-
export const ELEVEN_SFX_MAX_SECONDS = 22;
|
|
1690
|
-
export const ELEVEN_SFX_DEFAULT_SECONDS = 4;
|
|
1972
|
+
export const AUDIO_MODELS = ['seed-audio', 'eleven-sfx', 'inworld-tts-2'];
|
|
1691
1973
|
/**
|
|
1692
1974
|
* Byte-for-byte the desktop's `clampAudioDuration` in slate/src/shared/pricing.ts,
|
|
1693
1975
|
* INCLUDING the non-finite arm — that one matters: `Math.max(min, NaN)` is NaN,
|
|
@@ -1711,6 +1993,21 @@ function clampInt(value, min, max) {
|
|
|
1711
1993
|
// injects "... N seconds" into the prompt and bills seed-audio-{N}s, so
|
|
1712
1994
|
// display == billing with no amendment to the pricing law. The server probes
|
|
1713
1995
|
// the returned audio.duration afterwards and logs SEED AUDIO BILLING DRIFT.
|
|
1996
|
+
/**
|
|
1997
|
+
* Characters → the billed bucket, byte-identical to the desktop's
|
|
1998
|
+
* `clampTtsCharacters`. Rounds UP to the next 250 and clamps to [250, 2000].
|
|
1999
|
+
*
|
|
2000
|
+
* The non-finite arm matters for the same reason it does on the duration side:
|
|
2001
|
+
* `Math.ceil(NaN)` is NaN, which would build the key `inworld-tts-2-NaNc` here
|
|
2002
|
+
* while the desktop quotes the first bucket. Divergence at a value neither side
|
|
2003
|
+
* can bill is still divergence.
|
|
2004
|
+
*/
|
|
2005
|
+
function clampTtsCharacters(value) {
|
|
2006
|
+
if (value == null || !Number.isFinite(value))
|
|
2007
|
+
return TTS_BUCKET_CHARS;
|
|
2008
|
+
const buckets = Math.ceil(value / TTS_BUCKET_CHARS);
|
|
2009
|
+
return Math.min(TTS_BUCKET_COUNT, Math.max(1, buckets)) * TTS_BUCKET_CHARS;
|
|
2010
|
+
}
|
|
1714
2011
|
export function audioCostKey(input) {
|
|
1715
2012
|
if (input.model === 'seed-audio') {
|
|
1716
2013
|
// `?? default` before the clamp, not `?? 0` — the desktop resolves a missing
|
|
@@ -1723,6 +2020,11 @@ export function audioCostKey(input) {
|
|
|
1723
2020
|
const secs = clampAudioSeconds(input.durationSeconds ?? ELEVEN_SFX_DEFAULT_SECONDS, ELEVEN_SFX_MIN_SECONDS, ELEVEN_SFX_MAX_SECONDS, ELEVEN_SFX_DEFAULT_SECONDS);
|
|
1724
2021
|
return `eleven-sfx-${secs}s`;
|
|
1725
2022
|
}
|
|
2023
|
+
if (input.model === TTS_MODEL) {
|
|
2024
|
+
// `c` for characters, so the key can never be mistaken for a seconds key by
|
|
2025
|
+
// the `-\d+s$` tests the proxy and the desktop both run.
|
|
2026
|
+
return `${TTS_MODEL}-${clampTtsCharacters(input.characters)}c`;
|
|
2027
|
+
}
|
|
1726
2028
|
throw new Error(`Unknown audio model: ${input.model}`);
|
|
1727
2029
|
}
|
|
1728
2030
|
/**
|
|
@@ -1841,33 +2143,39 @@ function spokenTextByAssetId(assetIds, spoken) {
|
|
|
1841
2143
|
});
|
|
1842
2144
|
return Object.keys(out).length > 0 ? out : undefined;
|
|
1843
2145
|
}
|
|
1844
|
-
// Maps a
|
|
1845
|
-
//
|
|
1846
|
-
//
|
|
1847
|
-
//
|
|
2146
|
+
// Maps a model id to its bundled prompting skill (frontmatter `name:`), so
|
|
2147
|
+
// guidance text points at a skill that actually exists.
|
|
2148
|
+
//
|
|
2149
|
+
// 🚨 ONE RESOLVER. This was a second hand-typed alias table beside
|
|
2150
|
+
// `resolveGuideTopic()`, and it had already fallen behind: it knew nothing
|
|
2151
|
+
// about LTX-2.5, so every LTX generation was told to read the cost skill
|
|
2152
|
+
// instead of its own guide. Delegate; the ordering traps (2.5 before seedance,
|
|
2153
|
+
// seed-audio before seedance, minimax before both) are solved once, there.
|
|
1848
2154
|
function promptingSkillFor(model) {
|
|
1849
|
-
|
|
1850
|
-
return 'slates-prompting-kling-v3';
|
|
1851
|
-
if (model.startsWith('veo'))
|
|
1852
|
-
return 'slates-prompting-veo-3';
|
|
1853
|
-
// 2.5 BEFORE the generic seedance test — "seedance-2.5" also starts with
|
|
1854
|
-
// "seedance", and falling through hands 2.0's guide to a model with different
|
|
1855
|
-
// limits, a different resolution ladder and an extra task type.
|
|
1856
|
-
if (model.startsWith('seedance-2.5'))
|
|
1857
|
-
return 'slates-prompting-seedance-2-5';
|
|
1858
|
-
if (model.startsWith('seedance'))
|
|
1859
|
-
return 'slates-prompting-seedance';
|
|
1860
|
-
if (model.startsWith('omni-flash'))
|
|
1861
|
-
return 'slates-prompting-omni-flash';
|
|
1862
|
-
// ONE skill covers both H3 seats — the prompt grammar is identical and only
|
|
1863
|
-
// the ladder and the reference transport differ — so a prefix is right here.
|
|
1864
|
-
if (model.startsWith('minimax-h3'))
|
|
1865
|
-
return 'slates-prompting-minimax-h3';
|
|
1866
|
-
return 'slates-cost-discipline';
|
|
2155
|
+
return resolveGuideTopic(model) ?? 'slates-cost-discipline';
|
|
1867
2156
|
}
|
|
2157
|
+
/**
|
|
2158
|
+
* The per-model prompting guides for the whole video roster, DERIVED.
|
|
2159
|
+
*
|
|
2160
|
+
* `slates_generate_video`'s description hand-typed five of them and omitted
|
|
2161
|
+
* `slates-prompting-ltx-2-5` for as long as LTX shipped — an agent reading the
|
|
2162
|
+
* description could not learn the guide existed. A hand-typed index of a
|
|
2163
|
+
* generated corpus is a stale index; it is only a matter of when.
|
|
2164
|
+
*/
|
|
2165
|
+
const VIDEO_MODEL_GUIDES = [
|
|
2166
|
+
...new Set(VIDEO_MODELS.map((m) => promptingSkillFor(m))),
|
|
2167
|
+
].join(' / ');
|
|
1868
2168
|
export const generateVideo = {
|
|
1869
2169
|
id: 'slates_generate_video',
|
|
1870
|
-
|
|
2170
|
+
billable: true,
|
|
2171
|
+
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (' +
|
|
2172
|
+
VIDEO_MODEL_GUIDES +
|
|
2173
|
+
') — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). ' +
|
|
2174
|
+
CONFIRM_GATE_SENTENCE +
|
|
2175
|
+
' Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8"). ' +
|
|
2176
|
+
// GENERATED from the skill's own slop-token list. Always in context on both
|
|
2177
|
+
// surfaces, so it survives an agent that skips slates_get_prompting_guide.
|
|
2178
|
+
describeBannedTokens('video'),
|
|
1871
2179
|
input: z.object({
|
|
1872
2180
|
prompt: z.string().min(1).max(4000),
|
|
1873
2181
|
// ROUTING doctrine only. Every capability number was stripped on 2026-08-16
|
|
@@ -1876,7 +2184,14 @@ export const generateVideo = {
|
|
|
1876
2184
|
// "seedance-2.5 480p/720p" were both stated here AND there, and the two
|
|
1877
2185
|
// copies disagreed — and the second of those went stale on 2026-08-24 when
|
|
1878
2186
|
// 2.5 gained 1080p, which is exactly the failure mode generating it fixes.
|
|
1879
|
-
model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate
|
|
2187
|
+
model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id — duration and videoResolution are separate ` +
|
|
2188
|
+
`params (registry cost keys like "kling-v3-standard-8s" auto-resolve). All are VIDEO-only.\n` +
|
|
2189
|
+
// GENERATED from MODEL_FACTS. The paragraph that stood here restated it by
|
|
2190
|
+
// hand and had already drifted a phrase at a time.
|
|
2191
|
+
`${describeRouting('video', 'generate')}\n` +
|
|
2192
|
+
`Full table: the slates-model-selection skill. Each model's legal aspect ratios, durations and ` +
|
|
2193
|
+
`resolutions are in those params' own descriptions — read them there, not from memory. ` +
|
|
2194
|
+
`For per-call cost, call slates_estimate_generation_cost.`),
|
|
1880
2195
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
|
|
1881
2196
|
// 🚨 THESE THREE DESCRIPTIONS ARE GENERATED FROM `MODEL_CAPABILITIES`.
|
|
1882
2197
|
// Never hand-write a ratio, resolution or duration into them again — every
|
|
@@ -1886,32 +2201,39 @@ export const generateVideo = {
|
|
|
1886
2201
|
aspectRatio: zEnum(VIDEO_ASPECT_RATIOS).optional().describe(`NOT every model takes every value — an out-of-set ratio is REFUSED before submit, not silently ignored. Per model: ${describeAspectRatios(VIDEO_MODELS, AGENT_ROUTE_PROVIDER)}`),
|
|
1887
2202
|
duration: z.number().int().min(VIDEO_DURATION_BOUNDS.min).max(VIDEO_DURATION_BOUNDS.max).optional().describe(`Seconds. Cost scales linearly — a 30s seedance-2.5 take is several hundred credits, so be explicit rather than defaulting. Per model: ${describeDurations(VIDEO_MODELS)}`),
|
|
1888
2203
|
videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe(`An unsupported resolution is REJECTED, never downgraded. Per model: ${describeVideoResolutions(VIDEO_MODELS)}. 4K video is Pro-only (the server returns PRO_REQUIRED for a base-tier account).`),
|
|
1889
|
-
|
|
1890
|
-
|
|
2204
|
+
// 🚨 EVERY DESCRIPTION BELOW IS A CONSTRAINT OR A GENERATED TABLE — never
|
|
2205
|
+
// craft, never rationale, never a worked example. This op is the single
|
|
2206
|
+
// largest thing in the desktop's prompt-cached prefix (measured 15,357 of
|
|
2207
|
+
// 112,114 bytes on 2026-09-02, 13.7%), and almost all of the excess was
|
|
2208
|
+
// reasoning that belongs in slates-prompting-* where it is read once, on
|
|
2209
|
+
// demand, by the one session that needs it. If you are about to explain
|
|
2210
|
+
// WHY here, you are writing the skill in the wrong file.
|
|
2211
|
+
firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video (UUID or badge code, resolved at call time).'),
|
|
2212
|
+
lastFrameAssetId: z.string().optional().describe('Ending frame. Veo and Seedance only; pairs with firstFrameAssetId.'),
|
|
1891
2213
|
ingredientAssetIds: z.array(z.string()).max(30).optional().describe(
|
|
1892
2214
|
// Caps DERIVED from MODEL_CAPABILITIES — the hand-typed list omitted
|
|
1893
2215
|
// kling-v3.0-omni-pro entirely and read as if 7 were an Omni Flash-only rule.
|
|
1894
|
-
`Visual reference / ingredient assets
|
|
1895
|
-
characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets
|
|
1896
|
-
environmentAssetIds: z.array(z.string()).optional().describe('Environment
|
|
1897
|
-
styleAssetIds: z.array(z.string()).optional().describe('Style
|
|
1898
|
-
videoReferenceAssetId: z.string().optional().describe('DEPRECATED —
|
|
1899
|
-
videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach.
|
|
1900
|
-
audioReferenceAssetId: z.string().optional().describe('DEPRECATED —
|
|
2216
|
+
`Visual reference / ingredient assets. Cap per model, combined across the ingredient/character/environment/style params: ${describeReferenceImageCaps(VIDEO_MODELS)}. 2-4 strong references beat both extremes.`),
|
|
2217
|
+
characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets — keeps a character consistent.'),
|
|
2218
|
+
environmentAssetIds: z.array(z.string()).optional().describe('Environment references — keeps a location consistent.'),
|
|
2219
|
+
styleAssetIds: z.array(z.string()).optional().describe('Style references — locks the look.'),
|
|
2220
|
+
videoReferenceAssetId: z.string().optional().describe('DEPRECATED — use videoReferenceAssetIds. Kept working: shipped CLI/MCP builds send this shape.'),
|
|
2221
|
+
videoReferenceSeconds: z.number().optional().describe('DEPRECATED — the singular partner of videoReferenceSecondsEach.'),
|
|
2222
|
+
audioReferenceAssetId: z.string().optional().describe('DEPRECATED — use audioReferenceAssetIds. Carries no spoken text.'),
|
|
1901
2223
|
// ── Multimodal references, the plural surface ──
|
|
1902
2224
|
// The capacity sentences are DERIVED from MODEL_FACTS (see
|
|
1903
2225
|
// multimodalRefSummary) rather than hand-typed, so a cap change in one
|
|
1904
2226
|
// place cannot leave a stale number in a description an LLM reads.
|
|
1905
|
-
videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS
|
|
1906
|
-
videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each
|
|
1907
|
-
audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO
|
|
1908
|
-
audioReferenceSpokenText: z.array(z.string()).optional().describe('
|
|
2227
|
+
videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS, cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')} Billing switches to the vref key (input+output seconds) — pass videoReferenceSecondsEach. Over the cap is REFUSED, never trimmed.`),
|
|
2228
|
+
videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each clip\'s duration in seconds. Feeds the vref cost key; the server re-probes and corrects an understated value upward.'),
|
|
2229
|
+
audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO, cited as "audio 1", "audio 2"… in the order given. No billing surcharge. ${multimodalRefSummary('seedance-2')} ${multimodalRefSummary('seedance-2.5')}`),
|
|
2230
|
+
audioReferenceSpokenText: z.array(z.string()).optional().describe('The exact words in each reference clip — same order and length as audioReferenceAssetIds, "" for a clip with no speech. The model RE-TRANSCRIBES a take rather than using it verbatim, so audio decides voice/accent/timing and only this decides the WORDS. Omit it and the words are a guess.'),
|
|
1909
2231
|
sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: enable audio generation. Default true.'),
|
|
1910
2232
|
audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
|
|
1911
2233
|
generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
|
|
1912
|
-
seedanceFace: z.boolean().optional().describe('Seedance ONLY:
|
|
1913
|
-
seedanceRealFace: z.boolean().optional().describe('Seedance ONLY:
|
|
1914
|
-
realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace:
|
|
2234
|
+
seedanceFace: z.boolean().optional().describe('Seedance ONLY: a reference shows an AI CHARACTER\'s face. Faces are blocked on the default route, so this reroutes to a face-capable provider at ~45% more. A REAL person fails here with [REAL_FACE_DETECTED] — see seedanceRealFace.'),
|
|
2235
|
+
seedanceRealFace: z.boolean().optional().describe('Seedance ONLY: a reference shows a REAL person. Premium route, roughly 2x the AI-face price — quote it first. REQUIRES realFaceConsent=true.'),
|
|
2236
|
+
realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace: true ONLY after the user has explicitly confirmed they hold rights/consent to this likeness and it does not impersonate or misrepresent them. Refused without it; public figures fail on every route.'),
|
|
1915
2237
|
negativePrompt: z.string().optional(),
|
|
1916
2238
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
1917
2239
|
confirm: z.boolean().optional().describe('Set true after explicit user OK to bypass the confirm gate (which fires for almost every video gen since they\'re expensive).'),
|
|
@@ -1940,11 +2262,16 @@ export const generateVideo = {
|
|
|
1940
2262
|
// no asset to reference later, and a failed gen leaves the user with
|
|
1941
2263
|
// nothing. The MCP-only headless path that exists for image gen is
|
|
1942
2264
|
// not reasonable for video given the cost.
|
|
2265
|
+
// Non-blocking prompt hygiene, computed once. Reported on the gates that
|
|
2266
|
+
// fire BEFORE any spend, where a rewrite is still free.
|
|
2267
|
+
const promptWarning = bannedTokenWarning(input.prompt, 'video');
|
|
1943
2268
|
if (!input.projectId) {
|
|
1944
2269
|
return ok({
|
|
1945
2270
|
requires_clarification: true,
|
|
1946
2271
|
missing: ['projectId'],
|
|
1947
|
-
|
|
2272
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
2273
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
2274
|
+
'projectId is required for video generation. Use slates_list_projects to find one or slates_create_project to make a new one. Video gens cost tens to a few hundred credits per call — they need to land in a project so the user sees the progress card and the result.',
|
|
1948
2275
|
});
|
|
1949
2276
|
}
|
|
1950
2277
|
if (!input.aspectRatio || !input.duration) {
|
|
@@ -1956,7 +2283,9 @@ export const generateVideo = {
|
|
|
1956
2283
|
return ok({
|
|
1957
2284
|
requires_clarification: true,
|
|
1958
2285
|
missing,
|
|
1959
|
-
|
|
2286
|
+
...(promptWarning ? { prompt_warning: promptWarning } : {}),
|
|
2287
|
+
message: (promptWarning ? `${promptWarning}\n\n` : '') +
|
|
2288
|
+
`Missing required field(s): ${missing.join(', ')}. ` +
|
|
1960
2289
|
`Read the slates-cost-discipline + ${promptingSkillFor(input.model)} skills, ` +
|
|
1961
2290
|
// Generated from MODEL_CAPABILITIES. The prose that stood here claimed
|
|
1962
2291
|
// "Veo locks to 16:9", "Kling/Seedance support 1:1 16:9 9:16 4:3 3:4
|
|
@@ -2002,6 +2331,34 @@ export const generateVideo = {
|
|
|
2002
2331
|
});
|
|
2003
2332
|
}
|
|
2004
2333
|
}
|
|
2334
|
+
// LTX-2.5's SHAPE constraint: FRAMES, NEVER REFERENCES. fal publishes
|
|
2335
|
+
// text-to-video and image-to-video for `lightricks/ltx-2.5` and nothing
|
|
2336
|
+
// else — there is no reference-to-video endpoint on either seat, so there
|
|
2337
|
+
// is no transport for an ingredient, character, environment, style,
|
|
2338
|
+
// reference-video or reference-audio input.
|
|
2339
|
+
//
|
|
2340
|
+
// This has to be an EXPLICIT refusal rather than a silent no-op: accepting
|
|
2341
|
+
// reference ids we cannot send is the exact "no error, no warning, no
|
|
2342
|
+
// images in the request" failure `seedance-2.5-edit` shipped with. Start
|
|
2343
|
+
// and end FRAMES are unaffected — image-to-video carries `image_url` plus
|
|
2344
|
+
// an optional `end_image_url`, which is what `features.lastFrame` records.
|
|
2345
|
+
if (LTX_MODELS.has(input.model)) {
|
|
2346
|
+
const refImages = (input.ingredientAssetIds?.length ?? 0) +
|
|
2347
|
+
(input.characterAssetIds?.length ?? 0) +
|
|
2348
|
+
(input.environmentAssetIds?.length ?? 0) +
|
|
2349
|
+
(input.styleAssetIds?.length ?? 0);
|
|
2350
|
+
const refMedia = (input.videoReferenceAssetIds?.length ?? 0) +
|
|
2351
|
+
(input.audioReferenceAssetIds?.length ?? 0) +
|
|
2352
|
+
(input.videoReferenceAssetId ? 1 : 0) +
|
|
2353
|
+
(input.audioReferenceAssetId ? 1 : 0);
|
|
2354
|
+
if (refImages > 0 || refMedia > 0) {
|
|
2355
|
+
return ok({
|
|
2356
|
+
requires_clarification: true,
|
|
2357
|
+
missing: [],
|
|
2358
|
+
message: `${input.model} takes a prompt and up to two frames (start and/or end) — it has no reference endpoint at all, so reference images, video and audio cannot be sent. Drop them, or switch to minimax-h3, which reads ${getModelCapability('minimax-h3')?.maxIngredientImages ?? 9} images plus reference video and audio.`,
|
|
2359
|
+
});
|
|
2360
|
+
}
|
|
2361
|
+
}
|
|
2005
2362
|
// MiniMax H3's SHAPE constraints. Counts and caps are read from the
|
|
2006
2363
|
// capability SSOT; only the endpoint SHAPE is stated here, because it is
|
|
2007
2364
|
// not a number the registry models: fal publishes text-to-video,
|
|
@@ -2229,7 +2586,7 @@ export const generateVideo = {
|
|
|
2229
2586
|
}
|
|
2230
2587
|
const totalCents = creditCost(entry);
|
|
2231
2588
|
// Pre-flight confirm gate. Fires when:
|
|
2232
|
-
// (a) cost
|
|
2589
|
+
// (a) cost above CONFIRM_CREDITS (the cost gate), OR
|
|
2233
2590
|
// (b) any reference assets are involved (the look-first gate)
|
|
2234
2591
|
// When references are present, the response inlines them as image
|
|
2235
2592
|
// content blocks (and video keyframes for video refs) so the LLM
|
|
@@ -2268,6 +2625,7 @@ export const generateVideo = {
|
|
|
2268
2625
|
text: `Pre-flight for ${input.duration}s ${input.model} (${costKey}): ` +
|
|
2269
2626
|
`${fmtCredits(totalCents)}.` +
|
|
2270
2627
|
refSummary +
|
|
2628
|
+
(promptWarning ? `\n\n${promptWarning}` : '') +
|
|
2271
2629
|
`\n\nWhen ready, re-call slates_generate_video with confirm=true and the (possibly revised) prompt.`,
|
|
2272
2630
|
images: refImages,
|
|
2273
2631
|
data: {
|
|
@@ -2346,7 +2704,12 @@ export const generateVideo = {
|
|
|
2346
2704
|
}, refEcho);
|
|
2347
2705
|
}
|
|
2348
2706
|
return {
|
|
2349
|
-
text:
|
|
2707
|
+
text:
|
|
2708
|
+
// No frames come back with a video generation, so unlike the image op
|
|
2709
|
+
// this one has to name the op that fetches them.
|
|
2710
|
+
`${VIDEO_REVIEW_POINTER} ` +
|
|
2711
|
+
(promptWarning ? `${promptWarning} ` : '') +
|
|
2712
|
+
`Generated ${input.duration}s ${input.model} video into project ${input.projectId} ` +
|
|
2350
2713
|
`for ${fmtCredits(totalCents)}. ` +
|
|
2351
2714
|
`Prompt: "${input.prompt.slice(0, 60)}${input.prompt.length > 60 ? '...' : ''}"` +
|
|
2352
2715
|
(refEcho ? ` ${refEcho}` : ''),
|
|
@@ -2367,20 +2730,26 @@ export const generateVideo = {
|
|
|
2367
2730
|
// ── Generate audio ──────────────────────────────────────────────
|
|
2368
2731
|
export const generateAudio = {
|
|
2369
2732
|
id: 'slates_generate_audio',
|
|
2370
|
-
|
|
2733
|
+
billable: true,
|
|
2734
|
+
description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Three surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s), eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop), and inworld-tts-2 (one named voice saying one line; the prompt IS the words, billed per character). Which surface for which job: read the slates-model-selection skill. ' +
|
|
2371
2735
|
'🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
|
|
2372
|
-
'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
|
|
2373
|
-
'projectId is REQUIRED (no headless path).
|
|
2736
|
+
'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
|
|
2737
|
+
'projectId is REQUIRED (no headless path). ' +
|
|
2738
|
+
CONFIRM_GATE_SENTENCE +
|
|
2739
|
+
' No skill files installed? Call slates_get_prompting_guide first.',
|
|
2374
2740
|
input: z.object({
|
|
2375
2741
|
projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
|
|
2376
2742
|
model: z
|
|
2377
2743
|
.enum(AUDIO_MODELS)
|
|
2378
|
-
.describe(
|
|
2744
|
+
.describe(
|
|
2745
|
+
// GENERATED from MODEL_FACTS — the audio lane's routing lived only in the
|
|
2746
|
+
// system prompt and in a one-line paraphrase here.
|
|
2747
|
+
`Audio surface.\n${describeRouting('audio')}\nFull table: the slates-model-selection skill.`),
|
|
2379
2748
|
prompt: z
|
|
2380
2749
|
.string()
|
|
2381
2750
|
.min(1)
|
|
2382
2751
|
.max(5000)
|
|
2383
|
-
.describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars.'),
|
|
2752
|
+
.describe('seed-audio: ONE plain sentence describing the scene (no production jargon, no "SFX:" prefixes; name the crowd/room size). eleven-sfx: the effect described by its physical CAUSE ("heavy oak door slams shut in a stone hallway"), max 450 chars. inworld-tts-2: THE WORDS TO SPEAK, verbatim, max ' + TTS_MAX_CHARACTERS + ' — its length is the bill.'),
|
|
2384
2753
|
durationSeconds: z
|
|
2385
2754
|
.number()
|
|
2386
2755
|
.optional()
|
|
@@ -2389,6 +2758,18 @@ export const generateAudio = {
|
|
|
2389
2758
|
.string()
|
|
2390
2759
|
.optional()
|
|
2391
2760
|
.describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
|
|
2761
|
+
voiceId: z
|
|
2762
|
+
.string()
|
|
2763
|
+
.optional()
|
|
2764
|
+
.describe('inworld-tts-2 — the voice to speak in. One of these three is required there.'),
|
|
2765
|
+
voiceReferenceAssetId: z
|
|
2766
|
+
.string()
|
|
2767
|
+
.optional()
|
|
2768
|
+
.describe('inworld-tts-2 — clone the voice from this AUDIO asset (5-15s of one clean speaker).'),
|
|
2769
|
+
voiceDescription: z
|
|
2770
|
+
.string()
|
|
2771
|
+
.optional()
|
|
2772
|
+
.describe('inworld-tts-2 — build a voice from this description, for a character with no recording.'),
|
|
2392
2773
|
speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
|
|
2393
2774
|
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
|
|
2394
2775
|
pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
|
|
@@ -2409,13 +2790,48 @@ export const generateAudio = {
|
|
|
2409
2790
|
}),
|
|
2410
2791
|
run: async (input, ctx) => {
|
|
2411
2792
|
// ── Per-surface clarification + constraint gates ──
|
|
2793
|
+
//
|
|
2794
|
+
// `null` for the TTS seat is load-bearing rather than a placeholder: that
|
|
2795
|
+
// surface has NO duration dimension at all (speech length falls out of the
|
|
2796
|
+
// text), so there is no default that would be honest. A number here would
|
|
2797
|
+
// flow into `audioCostKey` and quote a per-second price for a per-character
|
|
2798
|
+
// generation.
|
|
2412
2799
|
const cfgDefaults = {
|
|
2413
2800
|
'seed-audio': SEED_AUDIO_DEFAULT_SECONDS,
|
|
2414
2801
|
'eleven-sfx': ELEVEN_SFX_DEFAULT_SECONDS,
|
|
2802
|
+
'inworld-tts-2': null,
|
|
2415
2803
|
};
|
|
2416
|
-
const seconds = input.durationSeconds ?? cfgDefaults[input.model];
|
|
2804
|
+
const seconds = input.durationSeconds ?? cfgDefaults[input.model] ?? undefined;
|
|
2805
|
+
if (input.model === TTS_MODEL) {
|
|
2806
|
+
// The text IS the prompt on this surface — there is no separate field,
|
|
2807
|
+
// because "the words that get spoken" and "what you asked for" are the
|
|
2808
|
+
// same thing here. That also means the character count is knowable
|
|
2809
|
+
// client-side, which is what makes the quote below exact.
|
|
2810
|
+
if (input.prompt.length > TTS_MAX_CHARACTERS) {
|
|
2811
|
+
throw new Error(`${TTS_MODEL} accepts up to ${TTS_MAX_CHARACTERS} characters in one take — this text is ${input.prompt.length}. Split it into separate lines and generate each one.`);
|
|
2812
|
+
}
|
|
2813
|
+
if (input.durationSeconds != null) {
|
|
2814
|
+
throw new Error(`${TTS_MODEL} has no duration parameter — speech length falls out of the text, and it bills per character. Drop durationSeconds.`);
|
|
2815
|
+
}
|
|
2816
|
+
// EXACTLY ONE voice source. Zero is a clarification (the agent should ask
|
|
2817
|
+
// who is speaking, not guess); two or more is an error, because the three
|
|
2818
|
+
// paths are genuinely different generations and silently picking one would
|
|
2819
|
+
// spend credits on a voice the caller did not ask for.
|
|
2820
|
+
const voiceSources = [input.voiceId, input.voiceReferenceAssetId, input.voiceDescription]
|
|
2821
|
+
.filter((v) => typeof v === 'string' && v.length > 0);
|
|
2822
|
+
if (voiceSources.length === 0) {
|
|
2823
|
+
return ok({
|
|
2824
|
+
requires_clarification: true,
|
|
2825
|
+
missing: ['voiceId'],
|
|
2826
|
+
message: `${TTS_MODEL} needs a voice. Ask the user WHICH CHARACTER is speaking and pass that character's voice as voiceId — a voice is a field on a character, not a thing to pick at generation time. To make a NEW voice, pass voiceReferenceAssetId (a clip to clone) or voiceDescription (words, for a character with no recording).`,
|
|
2827
|
+
});
|
|
2828
|
+
}
|
|
2829
|
+
if (voiceSources.length > 1) {
|
|
2830
|
+
throw new Error(`${TTS_MODEL} takes exactly one voice source — pass voiceId OR voiceReferenceAssetId OR voiceDescription, not ${voiceSources.length} of them.`);
|
|
2831
|
+
}
|
|
2832
|
+
}
|
|
2417
2833
|
if (input.model === 'seed-audio') {
|
|
2418
|
-
if (seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
|
|
2834
|
+
if (seconds == null || seconds < SEED_AUDIO_MIN_SECONDS || seconds > SEED_AUDIO_MAX_SECONDS) {
|
|
2419
2835
|
return ok({
|
|
2420
2836
|
requires_clarification: true,
|
|
2421
2837
|
missing: ['durationSeconds'],
|
|
@@ -2427,7 +2843,7 @@ export const generateAudio = {
|
|
|
2427
2843
|
}
|
|
2428
2844
|
}
|
|
2429
2845
|
if (input.model === 'eleven-sfx') {
|
|
2430
|
-
if (seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
|
|
2846
|
+
if (seconds == null || seconds < ELEVEN_SFX_MIN_SECONDS || seconds > ELEVEN_SFX_MAX_SECONDS) {
|
|
2431
2847
|
return ok({
|
|
2432
2848
|
requires_clarification: true,
|
|
2433
2849
|
missing: ['durationSeconds'],
|
|
@@ -2443,6 +2859,10 @@ export const generateAudio = {
|
|
|
2443
2859
|
const refInputs = [];
|
|
2444
2860
|
for (const ref of input.audioReferenceAssetIds ?? [])
|
|
2445
2861
|
refInputs.push({ ref, role: 'audio reference' });
|
|
2862
|
+
// Resolved through the SAME resolver as every other asset ref, so a badge
|
|
2863
|
+
// code ("AUD-S1") works here exactly as it does everywhere else.
|
|
2864
|
+
if (input.voiceReferenceAssetId)
|
|
2865
|
+
refInputs.push({ ref: input.voiceReferenceAssetId, role: 'voice reference' });
|
|
2446
2866
|
if (input.imageReferenceAssetId)
|
|
2447
2867
|
refInputs.push({ ref: input.imageReferenceAssetId, role: 'image reference' });
|
|
2448
2868
|
const resolvedRefs = refInputs.length > 0
|
|
@@ -2457,6 +2877,7 @@ export const generateAudio = {
|
|
|
2457
2877
|
const costKey = audioCostKey({
|
|
2458
2878
|
model: input.model,
|
|
2459
2879
|
durationSeconds: seconds,
|
|
2880
|
+
characters: input.model === TTS_MODEL ? input.prompt.length : undefined,
|
|
2460
2881
|
});
|
|
2461
2882
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
2462
2883
|
if (!entry) {
|
|
@@ -2489,6 +2910,9 @@ export const generateAudio = {
|
|
|
2489
2910
|
prompt: input.prompt,
|
|
2490
2911
|
durationSeconds: seconds,
|
|
2491
2912
|
voice: input.voice,
|
|
2913
|
+
voiceId: input.voiceId,
|
|
2914
|
+
voiceReferenceAssetId: rid(input.voiceReferenceAssetId),
|
|
2915
|
+
voiceDescription: input.voiceDescription,
|
|
2492
2916
|
speed: input.speed,
|
|
2493
2917
|
volume: input.volume,
|
|
2494
2918
|
pitch: input.pitch,
|
|
@@ -2524,7 +2948,8 @@ export const generateAudio = {
|
|
|
2524
2948
|
// ── Generate lip-sync ───────────────────────────────────────────
|
|
2525
2949
|
export const generateLipSync = {
|
|
2526
2950
|
id: 'slates_generate_lip_sync',
|
|
2527
|
-
|
|
2951
|
+
billable: true,
|
|
2952
|
+
description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip, sourceType=image animates a still avatar (avatar-standard, or avatar-pro for the premium seat). Quote each with slates_estimate_generation_cost rather than from memory. Audio from TTS (ttsText + ttsVoice) or an uploaded file. Always 5 seconds. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
|
|
2528
2953
|
input: z.object({
|
|
2529
2954
|
projectId: z.string().uuid().describe('Slates project the source asset lives in. The new lip-synced video lands here.'),
|
|
2530
2955
|
sourceAssetId: z.string().uuid().describe('Asset id of the still image (avatar flow) or video clip (lip-sync flow). Must already exist in the project — use slates_upload_reference_image or slates_generate_image / slates_generate_video first if needed.'),
|
|
@@ -2643,12 +3068,14 @@ export const generateLipSync = {
|
|
|
2643
3068
|
// ── Generate motion transfer ────────────────────────────────────
|
|
2644
3069
|
export const generateMotionTransfer = {
|
|
2645
3070
|
id: 'slates_generate_motion_transfer',
|
|
2646
|
-
|
|
3071
|
+
billable: true,
|
|
3072
|
+
description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std or kling-mc-pro, structured skeleton/depth retargeting, always 5s. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. ' +
|
|
3073
|
+
CONFIRM_GATE_SENTENCE,
|
|
2647
3074
|
input: z.object({
|
|
2648
3075
|
projectId: z.string().uuid().describe('Slates project. Both source and target assets must live here.'),
|
|
2649
3076
|
sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s.'),
|
|
2650
3077
|
targetImageAssetId: z.string().uuid().describe('Asset id of the target image (the character that will perform the motion). Must already exist in the project.'),
|
|
2651
|
-
motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std
|
|
3078
|
+
motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std general motion; kling-mc-pro cleaner anatomy — default. Quote both with slates_estimate_generation_cost.'),
|
|
2652
3079
|
characterOrientation: z.enum(['video', 'image']).optional().describe('"video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
|
|
2653
3080
|
prompt: z.string().optional().describe('Optional refinement. Read slates-prompting-motion-transfer.'),
|
|
2654
3081
|
klingProvider: z.enum(['fal', 'kling']).optional().describe('Provider routing. "fal" (default) uses Slates credits.'),
|
|
@@ -2683,7 +3110,15 @@ export const generateMotionTransfer = {
|
|
|
2683
3110
|
target_ref: target,
|
|
2684
3111
|
message: `Cost: ${fmtCredits(totalCents)} for 5s ${motionModel} (${costKey}). ` +
|
|
2685
3112
|
`Transferring motion from ${source} onto ${target}. ` +
|
|
2686
|
-
|
|
3113
|
+
// The saving is READ from the registry, never guessed: "~10 credits"
|
|
3114
|
+
// was hand-typed and is a rate change away from being a lie.
|
|
3115
|
+
`Re-call with confirm=true after the user explicitly OKs the spend${motionModel === 'kling-mc-pro'
|
|
3116
|
+
? (() => {
|
|
3117
|
+
const std = registry.models.find((m) => m.model === 'kling-mc-std-5s');
|
|
3118
|
+
const saving = std ? totalCents - creditCost(std) : 0;
|
|
3119
|
+
return saving > 0 ? `, or pick kling-mc-std to save ${fmtCredits(saving)}` : '';
|
|
3120
|
+
})()
|
|
3121
|
+
: ''}. ` +
|
|
2687
3122
|
`When discussing with the user, refer to the assets by those codes — they'll match the gallery badges.`,
|
|
2688
3123
|
});
|
|
2689
3124
|
}
|
|
@@ -2737,13 +3172,27 @@ export const generateMotionTransfer = {
|
|
|
2737
3172
|
// ── Edit video (Kling O3 video-to-video) ────────────────────────
|
|
2738
3173
|
export const editVideo = {
|
|
2739
3174
|
id: 'slates_edit_video',
|
|
2740
|
-
|
|
3175
|
+
billable: true,
|
|
3176
|
+
description:
|
|
3177
|
+
// 🚨 NO PRICES, NO HAND-TYPED WINDOWS. This description carried five dollar
|
|
3178
|
+
// figures and three duration/resolution claims. The prices contradicted the
|
|
3179
|
+
// agent's own REAL NUMBERS ONLY rule (it may not repeat a figure it cannot
|
|
3180
|
+
// point to in a tool result) and go stale on the next rate change; the
|
|
3181
|
+
// windows are owned by MODEL_CAPABILITIES and are generated below into the
|
|
3182
|
+
// params that enforce them.
|
|
3183
|
+
'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default — subject/style refs via elements), omni-flash-edit (PROMPT-ONLY, no refs, the cheapest seat), or seedance-2.5-edit (the only engine that takes a clip over 15s; seedanceFace:true for AI-character faces). Clip-length and resolution windows per engine are on the `model` param. Cost is per second of OUTPUT (≈ clip length, rounded up), and an edit bills input + output seconds on every provider — read the quote from the confirm gate or slates_estimate_generation_cost, never from memory. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; Seedance edit/relocate for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
|
|
2741
3184
|
input: z.object({
|
|
2742
3185
|
projectId: z.string().uuid().describe('Project the source clip lives in.'),
|
|
2743
3186
|
sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
|
|
2744
3187
|
prompt: z.string().min(1).max(2500).describe('The change, not the whole scene — e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
|
|
2745
3188
|
// Clip windows and resolutions GENERATED from MODEL_CAPABILITIES.
|
|
2746
|
-
model: zEnum(EDIT_VIDEO_MODELS).optional().describe(
|
|
3189
|
+
model: zEnum(EDIT_VIDEO_MODELS).optional().describe(
|
|
3190
|
+
// Routing GENERATED from MODEL_FACTS; the paragraph that stood here was
|
|
3191
|
+
// hand-written and carried per-second prices, which drift and which the
|
|
3192
|
+
// agent's own REAL NUMBERS ONLY rule forbids it repeating.
|
|
3193
|
+
`Default kling-v3.0-omni-edit. Neither omni-flash-edit nor seedance-2.5-edit takes character/style refs on this op.\n` +
|
|
3194
|
+
`${describeRouting('video', 'edit')}\n` +
|
|
3195
|
+
`Source-clip length per engine: ${describeDurations(EDIT_VIDEO_MODELS)}. Output resolution: ${describeVideoResolutions(EDIT_VIDEO_MODELS)}`),
|
|
2747
3196
|
characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2748
3197
|
styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
|
|
2749
3198
|
keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
|
|
@@ -3431,7 +3880,12 @@ export const reorderScenes = {
|
|
|
3431
3880
|
};
|
|
3432
3881
|
export const updateFrame = {
|
|
3433
3882
|
id: 'slates_update_frame',
|
|
3434
|
-
description:
|
|
3883
|
+
description:
|
|
3884
|
+
// ⚠️ `frameType` and `motionPrompt` are GONE. They were a weaker duplicate
|
|
3885
|
+
// of what the Shot in this slot already encodes — the image's role and the
|
|
3886
|
+
// beat's words — and they were backfilled into Shots on 2026-08-31. Use
|
|
3887
|
+
// slates_update_shot for either.
|
|
3888
|
+
'Update a slot: its shot label, notes, bound asset (assetId=null unbinds), scene or position. The BEAT — its line, references, model, prompt and framing — lives on the Shot in this slot; use slates_update_shot for that.',
|
|
3435
3889
|
input: z.object({
|
|
3436
3890
|
frameId: z.string().uuid(),
|
|
3437
3891
|
shotLabel: z.string().optional(),
|
|
@@ -3439,8 +3893,6 @@ export const updateFrame = {
|
|
|
3439
3893
|
assetId: z.string().uuid().nullable().optional(),
|
|
3440
3894
|
sceneId: z.string().uuid().nullable().optional(),
|
|
3441
3895
|
position: z.number().int().min(0).optional(),
|
|
3442
|
-
frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
|
|
3443
|
-
motionPrompt: z.string().nullable().optional(),
|
|
3444
3896
|
}),
|
|
3445
3897
|
async run(input, ctx) {
|
|
3446
3898
|
return ok(await ctx.desktop().post('/agent/frames/update', {
|
|
@@ -3451,8 +3903,6 @@ export const updateFrame = {
|
|
|
3451
3903
|
assetId: input.assetId,
|
|
3452
3904
|
sceneId: input.sceneId,
|
|
3453
3905
|
position: input.position,
|
|
3454
|
-
frameType: input.frameType,
|
|
3455
|
-
motionPrompt: input.motionPrompt,
|
|
3456
3906
|
},
|
|
3457
3907
|
}));
|
|
3458
3908
|
},
|
|
@@ -3472,7 +3922,7 @@ export const updateFrame = {
|
|
|
3472
3922
|
*/
|
|
3473
3923
|
export const batchUpdateFrames = {
|
|
3474
3924
|
id: 'slates_batch_update_frames',
|
|
3475
|
-
description: 'Update MANY
|
|
3925
|
+
description: 'Update MANY slots in one call — shot labels, notes, asset binding, scene/position. Prefer this over repeated slates_update_frame when re-arranging a scene: it is one round-trip and one UI refresh, and every id is validated before anything is written, so the batch never lands half-applied. To write the BEATS themselves, use slates_create_shot / slates_update_shot.',
|
|
3476
3926
|
input: z.object({
|
|
3477
3927
|
updates: z
|
|
3478
3928
|
.array(z.object({
|
|
@@ -3482,8 +3932,6 @@ export const batchUpdateFrames = {
|
|
|
3482
3932
|
assetId: z.string().uuid().nullable().optional(),
|
|
3483
3933
|
sceneId: z.string().uuid().nullable().optional(),
|
|
3484
3934
|
position: z.number().int().min(0).optional(),
|
|
3485
|
-
frameType: z.enum(['first', 'last', 'ingredient']).nullable().optional(),
|
|
3486
|
-
motionPrompt: z.string().nullable().optional(),
|
|
3487
3935
|
}))
|
|
3488
3936
|
.min(1),
|
|
3489
3937
|
}),
|
|
@@ -3497,8 +3945,6 @@ export const batchUpdateFrames = {
|
|
|
3497
3945
|
assetId: u.assetId,
|
|
3498
3946
|
sceneId: u.sceneId,
|
|
3499
3947
|
position: u.position,
|
|
3500
|
-
frameType: u.frameType,
|
|
3501
|
-
motionPrompt: u.motionPrompt,
|
|
3502
3948
|
},
|
|
3503
3949
|
})),
|
|
3504
3950
|
}));
|
|
@@ -3515,6 +3961,763 @@ export const deleteFrame = {
|
|
|
3515
3961
|
// ── Prompting guides (local lookup — no transport) ──────────────
|
|
3516
3962
|
// Model-id → guide-name aliasing. Order matters: kling-mc-* (motion
|
|
3517
3963
|
// transfer) must match before the generic kling-v3* check.
|
|
3964
|
+
// ── Shots — the prompt bar, serialized ──────────────────────────
|
|
3965
|
+
//
|
|
3966
|
+
// A Shot is a NAMED generation recipe: what to make, with what, on which model,
|
|
3967
|
+
// at what settings. It exists to remove one pipeline constraint from this
|
|
3968
|
+
// surface — `slates_add_frame` requires a non-null `assetId`, so an agent could
|
|
3969
|
+
// not plan a shot before its image existed. A Shot takes no asset at all.
|
|
3970
|
+
//
|
|
3971
|
+
// 🚨 IT STORES A RAW PROMPT AND REFERENCES, NEVER A COMPOSED PROMPT. The
|
|
3972
|
+
// composer is the only thing that numbers anything; a stored "image 3" is a lie
|
|
3973
|
+
// the moment a reference moves. `slates_get_shot` returns the composed prompt so
|
|
3974
|
+
// the agent can audit its own work through the exact resolver the request uses.
|
|
3975
|
+
/** Role → its `string[]` param, GENERATED from the role list. Never hand-typed:
|
|
3976
|
+
* `attachmentRoles`/`shot-spec` is the ONE role list, and a sixth role has to
|
|
3977
|
+
* appear here without anyone remembering to add it. */
|
|
3978
|
+
function shotRefShape(described) {
|
|
3979
|
+
return Object.fromEntries(ORDERED_ATTACHMENT_ROLES.map((role) => [
|
|
3980
|
+
role,
|
|
3981
|
+
described
|
|
3982
|
+
? z
|
|
3983
|
+
.array(z.string())
|
|
3984
|
+
.optional()
|
|
3985
|
+
.describe(`${ATTACHMENT_ROLE_DESCRIPTION[role]} UUIDs or badge codes ("IMG-A8").`)
|
|
3986
|
+
: z.array(z.string()).optional(),
|
|
3987
|
+
]));
|
|
3988
|
+
}
|
|
3989
|
+
const SHOT_REFS_LEAD = 'Attachments by ROLE, ordered within each role. The role decides the sentence the model is told, so a subject reference and a plain one are not interchangeable.';
|
|
3990
|
+
const shotRefsSchema = z.object(shotRefShape(true)).optional().describe(SHOT_REFS_LEAD);
|
|
3991
|
+
/**
|
|
3992
|
+
* 🚨 THE SAME SHAPE, DESCRIBED ONCE.
|
|
3993
|
+
*
|
|
3994
|
+
* `params`, `refs` and the script fields are identical across create / update /
|
|
3995
|
+
* duplicate, and `zodToJsonSchema` inlines every description into all three —
|
|
3996
|
+
* 2.7 KB of the same prose, three times, in the desktop's prompt-cached prefix
|
|
3997
|
+
* on every turn. `slates_create_shot` is the op that documents the shape and it
|
|
3998
|
+
* is always in context beside these; repeating the table here bought nothing
|
|
3999
|
+
* but bytes. The Zod ENUMS stay on both, so enforcement is unchanged — only the
|
|
4000
|
+
* prose is deduplicated.
|
|
4001
|
+
*/
|
|
4002
|
+
const SEE_CREATE_SHOT = 'Same shape as slates_create_shot — see it for what each field means. ';
|
|
4003
|
+
const shotRefsSchemaTerse = z
|
|
4004
|
+
.object(shotRefShape(false))
|
|
4005
|
+
.optional()
|
|
4006
|
+
.describe(SEE_CREATE_SHOT + SHOT_REFS_LEAD);
|
|
4007
|
+
// The full ratio vocabulary a Shot can hold — image OR video, because a Shot is
|
|
4008
|
+
// any generation. Per-model narrowing happens at create time for video (the same
|
|
4009
|
+
// `assertVideoCapabilities` gate `slates_generate_video` uses) and at generate
|
|
4010
|
+
// time for everything.
|
|
4011
|
+
const SHOT_ASPECT_RATIOS = [...new Set([...VIDEO_ASPECT_RATIOS, ...IMAGE_ASPECT_RATIOS])];
|
|
4012
|
+
// 🚨 THE PER-MODEL CAPABILITY TABLES ARE DELIBERATELY NOT REPEATED HERE.
|
|
4013
|
+
// `slates_generate_video`'s param descriptions already carry them and are always
|
|
4014
|
+
// in context on both surfaces; embedding them again — in three shot ops, each
|
|
4015
|
+
// taking these same params — would put FOUR more copies of a table that grows on
|
|
4016
|
+
// every model addition into the desktop's cached prefix. Measured: it was 15.3%
|
|
4017
|
+
// of the whole tool surface, almost all of it those three strings.
|
|
4018
|
+
// The vocabulary is still ENFORCED (a Zod enum built from MODEL_CAPABILITIES),
|
|
4019
|
+
// and the per-model narrowing is enforced by `assertShotCapabilities` at save
|
|
4020
|
+
// time — which is stronger than prose, not weaker.
|
|
4021
|
+
function shotParamsShape(described) {
|
|
4022
|
+
const d = (node, text) => (described ? node.describe(text) : node);
|
|
4023
|
+
return {
|
|
4024
|
+
aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the chosen model when the Shot is saved — see slates_generate_video for the per-model sets.'),
|
|
4025
|
+
duration: d(z.number().int().min(1).max(360).optional(), 'Seconds — video or audio. Required before a video Shot can be priced or fired; validated against the model when saved.'),
|
|
4026
|
+
videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the chosen model when the Shot is saved.'),
|
|
4027
|
+
imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
|
|
4028
|
+
gptQuality: d(z.enum(['medium', 'high']).optional(), 'gpt-image-2 only.'),
|
|
4029
|
+
imageQuantity: d(z.number().int().min(1).max(4).optional(), 'Image models only — how many to make per fire.'),
|
|
4030
|
+
negativePrompt: z.string().optional(),
|
|
4031
|
+
sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
|
|
4032
|
+
seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider at ~45% more."),
|
|
4033
|
+
audioDurationSeconds: d(z.number().int().min(1).max(120).optional(), 'Audio lane. On seed-audio the requested duration IS the bill.'),
|
|
4034
|
+
};
|
|
4035
|
+
}
|
|
4036
|
+
const shotParamsSchema = z.object(shotParamsShape(true)).optional();
|
|
4037
|
+
const shotParamsSchemaTerse = z
|
|
4038
|
+
.object(shotParamsShape(false))
|
|
4039
|
+
.optional()
|
|
4040
|
+
.describe(SEE_CREATE_SHOT + 'Generation settings for the Shot.');
|
|
4041
|
+
/**
|
|
4042
|
+
* The script layer, as op params — GENERATED from `SCRIPT_FIELD_DESCRIPTION`.
|
|
4043
|
+
*
|
|
4044
|
+
* 🚨 THE PROSE IS NEVER HAND-TYPED HERE. `shot-spec.ts` owns what each field
|
|
4045
|
+
* MEANS, with `satisfies Record<ScriptField, string>` making a ninth field a
|
|
4046
|
+
* compile error in the descriptions too. An op that spelled these out would
|
|
4047
|
+
* ship a field with no explanation — the same failure as a column nothing
|
|
4048
|
+
* renders.
|
|
4049
|
+
*
|
|
4050
|
+
* 🚨 AND NONE OF THEM IS SENT TO A MODEL. They are a planning and counting
|
|
4051
|
+
* surface; the prompt is the only thing the request carries. The one exception
|
|
4052
|
+
* is pre-existing: a multiShotSegment still prepends its own camera and
|
|
4053
|
+
* shotSize to its own segment prompt.
|
|
4054
|
+
*/
|
|
4055
|
+
function shotScriptShape(described) {
|
|
4056
|
+
const text = Object.fromEntries(SCRIPT_TEXT_FIELDS.map((field) => [
|
|
4057
|
+
field,
|
|
4058
|
+
described
|
|
4059
|
+
? z.string().max(2000).nullable().optional().describe(SCRIPT_FIELD_DESCRIPTION[field])
|
|
4060
|
+
: z.string().max(2000).nullable().optional(),
|
|
4061
|
+
]));
|
|
4062
|
+
return {
|
|
4063
|
+
...text,
|
|
4064
|
+
continues: described
|
|
4065
|
+
? z.boolean().optional().describe(SCRIPT_FIELD_DESCRIPTION.continues)
|
|
4066
|
+
: z.boolean().optional(),
|
|
4067
|
+
};
|
|
4068
|
+
}
|
|
4069
|
+
const shotScriptSchema = shotScriptShape(true);
|
|
4070
|
+
/** Same fields, described on `slates_create_shot` only — see SEE_CREATE_SHOT. */
|
|
4071
|
+
const shotScriptSchemaTerse = shotScriptShape(false);
|
|
4072
|
+
/** The framing vocabulary an agent should know about — generated from the
|
|
4073
|
+
* bucket lists so a seventh bucket cannot ship undescribed, and worded so it
|
|
4074
|
+
* is unmistakably a COUNTING aid rather than a closed set. */
|
|
4075
|
+
const FRAMING_NOTE = `shotSize and camera are FREE TEXT and are never rejected or rewritten. ` +
|
|
4076
|
+
`They are bucketed only for the variety count — shot size into ` +
|
|
4077
|
+
`${SHOT_SIZE_BUCKETS.join(' / ')}, camera into ${CAMERA_MOVE_BUCKETS.join(' / ')} — ` +
|
|
4078
|
+
`and anything unrecognised counts as "other", which is a fine answer.`;
|
|
4079
|
+
/** What the caller actually sent for the script half, by omission. Same rule
|
|
4080
|
+
* as `shotParamsPatch`: an `undefined` value is not an absent key, and a
|
|
4081
|
+
* spread of undefineds is silent data loss on a merging route. */
|
|
4082
|
+
function shotScriptPatch(input) {
|
|
4083
|
+
const out = {};
|
|
4084
|
+
const raw = input;
|
|
4085
|
+
for (const field of SCRIPT_TEXT_FIELDS) {
|
|
4086
|
+
if (raw[field] !== undefined)
|
|
4087
|
+
out[field] = raw[field];
|
|
4088
|
+
}
|
|
4089
|
+
if (input.continues !== undefined)
|
|
4090
|
+
out.continues = input.continues;
|
|
4091
|
+
return out;
|
|
4092
|
+
}
|
|
4093
|
+
/**
|
|
4094
|
+
* The `params` half of a spec patch — ONLY the keys the caller actually named.
|
|
4095
|
+
*
|
|
4096
|
+
* 🚨 AN `undefined` VALUE IS NOT AN ABSENT KEY, AND THE DIFFERENCE IS SILENT
|
|
4097
|
+
* DATA LOSS. `/agent/shots/update` merges `params` one level deep so that
|
|
4098
|
+
* "anything you omit is left exactly as it was" — but a spread copies keys
|
|
4099
|
+
* whose value is `undefined` too, so a hand-built object listing all ten fields
|
|
4100
|
+
* overwrote the nine the caller never mentioned with `undefined`, and the
|
|
4101
|
+
* tolerant reader on the far side then dropped them. `params: { duration: 9 }`
|
|
4102
|
+
* silently cleared the aspect ratio, the resolution, the negative prompt and
|
|
4103
|
+
* the face route. Building the object by omission is what keeps that promise.
|
|
4104
|
+
*
|
|
4105
|
+
* ONE builder, three callers (create / update / duplicate) — the ten field
|
|
4106
|
+
* names were hand-listed twice before this, which is the same near-miss in
|
|
4107
|
+
* waiting.
|
|
4108
|
+
*/
|
|
4109
|
+
function shotParamsPatch(p) {
|
|
4110
|
+
const out = {};
|
|
4111
|
+
if (!p)
|
|
4112
|
+
return out;
|
|
4113
|
+
for (const [k, v] of Object.entries(p))
|
|
4114
|
+
if (v !== undefined)
|
|
4115
|
+
out[k] = v;
|
|
4116
|
+
return out;
|
|
4117
|
+
}
|
|
4118
|
+
/**
|
|
4119
|
+
* `audioRefSpokenText` pairs POSITIONALLY with the audio references, so it can
|
|
4120
|
+
* only be sent alongside them.
|
|
4121
|
+
*
|
|
4122
|
+
* Refused rather than truncated or silently re-keyed: the whole point of the
|
|
4123
|
+
* field is that the WORDS are exact, and a misaligned array attaches one clip's
|
|
4124
|
+
* line to another clip with nothing on screen to say so. Same rule
|
|
4125
|
+
* `slates_generate_video` already applies.
|
|
4126
|
+
*/
|
|
4127
|
+
function checkSpokenTextAlignment(input) {
|
|
4128
|
+
if (input.audioRefSpokenText === undefined)
|
|
4129
|
+
return null;
|
|
4130
|
+
const clips = input.refs?.['audio-reference'];
|
|
4131
|
+
if (!clips || clips.length !== input.audioRefSpokenText.length) {
|
|
4132
|
+
return ok({
|
|
4133
|
+
requires_clarification: true,
|
|
4134
|
+
missing: ['refs["audio-reference"]'],
|
|
4135
|
+
message: `audioRefSpokenText pairs by position with refs["audio-reference"], so send both together ` +
|
|
4136
|
+
`and at the same length (${input.audioRefSpokenText.length} line(s) vs ${clips?.length ?? 0} clip(s)). ` +
|
|
4137
|
+
`Use "" for a clip with no speech.`,
|
|
4138
|
+
});
|
|
4139
|
+
}
|
|
4140
|
+
return null;
|
|
4141
|
+
}
|
|
4142
|
+
/** Every asset reference a Shot input carries, flat, for one `resolveAssetRefs`
|
|
4143
|
+
* pass. Derived from the role list so a new role resolves badge codes too. */
|
|
4144
|
+
function shotRefInputs(input) {
|
|
4145
|
+
const out = [];
|
|
4146
|
+
for (const role of ORDERED_ATTACHMENT_ROLES) {
|
|
4147
|
+
for (const r of input.refs?.[role] ?? [])
|
|
4148
|
+
out.push({ ref: r, role });
|
|
4149
|
+
}
|
|
4150
|
+
if (input.firstFrameAssetId)
|
|
4151
|
+
out.push({ ref: input.firstFrameAssetId, role: 'first frame' });
|
|
4152
|
+
if (input.lastFrameAssetId)
|
|
4153
|
+
out.push({ ref: input.lastFrameAssetId, role: 'last frame' });
|
|
4154
|
+
return out;
|
|
4155
|
+
}
|
|
4156
|
+
/**
|
|
4157
|
+
* Op input → the `ShotSpec` shape the desktop stores.
|
|
4158
|
+
*
|
|
4159
|
+
* Badge codes resolve to ids HERE, at call time, against the project as it
|
|
4160
|
+
* stands — never a mapping remembered from earlier in the conversation. A Shot
|
|
4161
|
+
* that recorded a guessed id would look fine in a listing and compose to
|
|
4162
|
+
* something else entirely.
|
|
4163
|
+
*/
|
|
4164
|
+
async function buildShotSpecInput(ctx, projectId, input) {
|
|
4165
|
+
const refInputs = shotRefInputs(input);
|
|
4166
|
+
const resolvedRefs = await resolveAssetRefs(ctx, projectId, refInputs.map((r) => r.ref));
|
|
4167
|
+
const rid = (v) => (v ? (resolvedRefs.get(v)?.id ?? v) : null);
|
|
4168
|
+
const refs = {};
|
|
4169
|
+
for (const role of ORDERED_ATTACHMENT_ROLES) {
|
|
4170
|
+
refs[role] = (input.refs?.[role] ?? []).map((v) => resolvedRefs.get(v)?.id ?? v);
|
|
4171
|
+
}
|
|
4172
|
+
// Positional in, KEYED out — the same re-keying `slates_generate_video` does,
|
|
4173
|
+
// and for the same reason: an index means different clips depending on how the
|
|
4174
|
+
// list was assembled, while an asset id cannot drift.
|
|
4175
|
+
const audioIds = refs['audio-reference'] ?? [];
|
|
4176
|
+
const spoken = {};
|
|
4177
|
+
(input.audioRefSpokenText ?? []).forEach((text, i) => {
|
|
4178
|
+
const id = audioIds[i];
|
|
4179
|
+
if (id && text?.trim())
|
|
4180
|
+
spoken[id] = text.trim();
|
|
4181
|
+
});
|
|
4182
|
+
return {
|
|
4183
|
+
spec: {
|
|
4184
|
+
prompt: input.prompt,
|
|
4185
|
+
model: input.model ?? null,
|
|
4186
|
+
// The prompt is recorded as written FOR this model. Swapping the model
|
|
4187
|
+
// later diverges from it and the desktop card says so — the prompt is
|
|
4188
|
+
// never rewritten (that is prompt enhancement, deleted 2026-08-01).
|
|
4189
|
+
authoredFor: input.model ?? null,
|
|
4190
|
+
params: shotParamsPatch(input.params),
|
|
4191
|
+
mentions: {
|
|
4192
|
+
characterIds: input.characterIds ?? [],
|
|
4193
|
+
environmentIds: input.environmentIds ?? [],
|
|
4194
|
+
styleIds: input.styleIds ?? [],
|
|
4195
|
+
},
|
|
4196
|
+
refs,
|
|
4197
|
+
firstFrameAssetId: rid(input.firstFrameAssetId),
|
|
4198
|
+
lastFrameAssetId: rid(input.lastFrameAssetId),
|
|
4199
|
+
audioRefSpokenText: spoken,
|
|
4200
|
+
},
|
|
4201
|
+
refEcho: describeResolvedRefs(refInputs, resolvedRefs),
|
|
4202
|
+
};
|
|
4203
|
+
}
|
|
4204
|
+
/**
|
|
4205
|
+
* The capability gate, applied to a SAVED recipe.
|
|
4206
|
+
*
|
|
4207
|
+
* It enforces exactly what the matching generate op enforces and no more: video
|
|
4208
|
+
* goes through `assertVideoCapabilities` (the MODEL_CAPABILITIES SSOT), image
|
|
4209
|
+
* does not, because `slates_generate_image`'s own per-model ratio check is the
|
|
4210
|
+
* named open follow-up in the capability plan. A Shot that refused what
|
|
4211
|
+
* `slates_generate_image` accepts would be a THIRD opinion about the same
|
|
4212
|
+
* model, which is worse than the gap.
|
|
4213
|
+
*/
|
|
4214
|
+
function assertShotCapabilities(model, params) {
|
|
4215
|
+
if (!model || !VIDEO_MODELS.includes(model))
|
|
4216
|
+
return null;
|
|
4217
|
+
const err = assertVideoCapabilities({
|
|
4218
|
+
model,
|
|
4219
|
+
aspectRatio: params?.aspectRatio,
|
|
4220
|
+
videoResolution: params?.videoResolution,
|
|
4221
|
+
duration: params?.duration,
|
|
4222
|
+
});
|
|
4223
|
+
return err ? ok(err) : null;
|
|
4224
|
+
}
|
|
4225
|
+
/**
|
|
4226
|
+
* The registry cost key for a saved Shot, through the SAME builders every quote
|
|
4227
|
+
* in this file uses (`videoCostKey` / `imageCostKey` / `audioCostKey`).
|
|
4228
|
+
*
|
|
4229
|
+
* Returns null when the Shot cannot be priced — no model, or a model this
|
|
4230
|
+
* surface does not carry. The caller REPORTS that rather than quoting zero: a
|
|
4231
|
+
* missing price displayed as free is the failure mode the whole pricing
|
|
4232
|
+
* contract exists to prevent.
|
|
4233
|
+
*/
|
|
4234
|
+
function shotCostKey(detail) {
|
|
4235
|
+
const model = detail.model;
|
|
4236
|
+
if (!model)
|
|
4237
|
+
return null;
|
|
4238
|
+
const p = detail.params;
|
|
4239
|
+
// Prefer the CLAMPED values the desktop will actually fire with; a listing row
|
|
4240
|
+
// has none, so it falls back to the raw ones and is announced as a floor.
|
|
4241
|
+
const fires = detail.firesWith;
|
|
4242
|
+
if (AUDIO_MODELS.includes(model)) {
|
|
4243
|
+
// The TTS seat prices on the TEXT, and a Shot carries no text field — so a
|
|
4244
|
+
// Shot cannot be a TTS generation and cannot be quoted as one. Explicit,
|
|
4245
|
+
// because the `!seconds` line below would also return null here and that
|
|
4246
|
+
// would read as "duration missing" for a surface that has no duration.
|
|
4247
|
+
if (model === TTS_MODEL)
|
|
4248
|
+
return null;
|
|
4249
|
+
const seconds = fires?.audioDurationSeconds ?? p.audioDurationSeconds;
|
|
4250
|
+
if (!seconds)
|
|
4251
|
+
return null;
|
|
4252
|
+
return audioCostKey({ model: model, durationSeconds: seconds });
|
|
4253
|
+
}
|
|
4254
|
+
if (VIDEO_MODELS.includes(model)) {
|
|
4255
|
+
const duration = fires?.duration ?? p.duration;
|
|
4256
|
+
if (!duration)
|
|
4257
|
+
return null;
|
|
4258
|
+
const billed = (d) => (d > 0 ? Math.ceil(d - 0.05) : 0);
|
|
4259
|
+
return videoCostKey({
|
|
4260
|
+
model: model,
|
|
4261
|
+
duration,
|
|
4262
|
+
videoResolution: fires?.videoResolution ??
|
|
4263
|
+
p.videoResolution ??
|
|
4264
|
+
defaultVideoResolutionFor(model),
|
|
4265
|
+
sound: p.sound,
|
|
4266
|
+
seedanceFace: p.seedanceFace,
|
|
4267
|
+
// `references` is absent on a LISTING row (it does not compose), so both
|
|
4268
|
+
// of these read 0 there. That is why a listing quote is announced as a
|
|
4269
|
+
// floor and `slates_get_shot` is the exact one.
|
|
4270
|
+
referenceImages: (detail.references ?? []).filter((r) => r.kind === 'image').length,
|
|
4271
|
+
videoRefSeconds: (detail.references ?? [])
|
|
4272
|
+
.filter((r) => r.kind === 'video')
|
|
4273
|
+
.reduce((n, r) => n + billed(r.durationSeconds ?? 0), 0),
|
|
4274
|
+
});
|
|
4275
|
+
}
|
|
4276
|
+
if (IMAGE_MODELS.includes(model)) {
|
|
4277
|
+
return imageCostKey(model, (fires?.imageResolution ?? p.imageResolution) ??
|
|
4278
|
+
(model === 'nano-banana-2-lite' ? '1k' : '2k'), p.gptQuality ?? 'medium');
|
|
4279
|
+
}
|
|
4280
|
+
return null;
|
|
4281
|
+
}
|
|
4282
|
+
/** Credits for one Shot, and how many generations it fires.
|
|
4283
|
+
*
|
|
4284
|
+
* `imageQuantity` multiplies IMAGE models only — the same condition the
|
|
4285
|
+
* desktop's `estimateCostFor` applies and the only lane `/agent/shots/*` sends
|
|
4286
|
+
* a `count` for. Multiplying it blindly would quote a video Shot 3× for a
|
|
4287
|
+
* param its request never carries, and the card beside it would say ×1. */
|
|
4288
|
+
function shotQuote(detail, byKey) {
|
|
4289
|
+
const key = shotCostKey(detail);
|
|
4290
|
+
const isImage = !!detail.model && IMAGE_MODELS.includes(detail.model);
|
|
4291
|
+
const quantity = isImage ? (detail.params.imageQuantity ?? 1) || 1 : 1;
|
|
4292
|
+
const per = key != null ? byKey.get(key) : undefined;
|
|
4293
|
+
return { key, credits: (per ?? 0) * quantity, quantity };
|
|
4294
|
+
}
|
|
4295
|
+
export const createShot = {
|
|
4296
|
+
id: 'slates_create_shot',
|
|
4297
|
+
description: 'Write one beat of the piece — a Shot: its script line, its references with their roles, its model and params, and the prompt that fires. It needs NO image to exist, so a whole film can be written, arranged and priced before anything is generated. It lands in the storyboard automatically (the open scene, else the most recent storyboard) — never unfiled. ' +
|
|
4298
|
+
FRAMING_NOTE,
|
|
4299
|
+
input: z.object({
|
|
4300
|
+
projectId: z.string().uuid(),
|
|
4301
|
+
name: z.string().max(120).optional().describe('What to call it. Shown on the card; the prompt supplies one if you omit it.'),
|
|
4302
|
+
prompt: z.string().min(1).max(4000).describe('The RAW prompt, @mentions intact. Never write "image 1" yourself — the composer numbers references, and a hand-written number is wrong the moment one moves.'),
|
|
4303
|
+
model: z.string().optional().describe('Model id — the same ids slates_generate_image / slates_generate_video / slates_generate_audio take, and their descriptions carry the routing. Optional: a Shot can be planned before the model is decided.'),
|
|
4304
|
+
params: shotParamsSchema,
|
|
4305
|
+
refs: shotRefsSchema,
|
|
4306
|
+
firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video (UUID or badge code).'),
|
|
4307
|
+
lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code).'),
|
|
4308
|
+
audioRefSpokenText: z.array(z.string()).optional().describe('Same order and length as refs["audio-reference"]; use "" for a clip with no words. The model RE-TRANSCRIBES a supplied take, so only text decides the words.'),
|
|
4309
|
+
characterIds: z.array(z.string().uuid()).optional().describe('Characters the prompt @mentions — stored as ENTITY ids, so updating the character updates every Shot that names it.'),
|
|
4310
|
+
environmentIds: z.array(z.string().uuid()).optional(),
|
|
4311
|
+
styleIds: z.array(z.string().uuid()).optional(),
|
|
4312
|
+
frameId: z.string().uuid().optional().describe('Put it in this exact frame. Optional — omit it and the Shot files itself into a scene, creating a storyboard named after the project if there is none.'),
|
|
4313
|
+
sceneId: z.string().uuid().optional().describe('File it into this scene. Optional; ignored when frameId is given.'),
|
|
4314
|
+
...shotScriptSchema,
|
|
4315
|
+
}),
|
|
4316
|
+
async run(input, ctx) {
|
|
4317
|
+
const capErr = assertShotCapabilities(input.model, input.params);
|
|
4318
|
+
if (capErr)
|
|
4319
|
+
return capErr;
|
|
4320
|
+
const alignErr = checkSpokenTextAlignment(input);
|
|
4321
|
+
if (alignErr)
|
|
4322
|
+
return alignErr;
|
|
4323
|
+
const desktop = ctx.desktop();
|
|
4324
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4325
|
+
const { spec, refEcho } = await buildShotSpecInput(ctx, input.projectId, input);
|
|
4326
|
+
const r = await desktop.post('/agent/shots', {
|
|
4327
|
+
projectId: input.projectId,
|
|
4328
|
+
name: input.name,
|
|
4329
|
+
spec: { ...spec, ...shotScriptPatch(input) },
|
|
4330
|
+
frameId: input.frameId ?? null,
|
|
4331
|
+
sceneId: input.sceneId ?? null,
|
|
4332
|
+
});
|
|
4333
|
+
// The CODE is the address the user sees on the row — say it back so the
|
|
4334
|
+
// next call, and the next sentence to the user, can point at it.
|
|
4335
|
+
return ok(r.shot, `${r.shot?.code || 'Shot'} — "${r.shot?.name || 'Untitled'}". ${refEcho}`.trim());
|
|
4336
|
+
},
|
|
4337
|
+
};
|
|
4338
|
+
export const updateShot = {
|
|
4339
|
+
id: 'slates_update_shot',
|
|
4340
|
+
description: 'Change part of a Shot, or attach/detach it from a storyboard frame — anything you omit is left exactly as it was. ' +
|
|
4341
|
+
FRAMING_NOTE,
|
|
4342
|
+
input: z.object({
|
|
4343
|
+
shotId: z.string().describe('The Shot id, or its SHOT-A code as shown on the row.'),
|
|
4344
|
+
projectId: z.string().uuid().describe('The Shot\'s project — badge codes and entity ids resolve against it.'),
|
|
4345
|
+
name: z.string().max(120).optional(),
|
|
4346
|
+
prompt: z.string().max(4000).optional(),
|
|
4347
|
+
model: z.string().optional(),
|
|
4348
|
+
params: shotParamsSchemaTerse,
|
|
4349
|
+
refs: shotRefsSchemaTerse,
|
|
4350
|
+
firstFrameAssetId: z.string().optional(),
|
|
4351
|
+
lastFrameAssetId: z.string().optional(),
|
|
4352
|
+
audioRefSpokenText: z.array(z.string()).optional(),
|
|
4353
|
+
characterIds: z.array(z.string().uuid()).optional(),
|
|
4354
|
+
environmentIds: z.array(z.string().uuid()).optional(),
|
|
4355
|
+
styleIds: z.array(z.string().uuid()).optional(),
|
|
4356
|
+
attachFrameId: z.string().uuid().optional().describe('Attach this Shot to a storyboard frame.'),
|
|
4357
|
+
detachFrameId: z.string().uuid().optional().describe('Detach it from a frame. The Shot itself survives.'),
|
|
4358
|
+
posterAssetId: z.string().nullable().optional().describe('Which reference represents this Shot as a thumbnail. Defaulted automatically (first frame, else the first image reference, else the newest take) — only set it to OVERRIDE, and pass null to go back to the default.'),
|
|
4359
|
+
...shotScriptSchemaTerse,
|
|
4360
|
+
}),
|
|
4361
|
+
async run(input, ctx) {
|
|
4362
|
+
const capErr = assertShotCapabilities(input.model, input.params);
|
|
4363
|
+
if (capErr)
|
|
4364
|
+
return capErr;
|
|
4365
|
+
const alignErr = checkSpokenTextAlignment(input);
|
|
4366
|
+
if (alignErr)
|
|
4367
|
+
return alignErr;
|
|
4368
|
+
const desktop = ctx.desktop();
|
|
4369
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4370
|
+
const { spec } = await buildShotSpecInput(ctx, input.projectId, input);
|
|
4371
|
+
// Only send the halves the caller actually named; the route merges a PARTIAL
|
|
4372
|
+
// spec over the stored one, so an omitted field is never silently cleared.
|
|
4373
|
+
const patch = {};
|
|
4374
|
+
if (input.prompt !== undefined)
|
|
4375
|
+
patch.prompt = input.prompt;
|
|
4376
|
+
if (input.model !== undefined) {
|
|
4377
|
+
patch.model = input.model;
|
|
4378
|
+
// 🚨 `authoredFor` is NOT re-stamped on a model swap. The whole point of
|
|
4379
|
+
// recording it is that the prompt stays written for the model it was
|
|
4380
|
+
// written for — the grammars genuinely differ — so the card can say so.
|
|
4381
|
+
}
|
|
4382
|
+
if (input.params !== undefined)
|
|
4383
|
+
patch.params = spec.params;
|
|
4384
|
+
// 🚨 ONLY THE ROLES THE CALLER NAMED. `buildShotSpecInput` always returns a
|
|
4385
|
+
// COMPLETE refs record, and the route merges one level deep — so sending all
|
|
4386
|
+
// five would clear every role the caller never mentioned. Clearing a role is
|
|
4387
|
+
// explicit: send `[]`.
|
|
4388
|
+
if (input.refs !== undefined) {
|
|
4389
|
+
const built = spec.refs;
|
|
4390
|
+
const named = {};
|
|
4391
|
+
for (const role of ORDERED_ATTACHMENT_ROLES) {
|
|
4392
|
+
if (input.refs[role] !== undefined)
|
|
4393
|
+
named[role] = built[role];
|
|
4394
|
+
}
|
|
4395
|
+
if (Object.keys(named).length > 0)
|
|
4396
|
+
patch.refs = named;
|
|
4397
|
+
}
|
|
4398
|
+
if (input.firstFrameAssetId !== undefined)
|
|
4399
|
+
patch.firstFrameAssetId = spec.firstFrameAssetId;
|
|
4400
|
+
if (input.lastFrameAssetId !== undefined)
|
|
4401
|
+
patch.lastFrameAssetId = spec.lastFrameAssetId;
|
|
4402
|
+
if (input.audioRefSpokenText !== undefined)
|
|
4403
|
+
patch.audioRefSpokenText = spec.audioRefSpokenText;
|
|
4404
|
+
Object.assign(patch, shotScriptPatch(input));
|
|
4405
|
+
// Same rule for the three mention lists — naming one must not clear the
|
|
4406
|
+
// other two.
|
|
4407
|
+
{
|
|
4408
|
+
const built = spec.mentions;
|
|
4409
|
+
const named = {};
|
|
4410
|
+
if (input.characterIds !== undefined)
|
|
4411
|
+
named.characterIds = built.characterIds;
|
|
4412
|
+
if (input.environmentIds !== undefined)
|
|
4413
|
+
named.environmentIds = built.environmentIds;
|
|
4414
|
+
if (input.styleIds !== undefined)
|
|
4415
|
+
named.styleIds = built.styleIds;
|
|
4416
|
+
if (Object.keys(named).length > 0)
|
|
4417
|
+
patch.mentions = named;
|
|
4418
|
+
}
|
|
4419
|
+
const r = await desktop.post('/agent/shots/update', {
|
|
4420
|
+
id: input.shotId,
|
|
4421
|
+
data: {
|
|
4422
|
+
name: input.name,
|
|
4423
|
+
...(Object.keys(patch).length > 0 ? { spec: patch } : {}),
|
|
4424
|
+
attachFrameId: input.attachFrameId,
|
|
4425
|
+
detachFrameId: input.detachFrameId,
|
|
4426
|
+
posterAssetId: input.posterAssetId,
|
|
4427
|
+
},
|
|
4428
|
+
});
|
|
4429
|
+
return ok(r.shot);
|
|
4430
|
+
},
|
|
4431
|
+
};
|
|
4432
|
+
export const duplicateShot = {
|
|
4433
|
+
id: 'slates_duplicate_shot',
|
|
4434
|
+
description: 'Fork a saved Shot, changing the prompt, the model or any param on the COPY in the same call — make one, fork it five times, change one thing on each. The original is never touched.',
|
|
4435
|
+
input: z.object({
|
|
4436
|
+
shotId: z.string().describe('The Shot id, or its SHOT-A code.'),
|
|
4437
|
+
name: z.string().max(120).optional().describe('Name for the copy (default: the original plus "copy").'),
|
|
4438
|
+
prompt: z.string().max(4000).optional().describe('Replace the prompt on the copy. Omit to keep the original\'s.'),
|
|
4439
|
+
model: z.string().optional().describe('Point the copy at a different model — the A/B lever. The prompt is NOT rewritten, and the copy records which model it was written for.'),
|
|
4440
|
+
params: shotParamsSchemaTerse,
|
|
4441
|
+
frameId: z.string().uuid().nullable().optional().describe('Attach the copy to this frame. Omit to keep the original\'s frame; pass null to leave it unattached.'),
|
|
4442
|
+
}),
|
|
4443
|
+
async run(input, ctx) {
|
|
4444
|
+
const capErr = assertShotCapabilities(input.model, input.params);
|
|
4445
|
+
if (capErr)
|
|
4446
|
+
return capErr;
|
|
4447
|
+
const desktop = ctx.desktop();
|
|
4448
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4449
|
+
// Only the halves actually named. The route merges two levels deep, so an
|
|
4450
|
+
// omitted param on the copy keeps the original's value rather than clearing it.
|
|
4451
|
+
const spec = {};
|
|
4452
|
+
if (input.prompt !== undefined)
|
|
4453
|
+
spec.prompt = input.prompt;
|
|
4454
|
+
if (input.model !== undefined)
|
|
4455
|
+
spec.model = input.model;
|
|
4456
|
+
if (input.params !== undefined)
|
|
4457
|
+
spec.params = shotParamsPatch(input.params);
|
|
4458
|
+
const r = await desktop.post('/agent/shots/duplicate', {
|
|
4459
|
+
id: input.shotId,
|
|
4460
|
+
name: input.name,
|
|
4461
|
+
...(Object.keys(spec).length > 0 ? { spec } : {}),
|
|
4462
|
+
frameId: input.frameId,
|
|
4463
|
+
});
|
|
4464
|
+
return ok(r.shot, `Forked into "${r.shot?.name || 'Untitled'}".`);
|
|
4465
|
+
},
|
|
4466
|
+
};
|
|
4467
|
+
/**
|
|
4468
|
+
* The variety strip, in words. GENERATED from the report — never hand-typed,
|
|
4469
|
+
* and never a judgement: it states what is there and stops.
|
|
4470
|
+
*
|
|
4471
|
+
* 🔑 IT RIDES THE OP RESULT, NOT A SKILL. Measured on the 2026-08-30 eval
|
|
4472
|
+
* harness: a rule inlined into an op description moved compliance from 0/8 to
|
|
4473
|
+
* 30/32, while the same guidance behind `slates_get_prompting_guide` sat at 13%
|
|
4474
|
+
* before and 13% after. Guidance the agent must CHOOSE to fetch does not reach
|
|
4475
|
+
* it — so the counts arrive in the result it is already reading.
|
|
4476
|
+
*
|
|
4477
|
+
* Slates counts; the agent judges. No suggested shot size, no auto-varied
|
|
4478
|
+
* camera, no "we changed this for you".
|
|
4479
|
+
*/
|
|
4480
|
+
function describeVarietyReport(v) {
|
|
4481
|
+
if (!v || v.cuts === 0)
|
|
4482
|
+
return '';
|
|
4483
|
+
const parts = [];
|
|
4484
|
+
const topSize = v.shotSizes.find((b) => b.bucket !== 'other');
|
|
4485
|
+
if (topSize && topSize.count >= 2)
|
|
4486
|
+
parts.push(`${topSize.count}/${v.cuts} ${topSize.bucket}`);
|
|
4487
|
+
const topMove = v.cameraMoves.find((b) => b.bucket !== 'other');
|
|
4488
|
+
if (topMove && topMove.count >= 2)
|
|
4489
|
+
parts.push(`${topMove.count} ${topMove.bucket}`);
|
|
4490
|
+
for (const run of v.runs)
|
|
4491
|
+
parts.push(`${run.length} ${run.bucket} in a row`);
|
|
4492
|
+
// 🚨 TWO DENOMINATORS, NAMED. Rhythm is counted in CUTS and money in
|
|
4493
|
+
// GENERATIONS; several cuts routinely live inside one generation.
|
|
4494
|
+
const head = `${v.generations} generation(s) · ${v.cuts} cut(s) · ` +
|
|
4495
|
+
`${v.runtimeSeconds == null ? 'no runtime yet' : `${v.runtimeSeconds}s`}` +
|
|
4496
|
+
(v.cutsWithoutDuration > 0 ? ` (${v.cutsWithoutDuration} cut(s) have no duration)` : '');
|
|
4497
|
+
const variety = parts.length > 0 ? `\nVARIETY: ${parts.join(' · ')}` : '';
|
|
4498
|
+
const words = v.words > 0
|
|
4499
|
+
? `\nWORDS: ${v.words} written` +
|
|
4500
|
+
(v.wordBudget != null
|
|
4501
|
+
? `, ~${v.wordBudget} fit the runtime at ${SPEECH_RATE.conversational.wpm} wpm` +
|
|
4502
|
+
` (measured across ${SPEECH_RATE.conversational.n} real ads)`
|
|
4503
|
+
: '')
|
|
4504
|
+
: '';
|
|
4505
|
+
const overLong = v.overLongLines.length > 0
|
|
4506
|
+
? `\n⚠ ${v.overLongLines.length} line(s) cannot fit their cut at ANY plausible delivery` +
|
|
4507
|
+
` (over ${SPEECH_RATE.ceiling.wpm} wpm, the fastest read in the corpus).` +
|
|
4508
|
+
` Split the line or merge the cut — slates_split_shot / slates_merge_shots.`
|
|
4509
|
+
: '';
|
|
4510
|
+
return `${head}${variety}${words}${overLong}`;
|
|
4511
|
+
}
|
|
4512
|
+
export const listShots = {
|
|
4513
|
+
id: 'slates_list_shots',
|
|
4514
|
+
description: "Read the shot list — every Shot as a compact row IN BOARD ORDER (scene, then position), with the piece's cut count, runtime, credit floor and its variety distribution. Read this before firing a set: if one shot size is the plurality or three cuts in a row share a camera move, the batch is wrong before a credit is spent.",
|
|
4515
|
+
input: z.object({
|
|
4516
|
+
projectId: z.string().uuid(),
|
|
4517
|
+
storyboardId: z.string().uuid().optional().describe('Only Shots attached to a frame in this storyboard.'),
|
|
4518
|
+
frameId: z.string().uuid().optional().describe('Only Shots attached to this frame.'),
|
|
4519
|
+
}),
|
|
4520
|
+
async run(input, ctx) {
|
|
4521
|
+
const desktop = ctx.desktop();
|
|
4522
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4523
|
+
const r = await desktop.get('/agent/shots', {
|
|
4524
|
+
projectId: input.projectId,
|
|
4525
|
+
storyboardId: input.storyboardId,
|
|
4526
|
+
frameId: input.frameId,
|
|
4527
|
+
});
|
|
4528
|
+
const rows = r.shots ?? [];
|
|
4529
|
+
// Deliberately does NOT compose each Shot — that is what slates_get_shot is
|
|
4530
|
+
// for. A listing that composed every row would make browsing cost as much as
|
|
4531
|
+
// auditing.
|
|
4532
|
+
const registry = await ctx.cloud().get('/api/agent/models');
|
|
4533
|
+
const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
|
|
4534
|
+
let total = 0;
|
|
4535
|
+
let unpriced = 0;
|
|
4536
|
+
const shots = rows.map((s) => {
|
|
4537
|
+
const q = shotQuote(s, byKey);
|
|
4538
|
+
if (q.key == null || !byKey.has(q.key))
|
|
4539
|
+
unpriced += 1;
|
|
4540
|
+
total += q.credits;
|
|
4541
|
+
return {
|
|
4542
|
+
id: s.id,
|
|
4543
|
+
code: s.code,
|
|
4544
|
+
name: s.name,
|
|
4545
|
+
scene: s.sceneName,
|
|
4546
|
+
position: s.position,
|
|
4547
|
+
model: s.model,
|
|
4548
|
+
references: s.referenceCount,
|
|
4549
|
+
cuts: s.cuts,
|
|
4550
|
+
runtime_seconds: s.runtimeSeconds,
|
|
4551
|
+
speaker: s.speaker,
|
|
4552
|
+
line: s.line,
|
|
4553
|
+
delivery: s.delivery,
|
|
4554
|
+
action: s.action,
|
|
4555
|
+
prop: s.prop,
|
|
4556
|
+
shot_size: s.shotSize,
|
|
4557
|
+
camera: s.camera,
|
|
4558
|
+
continues: s.continues,
|
|
4559
|
+
credits: q.credits,
|
|
4560
|
+
};
|
|
4561
|
+
});
|
|
4562
|
+
return ok({ shots, total_credits: total, unpriced, variety: r.variety }, `${shots.length} shot(s), at least ${fmtCredits(total)} to fire them all` +
|
|
4563
|
+
(unpriced > 0 ? ` (${unpriced} could not be priced — no model or no duration set).` : '.') +
|
|
4564
|
+
' 🚨 That is a FLOOR, not the bill: a listing does not compose, so the two dimensions that' +
|
|
4565
|
+
' depend on the reference set — Seedance reference-clip seconds and MiniMax reference images' +
|
|
4566
|
+
' past the free five — are missing from it. slates_get_shot prices one exactly, and' +
|
|
4567
|
+
' slates_generate_from_shots quotes the set exactly before it fires anything.' +
|
|
4568
|
+
(describeVarietyReport(r.variety) ? `
|
|
4569
|
+
|
|
4570
|
+
${describeVarietyReport(r.variety)}` : ''));
|
|
4571
|
+
},
|
|
4572
|
+
};
|
|
4573
|
+
export const getShot = {
|
|
4574
|
+
id: 'slates_get_shot',
|
|
4575
|
+
description: 'Read one Shot in full — the COMPOSED prompt the request will actually carry, its numbered references, anything it points at that no longer exists, and its exact credit quote. Audit your own work here before firing.',
|
|
4576
|
+
input: z.object({
|
|
4577
|
+
shotId: z.string().describe('The Shot id, or its SHOT-A code as shown on the row.'),
|
|
4578
|
+
}),
|
|
4579
|
+
async run(input, ctx) {
|
|
4580
|
+
const desktop = ctx.desktop();
|
|
4581
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4582
|
+
const r = await desktop.get('/agent/shots/get', { id: input.shotId });
|
|
4583
|
+
const registry = await ctx.cloud().get('/api/agent/models');
|
|
4584
|
+
const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
|
|
4585
|
+
const q = shotQuote(r.shot, byKey);
|
|
4586
|
+
return ok({ ...r.shot, cost_key: q.key, credits: q.credits }, `"${r.shot.name || 'Untitled'}" — ${r.shot.model ?? 'no model set'}, ` +
|
|
4587
|
+
(q.key && !r.shot.blocked
|
|
4588
|
+
? `${fmtCredits(q.credits)} (${q.key}).`
|
|
4589
|
+
: `CANNOT FIRE YET: ${r.shot.blocked ?? 'not priceable — set a model and a duration.'}`) +
|
|
4590
|
+
(r.shot.blocked ? '' : ` Fires with ${JSON.stringify(r.shot.firesWith)}.`) +
|
|
4591
|
+
`\nCOMPOSED PROMPT (what the model is told): ${r.shot.composedPrompt}`);
|
|
4592
|
+
},
|
|
4593
|
+
};
|
|
4594
|
+
/**
|
|
4595
|
+
* SPLIT and MERGE — the chop decision, and the only two cross-row operations on
|
|
4596
|
+
* this surface.
|
|
4597
|
+
*
|
|
4598
|
+
* 🔑 THIS IS THE EXECUTIVE CALL, AND IT IS WHY THEY ARE OPS. *"Sometimes you
|
|
4599
|
+
* might be using more dialogue in one single 30-second generation. Sometimes it
|
|
4600
|
+
* might just be a 4-second generation of one line."* Merging five lines into
|
|
4601
|
+
* one long take or splitting them into five short ones changes the rhythm AND
|
|
4602
|
+
* the price, and the agent has to be able to re-chop what it wrote.
|
|
4603
|
+
*
|
|
4604
|
+
* 🚨 NEITHER EDITS TEXT AS A SIDE EFFECT. Split moves the text after a caret
|
|
4605
|
+
* the caller placed; merge joins two texts at their boundary. Nothing else in
|
|
4606
|
+
* either row is rewritten. That property is what keeps a finished script
|
|
4607
|
+
* finished, and it is the acceptance test for anything added here later.
|
|
4608
|
+
*/
|
|
4609
|
+
export const splitShot = {
|
|
4610
|
+
id: 'slates_split_shot',
|
|
4611
|
+
description: 'Split one Shot into two at a caret in its line. The text after the caret moves to the new Shot, which INHERITS the model, params and references and lands directly after it in the scene. A mid-sentence split marks the second row as continuing the first — one sentence, two cuts, which is the signature voiceover move. Takes stay with the first row. Omit the caret to add a sibling cut with the same visuals and no words moved.',
|
|
4612
|
+
input: z.object({
|
|
4613
|
+
shotId: z.string().describe('The Shot id, or its SHOT-A code.'),
|
|
4614
|
+
caret: z
|
|
4615
|
+
.number()
|
|
4616
|
+
.int()
|
|
4617
|
+
.min(0)
|
|
4618
|
+
.optional()
|
|
4619
|
+
.describe("Character offset into the Shot's `line` to cut at. Omitted = the end of the line."),
|
|
4620
|
+
}),
|
|
4621
|
+
async run(input, ctx) {
|
|
4622
|
+
const desktop = ctx.desktop();
|
|
4623
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4624
|
+
const r = await desktop.post('/agent/shots/split', { id: input.shotId, caret: input.caret });
|
|
4625
|
+
return ok(r, `Split into ${r.first?.code || 'the first Shot'} and ` +
|
|
4626
|
+
`${r.second?.code || 'a new Shot'}. Duration was INHERITED, not divided — ` +
|
|
4627
|
+
`set it on each if the chop changed how long they run.`);
|
|
4628
|
+
},
|
|
4629
|
+
};
|
|
4630
|
+
export const mergeShots = {
|
|
4631
|
+
id: 'slates_merge_shots',
|
|
4632
|
+
description: "Merge two adjacent Shots into one. Texts join, references union, the FIRST Shot's model and params win, and the durations SUM — which may exceed the model's window, in which case it is shown and never blocked. Lossy in one direction: the second Shot's model and params are discarded, and there is no undo (the takes are the history).",
|
|
4633
|
+
input: z.object({
|
|
4634
|
+
firstId: z.string().describe('The Shot that survives — its model and params win. Id or SHOT-A code.'),
|
|
4635
|
+
secondId: z.string().describe('The Shot folded into it. Id or SHOT-A code.'),
|
|
4636
|
+
}),
|
|
4637
|
+
async run(input, ctx) {
|
|
4638
|
+
const desktop = ctx.desktop();
|
|
4639
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4640
|
+
const r = await desktop.post('/agent/shots/merge', {
|
|
4641
|
+
firstId: input.firstId,
|
|
4642
|
+
secondId: input.secondId,
|
|
4643
|
+
});
|
|
4644
|
+
return ok(r.shot, `Merged into ${r.shot?.code || 'one Shot'} — ` +
|
|
4645
|
+
`${r.shot?.cuts ?? 1} cut(s), ${r.shot?.runtimeSeconds ?? 'no'} second(s).`);
|
|
4646
|
+
},
|
|
4647
|
+
};
|
|
4648
|
+
export const generateFromShots = {
|
|
4649
|
+
id: 'slates_generate_from_shots',
|
|
4650
|
+
billable: true,
|
|
4651
|
+
description: 'Generate from saved Shots, ONE AFTER ANOTHER, with a single quote and a single approval for the whole set. It blocks until the last one lands, so a set of video Shots can outlast the HTTP timeout while the run keeps going — if that happens, poll slates_get_shot for each Shot\'s generationIds instead of re-firing, which double-spends.',
|
|
4652
|
+
input: z.object({
|
|
4653
|
+
shotIds: z.array(z.string()).min(1).max(20).describe('The Shots to fire, in order — ids or SHOT-A codes.'),
|
|
4654
|
+
confirm: z.boolean().optional().describe('Set true after explicit user OK on the TOTAL below.'),
|
|
4655
|
+
}),
|
|
4656
|
+
async run(input, ctx) {
|
|
4657
|
+
const desktop = ctx.desktop();
|
|
4658
|
+
await desktop.requireCapability('shots', 'saved Shots');
|
|
4659
|
+
// Resolve and price every Shot BEFORE anything fires. A dead id found
|
|
4660
|
+
// halfway through a batch means a partially-fired, partially-BILLED run.
|
|
4661
|
+
const details = [];
|
|
4662
|
+
for (const id of input.shotIds) {
|
|
4663
|
+
const r = await desktop.get('/agent/shots/get', { id });
|
|
4664
|
+
details.push(r.shot);
|
|
4665
|
+
}
|
|
4666
|
+
const registry = await ctx.cloud().get('/api/agent/models');
|
|
4667
|
+
const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
|
|
4668
|
+
const quotes = details.map((d) => ({ detail: d, ...shotQuote(d, byKey) }));
|
|
4669
|
+
const total = quotes.reduce((n, q) => n + q.credits, 0);
|
|
4670
|
+
const largest = quotes.reduce((m, q) => (q.credits > m ? q.credits : m), 0);
|
|
4671
|
+
const unpriced = quotes.filter((q) => q.key == null || !byKey.has(q.key));
|
|
4672
|
+
const blockedShots = details.filter((d) => d.blocked);
|
|
4673
|
+
if (!input.confirm) {
|
|
4674
|
+
// ONE approval for the set, itemised. N approvals would re-introduce the
|
|
4675
|
+
// friction the batch exists to remove; the safety is the STATED TOTAL,
|
|
4676
|
+
// prominent — count, total, and the largest single Shot.
|
|
4677
|
+
const lines = quotes.map((q) => ` - ${q.detail.name || 'Untitled'} · ${q.detail.model ?? 'no model'} · ` +
|
|
4678
|
+
(q.key ? fmtCredits(q.credits) : 'NOT PRICEABLE'));
|
|
4679
|
+
const blockedLines = blockedShots.map((d) => ` ✖ ${d.name || 'Untitled'} WILL NOT FIRE: ${d.blocked}`);
|
|
4680
|
+
const warnings = details
|
|
4681
|
+
.filter((d) => d.missing || d.unresolvedTokens?.length)
|
|
4682
|
+
.map((d) => ` ! ${d.name || 'Untitled'}: ${d.missing ? 'references something that no longer exists' : ''}` +
|
|
4683
|
+
`${d.unresolvedTokens?.length ? ` ${d.unresolvedTokens.join(', ')} match nothing saved (sent as written, no reference attached)` : ''}`);
|
|
4684
|
+
return ok({
|
|
4685
|
+
requires_confirm: true,
|
|
4686
|
+
count: quotes.length,
|
|
4687
|
+
total_credits: total,
|
|
4688
|
+
largest_single_credits: largest,
|
|
4689
|
+
blocked_count: blockedShots.length,
|
|
4690
|
+
shots: quotes.map((q) => ({
|
|
4691
|
+
id: q.detail.id,
|
|
4692
|
+
name: q.detail.name,
|
|
4693
|
+
model: q.detail.model,
|
|
4694
|
+
cost_key: q.key,
|
|
4695
|
+
credits: q.credits,
|
|
4696
|
+
blocked: q.detail.blocked,
|
|
4697
|
+
})),
|
|
4698
|
+
}, `Firing ${quotes.length} Shot(s) SEQUENTIALLY.\n` +
|
|
4699
|
+
`TOTAL ${fmtCredits(total)} · largest single ${fmtCredits(largest)}\n` +
|
|
4700
|
+
lines.join('\n') +
|
|
4701
|
+
(blockedLines.length > 0 ? `\n${blockedLines.join('\n')}` : '') +
|
|
4702
|
+
(unpriced.length > 0
|
|
4703
|
+
? `\n ! ${unpriced.length} shot(s) could not be priced — they will still be attempted and may fail.`
|
|
4704
|
+
: '') +
|
|
4705
|
+
(warnings.length > 0 ? `\n${warnings.join('\n')}` : '') +
|
|
4706
|
+
`\n\nRe-call with confirm: true after explicit user OK on that total.`);
|
|
4707
|
+
}
|
|
4708
|
+
const r = await desktop.post('/agent/shots/batch-generate', { shotIds: input.shotIds });
|
|
4709
|
+
const failedLines = (r.results ?? [])
|
|
4710
|
+
.filter((x) => x.status === 'failed')
|
|
4711
|
+
.map((x) => ` ✗ ${x.name || 'Untitled'}: ${x.error ?? 'failed'}`);
|
|
4712
|
+
return ok(r, `${r.succeeded} of ${r.total} generated for about ${fmtCredits(total)}.` +
|
|
4713
|
+
(failedLines.length > 0
|
|
4714
|
+
? // Reported, never retried: an agent that treats a failed render as
|
|
4715
|
+
// something to try again spends credits before anyone notices.
|
|
4716
|
+
`\n${failedLines.join('\n')}\nThese were NOT retried. Read each error, fix the Shot, and re-fire only what you meant to.`
|
|
4717
|
+
: '') +
|
|
4718
|
+
` ${VIDEO_REVIEW_POINTER}`);
|
|
4719
|
+
},
|
|
4720
|
+
};
|
|
3518
4721
|
function resolveGuideTopic(topic) {
|
|
3519
4722
|
const t = topic.trim().toLowerCase();
|
|
3520
4723
|
if (SKILLS[t])
|
|
@@ -3578,23 +4781,41 @@ function resolveGuideTopic(topic) {
|
|
|
3578
4781
|
t.startsWith('h3-')) {
|
|
3579
4782
|
return 'slates-prompting-minimax-h3';
|
|
3580
4783
|
}
|
|
4784
|
+
// LTX-2.5 — both seats share one skill. A PREFIX is right HERE (unlike every
|
|
4785
|
+
// rate, key and endpoint lookup, which must be exact) precisely because both
|
|
4786
|
+
// rows resolve to the same guide: `ltx-2-5-pro` matching the `ltx` prefix is
|
|
4787
|
+
// the intended outcome, not a collision.
|
|
4788
|
+
if (t.startsWith('ltx') || t === 'lightricks')
|
|
4789
|
+
return 'slates-prompting-ltx-2-5';
|
|
3581
4790
|
if (t.startsWith('kling-mc'))
|
|
3582
4791
|
return 'slates-prompting-motion-transfer';
|
|
3583
4792
|
if (t === 'edit-video' || t === 'video-edit' || t === 'edit video' || t === 'video edit')
|
|
3584
4793
|
return 'slates-prompting-kling-v3';
|
|
3585
4794
|
if (t.startsWith('kling-v3'))
|
|
3586
4795
|
return 'slates-prompting-kling-v3';
|
|
3587
|
-
// Audio —
|
|
3588
|
-
//
|
|
3589
|
-
//
|
|
3590
|
-
//
|
|
3591
|
-
//
|
|
3592
|
-
// guide, which is
|
|
4796
|
+
// Audio — the TTS seat FIRST, then seed-audio, then eleven-sfx.
|
|
4797
|
+
//
|
|
4798
|
+
// 🚨 "tts"/"voiceover" ROUTE HERE AGAIN (2026-09-05). They used to land on
|
|
4799
|
+
// seed-audio because there was no TTS surface at all; there is one now, so
|
|
4800
|
+
// leaving them there would hand a scene-renderer's guide to someone asking
|
|
4801
|
+
// about speech. They must still never reach the ElevenLabs guide, which is
|
|
4802
|
+
// SFX-only.
|
|
4803
|
+
if (t.startsWith('inworld') ||
|
|
4804
|
+
t === 'tts' ||
|
|
4805
|
+
t === 'text-to-speech' ||
|
|
4806
|
+
t === 'text to speech' ||
|
|
4807
|
+
t === 'voiceover' ||
|
|
4808
|
+
t === 'voice') {
|
|
4809
|
+
return 'slates-prompting-inworld-tts';
|
|
4810
|
+
}
|
|
4811
|
+
// seed-audio BEFORE the seedance check: "seed-audio" also starts with "seed",
|
|
4812
|
+
// and falling through would hand the video guide to the audio model (the
|
|
4813
|
+
// exact class of aliasing bug this comment block warns about). `dialogue`
|
|
4814
|
+
// stays here: dialogue inside a SCENE is what seed-audio is for, while a
|
|
4815
|
+
// single voice saying a single line is the TTS seat above.
|
|
3593
4816
|
if (t.startsWith('seed-audio') ||
|
|
3594
4817
|
t === 'seed audio' ||
|
|
3595
4818
|
t === 'audio' ||
|
|
3596
|
-
t === 'tts' ||
|
|
3597
|
-
t === 'voiceover' ||
|
|
3598
4819
|
t === 'dialogue') {
|
|
3599
4820
|
return 'slates-prompting-seed-audio';
|
|
3600
4821
|
}
|
|
@@ -3630,14 +4851,43 @@ function resolveGuideTopic(topic) {
|
|
|
3630
4851
|
}
|
|
3631
4852
|
return null;
|
|
3632
4853
|
}
|
|
4854
|
+
/**
|
|
4855
|
+
* The guide index, GENERATED from SKILLS.
|
|
4856
|
+
*
|
|
4857
|
+
* The list here was hand-typed and had drifted to 25 of 32 names — the prompting
|
|
4858
|
+
* guides for GPT Image 2, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
|
|
4859
|
+
* all missing, so an agent reading this description could not learn they exist.
|
|
4860
|
+
* A hand-typed index of a generated corpus is a stale index; it is only a matter
|
|
4861
|
+
* of when.
|
|
4862
|
+
*
|
|
4863
|
+
* Per-model guides are listed as BARE NAMES: the name is the description, and
|
|
4864
|
+
* `resolveGuideTopic()` resolves a model id to the right one anyway.
|
|
4865
|
+
*/
|
|
4866
|
+
function describeGuideTopics() {
|
|
4867
|
+
const names = Object.keys(SKILLS).sort();
|
|
4868
|
+
const perModel = names.filter((n) => n.startsWith('slates-prompting-'));
|
|
4869
|
+
const rest = names.filter((n) => !n.startsWith('slates-prompting-'));
|
|
4870
|
+
return (`Workflow and cross-cutting guides: ${rest.join(', ')}. ` +
|
|
4871
|
+
`Per-model prompting guides (or just pass the model id, which resolves to one of these): ` +
|
|
4872
|
+
`${perModel.join(', ')}. ` +
|
|
4873
|
+
`Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.`);
|
|
4874
|
+
}
|
|
3633
4875
|
export const getPromptingGuide = {
|
|
3634
4876
|
id: 'slates_get_prompting_guide',
|
|
3635
|
-
description:
|
|
4877
|
+
description:
|
|
4878
|
+
// 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
|
|
4879
|
+
// MEASURED at 13% compliance before and after the enforcement work — pointer
|
|
4880
|
+
// prose is the shape that does not move the agent. What replaced it is
|
|
4881
|
+
// structural: the never-use list rides the generate ops' descriptions and
|
|
4882
|
+
// the craft card rides the estimate result, so the facts arrive whether or
|
|
4883
|
+
// not this op is ever called.
|
|
4884
|
+
"Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
|
|
3636
4885
|
input: z.object({
|
|
3637
4886
|
topic: z
|
|
3638
4887
|
.string()
|
|
3639
4888
|
.min(1)
|
|
3640
|
-
.describe(
|
|
4889
|
+
.describe(`Guide name, model id, or style name. ${describeGuideTopics()}`),
|
|
4890
|
+
depth: z.enum(['card', 'full']).optional().describe('"card" returns just the levers block (a few hundred words — the same card slates_estimate_generation_cost already attached, so usually redundant). "full" (default) returns the whole guide, up to several thousand words.'),
|
|
3641
4891
|
}),
|
|
3642
4892
|
async run(input) {
|
|
3643
4893
|
const resolved = resolveGuideTopic(input.topic);
|
|
@@ -3645,12 +4895,46 @@ export const getPromptingGuide = {
|
|
|
3645
4895
|
if (!resolved || content === undefined) {
|
|
3646
4896
|
throw new Error(`Unknown guide topic: ${input.topic}. Valid topics: ${Object.keys(SKILLS).sort().join(', ')}`);
|
|
3647
4897
|
}
|
|
4898
|
+
if (input.depth === 'card') {
|
|
4899
|
+
const card = describeCraftCard(resolved);
|
|
4900
|
+
if (card) {
|
|
4901
|
+
return { text: card, data: { topic: resolved, depth: 'card', bytes: Buffer.byteLength(card, 'utf8') } };
|
|
4902
|
+
}
|
|
4903
|
+
// No card on this guide — returning nothing would read as "no guidance",
|
|
4904
|
+
// which is worse than a fall-through the result names.
|
|
4905
|
+
}
|
|
3648
4906
|
return {
|
|
3649
4907
|
text: content,
|
|
3650
|
-
data: { topic: resolved, bytes: Buffer.byteLength(content, 'utf8') },
|
|
4908
|
+
data: { topic: resolved, depth: 'full', bytes: Buffer.byteLength(content, 'utf8') },
|
|
3651
4909
|
};
|
|
3652
4910
|
},
|
|
3653
4911
|
};
|
|
4912
|
+
/**
|
|
4913
|
+
* The one op that changes what OTHER ops are visible.
|
|
4914
|
+
*
|
|
4915
|
+
* 🚨 IT EXISTS BECAUSE THE SURFACE IS 112 KB AND EVERY TURN PAYS FOR ALL OF IT.
|
|
4916
|
+
* The desktop Studio Agent sends `core` plus this; a group arrives when the
|
|
4917
|
+
* work needs it and stays for the rest of the run. On the MCP surface every op
|
|
4918
|
+
* is registered up front (a stdio server has no run to append to), so this
|
|
4919
|
+
* returns the same definitions as a plain listing — useful either way, since
|
|
4920
|
+
* it is also how an agent asks "what else can you do".
|
|
4921
|
+
*/
|
|
4922
|
+
export const loadTools = {
|
|
4923
|
+
id: 'slates_load_tools',
|
|
4924
|
+
description: 'Load a deferred group of tools for the rest of this session. The core surface is always present; these four groups are held back so every turn does not pay for the whole registry. ' +
|
|
4925
|
+
Object.entries(GROUP_SUMMARY)
|
|
4926
|
+
.map(([g, s]) => `"${g}": ${s}`)
|
|
4927
|
+
.join('. ') +
|
|
4928
|
+
'. Call it the moment the work needs one of those — the tools arrive in the same turn\'s result and stay loaded. On MCP clients every tool is already registered and this just lists the group.',
|
|
4929
|
+
input: z.object({
|
|
4930
|
+
group: z.enum(['library', 'timeline', 'admin', 'blender']).describe('Which group to load.'),
|
|
4931
|
+
}),
|
|
4932
|
+
async run(input) {
|
|
4933
|
+
const defs = toolDefinitions(ALL_OPERATIONS.filter((op) => groupFor(op.id) === input.group), { surface: 'mcp' });
|
|
4934
|
+
return ok({ group: input.group, tools: defs }, `Loaded the "${input.group}" group — ${defs.length} tool(s) now available:\n` +
|
|
4935
|
+
defs.map((d) => `${d.name}: ${d.description.split(/(?<=\.)\s/)[0]}`).join('\n'));
|
|
4936
|
+
},
|
|
4937
|
+
};
|
|
3654
4938
|
// ── Blender previs ──────────────────────────────────────────────
|
|
3655
4939
|
//
|
|
3656
4940
|
// The only ops that talk to a third transport: a localhost socket into a
|
|
@@ -3881,14 +5165,31 @@ export const ALL_OPERATIONS = [
|
|
|
3881
5165
|
updateFrame,
|
|
3882
5166
|
batchUpdateFrames,
|
|
3883
5167
|
deleteFrame,
|
|
5168
|
+
// ── Shots: the prompt bar, serialized ────────────────────────────────
|
|
5169
|
+
// Beside the storyboard ops because that is the neighbourhood they belong to
|
|
5170
|
+
// — structure, not spend. The one that spends sits last in the group.
|
|
5171
|
+
createShot,
|
|
5172
|
+
updateShot,
|
|
5173
|
+
duplicateShot,
|
|
5174
|
+
// Split and merge are the CHOP decision — where the rhythm and the price are
|
|
5175
|
+
// actually decided. They sit beside the writers, not with the spender.
|
|
5176
|
+
splitShot,
|
|
5177
|
+
mergeShots,
|
|
5178
|
+
listShots,
|
|
5179
|
+
getShot,
|
|
5180
|
+
generateFromShots,
|
|
3884
5181
|
getPromptingGuide,
|
|
5182
|
+
loadTools,
|
|
3885
5183
|
// ── Blender previs, LAST and deliberately ────────────────────────────
|
|
3886
|
-
//
|
|
3887
|
-
//
|
|
3888
|
-
// the
|
|
3889
|
-
//
|
|
3890
|
-
//
|
|
3891
|
-
//
|
|
5184
|
+
// These six landed at the TOP once, which put a third transport nobody
|
|
5185
|
+
// without Blender can reach ahead of `slates_get_workspace_state` in every
|
|
5186
|
+
// conversation the app has. They are a niche lane off the end of the surface,
|
|
5187
|
+
// and the list should read that way.
|
|
5188
|
+
//
|
|
5189
|
+
// ⚠️ POSITION NO LONGER CONTROLS COST. They are the `blender` tier group (see
|
|
5190
|
+
// surface.ts), so the desktop does not send them at all until
|
|
5191
|
+
// `slates_load_tools` asks for them. Order here is now the READING order on
|
|
5192
|
+
// the MCP surface, which sends everything — still last, same reason.
|
|
3892
5193
|
blenderStatus,
|
|
3893
5194
|
blenderExecute,
|
|
3894
5195
|
blenderScene,
|
|
@@ -3896,4 +5197,32 @@ export const ALL_OPERATIONS = [
|
|
|
3896
5197
|
blenderSearchDocs,
|
|
3897
5198
|
blenderRenderBlocking,
|
|
3898
5199
|
];
|
|
5200
|
+
// ── Surface metadata, stamped once at load ──────────────────────
|
|
5201
|
+
//
|
|
5202
|
+
// 🚨 DERIVED, NEVER HAND-SET PER OP. Every op gets all four MCP annotations and
|
|
5203
|
+
// its tier here, from the rules in `surface.ts`, so a new op cannot ship
|
|
5204
|
+
// un-annotated (which a host reads as "not destructive") or accidentally
|
|
5205
|
+
// deferred. The lockstep check re-derives the hints INDEPENDENTLY from each
|
|
5206
|
+
// op's own transport verbs, so a read-only claim on an op that posts is caught
|
|
5207
|
+
// rather than trusted.
|
|
5208
|
+
for (const op of ALL_OPERATIONS) {
|
|
5209
|
+
const mutable = op;
|
|
5210
|
+
mutable.annotations = annotate(mutable.id, mutable.billable);
|
|
5211
|
+
mutable.tier = tierFor(mutable.id);
|
|
5212
|
+
mutable.group = groupFor(mutable.id);
|
|
5213
|
+
}
|
|
5214
|
+
// A group naming an op that does not exist would silently defer nothing, and
|
|
5215
|
+
// the tool it meant to hold back would keep costing prefix bytes forever.
|
|
5216
|
+
{
|
|
5217
|
+
const ids = new Set(ALL_OPERATIONS.map((o) => o.id));
|
|
5218
|
+
for (const [group, members] of Object.entries(OPERATION_GROUPS)) {
|
|
5219
|
+
for (const id of members) {
|
|
5220
|
+
if (!ids.has(id)) {
|
|
5221
|
+
throw new Error(`[operations] OPERATION_GROUPS.${group} names "${id}", which is not in ALL_OPERATIONS. ` +
|
|
5222
|
+
`Fix the id or drop the entry — a phantom member defers nothing.`);
|
|
5223
|
+
}
|
|
5224
|
+
}
|
|
5225
|
+
}
|
|
5226
|
+
}
|
|
5227
|
+
export { toolDefinitions, toolDefinition, groupFor, tierFor, OPERATION_GROUPS, GROUP_SUMMARY } from './surface.js';
|
|
3899
5228
|
//# sourceMappingURL=index.js.map
|