@slatesvideo/shared 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api-url.d.ts +5 -3
- package/dist/api-url.js +5 -3
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/manual/content.d.ts +2 -0
- package/dist/manual/content.js +3 -0
- package/dist/manual/index.d.ts +5 -0
- package/dist/manual/index.js +20 -0
- package/dist/operations/index.d.ts +67 -17
- package/dist/operations/index.js +310 -87
- package/dist/prompts/agent-doctrine.js +1 -0
- package/dist/prompts/character-sheet.js +10 -0
- package/dist/prompts/model-capabilities.js +52 -3
- package/dist/prompts/model-facts.js +12 -4
- package/dist/prompts/prompting-tips.js +9 -3
- package/dist/prompts/reference-composer.js +14 -0
- package/dist/prompts/shot-spec.d.ts +30 -10
- package/dist/prompts/shot-spec.js +41 -9
- package/dist/skills/content.js +12 -12
- package/exports/slates-prompt-builder/generated/reference-character.md +1 -1
- package/exports/slates-prompt-builder/generated/reference-seedance.md +3 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +10 -10
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +1 -1
- package/skills/slates-character-identity.md +1 -1
- package/skills/slates-model-selection.md +10 -7
- package/skills/slates-prompting-elevenlabs.md +1 -1
- package/skills/slates-prompting-gpt-image-2-5.md +183 -0
- package/skills/slates-prompting-inworld-tts.md +174 -166
- package/skills/slates-prompting-lip-sync.md +1 -1
- package/skills/slates-prompting-nano-banana-2.md +1 -1
- package/skills/slates-prompting-seed-audio.md +1 -1
- package/skills/slates-prompting-seedance-2-5.md +3 -1
- package/skills/slates-prompting-seedance.md +3 -1
- package/skills/slates-ugc-influencer-ad.md +5 -3
- package/skills/slates-vision-feedback-loop.md +2 -2
- package/skills/slates-prompting-gpt-image-2.md +0 -109
package/dist/operations/index.js
CHANGED
|
@@ -14,6 +14,7 @@ import { SlatesCloudClient } from '../clients/cloud.js';
|
|
|
14
14
|
import { SlatesDesktopClient } from '../clients/desktop.js';
|
|
15
15
|
import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
|
|
16
16
|
import { SKILLS } from '../skills/content.js';
|
|
17
|
+
import { appManualSections } from '../manual/index.js';
|
|
17
18
|
// Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
|
|
18
19
|
// "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
|
|
19
20
|
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
|
|
@@ -183,6 +184,15 @@ export const TTS_MAX_CHARACTERS = (() => {
|
|
|
183
184
|
return max;
|
|
184
185
|
})();
|
|
185
186
|
export const TTS_BUCKET_COUNT = TTS_MAX_CHARACTERS / TTS_BUCKET_CHARS; // 8
|
|
187
|
+
/** The seat's cloning spec — reference-clip bounds, the design-prompt bounds
|
|
188
|
+
* and the workspace-wide clone rate — read from the SSOT for the same reason
|
|
189
|
+
* as the cap: every number in it was MEASURED and lives in exactly one row. */
|
|
190
|
+
export const TTS_VOICE_CLONE = (() => {
|
|
191
|
+
const spec = getModelCapability(TTS_MODEL)?.voiceClone;
|
|
192
|
+
if (!spec)
|
|
193
|
+
throw new Error(`MODEL_CAPABILITIES['${TTS_MODEL}'] must declare voiceClone`);
|
|
194
|
+
return spec;
|
|
195
|
+
})();
|
|
186
196
|
function creditCost(m) {
|
|
187
197
|
if (!m)
|
|
188
198
|
return 0;
|
|
@@ -201,7 +211,7 @@ function creditsFromDollars(dollars) {
|
|
|
201
211
|
// Shared describe-text for the background flag on every generate_* op. ONE
|
|
202
212
|
// sentence: it is repeated verbatim on seven ops, so every word costs seven
|
|
203
213
|
// times, and `slates_get_generation_status` explains the polling itself.
|
|
204
|
-
const BACKGROUND_DESCRIBE = 'Return generationId(s)
|
|
214
|
+
const BACKGROUND_DESCRIBE = 'Return generationId(s) now instead of blocking; poll slates_get_generation_status. Recommended for video.';
|
|
205
215
|
// ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
|
|
206
216
|
//
|
|
207
217
|
// QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
|
|
@@ -219,7 +229,7 @@ const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it
|
|
|
219
229
|
const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
|
|
220
230
|
'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
|
|
221
231
|
const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
|
|
222
|
-
'slates_get_asset_video_frames for video.';
|
|
232
|
+
'slates_get_asset_video_frames for video. For audio, audition the saved file; metadata alone does not establish voice similarity or delivery quality.';
|
|
223
233
|
// The image saved, but reading it back off disk failed (best-effort fetch). The
|
|
224
234
|
// agent has an asset and NO pixels, which is the one state where a quality
|
|
225
235
|
// claim would be pure invention — so this branch has to say so rather than
|
|
@@ -492,8 +502,9 @@ export const estimateGenerationCost = {
|
|
|
492
502
|
duration: z.number().int().min(1).max(360).optional().describe(`Seconds; cost scales linearly. Required with a video or PER-SECOND audio base id. Per-model windows: see slates_generate_video's duration. Audio: seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (⚠️ the requested duration IS the bill), eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. ⛔ NOT for ${TTS_MODEL} — pass \`characters\`.`),
|
|
493
503
|
characters: z.number().int().min(1).max(TTS_MAX_CHARACTERS).optional().describe(`${TTS_MODEL} only — the LENGTH OF THE TEXT to speak (${TTS_BUCKET_CHARS}-char buckets).`),
|
|
494
504
|
videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe('Video only. Omitted, each model quotes at its own default. Per-model ladders: see slates_generate_video\'s videoResolution.'),
|
|
495
|
-
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k
|
|
496
|
-
quality: z.enum(['medium', 'high']).optional().describe('
|
|
505
|
+
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k: GPT Image/seedream-5-lite).'),
|
|
506
|
+
quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
|
|
507
|
+
aspectRatio: z.string().optional().describe('Image only. 1:1/4:3/3:4 cost more than 16:9.'),
|
|
497
508
|
sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
|
|
498
509
|
seedanceFace: z.boolean().optional().describe('Seedance AI-face route (pricier key).'),
|
|
499
510
|
seedanceRealFace: z.boolean().optional().describe('Seedance consented real-face route (premium key).'),
|
|
@@ -504,15 +515,21 @@ export const estimateGenerationCost = {
|
|
|
504
515
|
const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
|
|
505
516
|
// 1) exact registry cost key
|
|
506
517
|
let key = byKey.has(input.model) ? input.model : null;
|
|
507
|
-
// 2) image base id + resolution (+ quality for
|
|
518
|
+
// 2) image base id + resolution (+ quality for GPT Image 2.5)
|
|
508
519
|
if (!key) {
|
|
509
520
|
// IMAGE_MODELS, never a second hand-typed copy: this list is declared
|
|
510
521
|
// below (a runtime read, so no temporal-dead-zone hazard) and is the same
|
|
511
522
|
// enum `slates_generate_image` accepts. Two copies is how the estimate op
|
|
512
523
|
// would quietly stop pricing the seventh image model.
|
|
513
524
|
const img = IMAGE_MODELS.find((m) => m === input.model);
|
|
525
|
+
// ⚙ NO INLINE DEFAULT. `imageCostKey`'s own parameter default is the ONE
|
|
526
|
+
// home for the fallback tier, and `pricing-consistency-check.mjs` pins it
|
|
527
|
+
// to the desktop's. This line read `?? 'medium'` while every generate path
|
|
528
|
+
// billed `high`, so the op the doctrine tells agents to call before every
|
|
529
|
+
// generation quoted 1 cr for a 2 cr job. Four separate copies of one
|
|
530
|
+
// default is what made that possible; there are now none.
|
|
514
531
|
if (img)
|
|
515
|
-
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality
|
|
532
|
+
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality, input.aspectRatio);
|
|
516
533
|
}
|
|
517
534
|
// 2a) audio base id → seconds. Both surfaces bill per second, so a
|
|
518
535
|
// duration is always required. Runs BEFORE the video resolver: it is
|
|
@@ -632,7 +649,7 @@ export const estimateGenerationCost = {
|
|
|
632
649
|
if (key == null || perCredits == null) {
|
|
633
650
|
// Every id in the error comes from the SSOT arrays. The image half was
|
|
634
651
|
// hand-typed and named three of six, so an agent that mis-spelled
|
|
635
|
-
// `gpt-image-2` was told the model did not exist.
|
|
652
|
+
// `gpt-image-2-5-flare` was told the model did not exist.
|
|
636
653
|
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | ${IMAGE_MODELS.join(' | ')}) plus duration/resolution params, or use slates_list_available_models with a filter.`);
|
|
637
654
|
}
|
|
638
655
|
const qty = input.quantity ?? 1;
|
|
@@ -1113,11 +1130,23 @@ export const generateCharacterIdentity = {
|
|
|
1113
1130
|
baseAssetId: z.string().uuid().describe('The base portrait asset the identity is generated from.'),
|
|
1114
1131
|
userNotes: z.string().optional().describe('Extra instruction, e.g. "use the woman on the left".'),
|
|
1115
1132
|
model: z
|
|
1116
|
-
.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2'])
|
|
1133
|
+
.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2-5-flare', 'gpt-image-2-5-sunburst'])
|
|
1117
1134
|
.optional()
|
|
1118
1135
|
.describe('Image model for the sheet. Omit for the default (nano-banana-2). Exists so the layout-vs-face tradeoff can be tested with comparison gens — do not switch without a receipt.'),
|
|
1119
1136
|
}),
|
|
1120
1137
|
async run(input, ctx) {
|
|
1138
|
+
// 🚨 THE ROSTER GATE, WHICH THIS OP NEVER HAD. Its `model` enum offers
|
|
1139
|
+
// seats an older desktop does not know, and the handler resolves an unknown
|
|
1140
|
+
// id by falling back to the Google/Banana path — so the op would quote one
|
|
1141
|
+
// model's price and generate another, silently. Same failure `generateImage`
|
|
1142
|
+
// and `editImage` gate against; this op was simply missed when the gate was
|
|
1143
|
+
// introduced, and the 2.5 swap is what makes it reachable in practice.
|
|
1144
|
+
if (isGptImageModel(input.model)) {
|
|
1145
|
+
await ctx.desktop().requireCapability('image-models-v3', `${input.model} character identity`);
|
|
1146
|
+
}
|
|
1147
|
+
else if (input.model === 'nano-banana-pro' || input.model === 'nano-banana-2-lite') {
|
|
1148
|
+
await ctx.desktop().requireCapability('image-models-v2', `${input.model} character identity`);
|
|
1149
|
+
}
|
|
1121
1150
|
return ok(await ctx.desktop().post('/agent/characters/generate-identity', {
|
|
1122
1151
|
characterId: input.characterId,
|
|
1123
1152
|
projectId: input.projectId,
|
|
@@ -1299,21 +1328,39 @@ async function previewAssets(ctx, refs) {
|
|
|
1299
1328
|
return out;
|
|
1300
1329
|
}
|
|
1301
1330
|
/** The exact `model` ids `slates_generate_image` accepts. */
|
|
1331
|
+
/**
|
|
1332
|
+
* The image models whose ladder includes the 3k (1440p) class — a MIRROR of
|
|
1333
|
+
* `imageResolutions` in slate's MODEL_REGISTRY, which this package cannot read.
|
|
1334
|
+
* Exported so `pricing-consistency-check.mjs` can prove the mirror still
|
|
1335
|
+
* matches; without that proof a model that gains 3k in the registry just goes
|
|
1336
|
+
* quietly unreachable through the op.
|
|
1337
|
+
*/
|
|
1338
|
+
export const THREE_K_IMAGE_MODELS = [
|
|
1339
|
+
'gpt-image-2-5-flare',
|
|
1340
|
+
'gpt-image-2-5-sunburst',
|
|
1341
|
+
'seedream-5-lite',
|
|
1342
|
+
];
|
|
1302
1343
|
export const IMAGE_MODELS = [
|
|
1303
1344
|
'nano-banana-2',
|
|
1304
1345
|
'nano-banana-2-lite',
|
|
1305
1346
|
'nano-banana-pro',
|
|
1306
|
-
'gpt-image-2',
|
|
1347
|
+
'gpt-image-2-5-flare',
|
|
1348
|
+
'gpt-image-2-5-sunburst',
|
|
1307
1349
|
'flux-2-max',
|
|
1308
1350
|
'seedream-5-lite',
|
|
1309
1351
|
];
|
|
1352
|
+
/** True for either GPT Image 2.5 seat. Flare and Sunburst differ in latency
|
|
1353
|
+
* and routing advice, never in pricing shape or param surface. */
|
|
1354
|
+
export function isGptImageModel(model) {
|
|
1355
|
+
return model === 'gpt-image-2-5-flare' || model === 'gpt-image-2-5-sunburst';
|
|
1356
|
+
}
|
|
1310
1357
|
/**
|
|
1311
1358
|
* The aspect ratios an image generation can carry — the UNION over what these
|
|
1312
1359
|
* models declare in `MODEL_CAPABILITIES`, generated so the enum cannot hold a
|
|
1313
1360
|
* value no model accepts. It used to be a hand-typed eleven-value list whose
|
|
1314
1361
|
* `9:21` exists in ZERO models; that phantom is gone by construction.
|
|
1315
1362
|
*
|
|
1316
|
-
* ⚠️ STILL A UNION, NOT A PER-MODEL CHECK.
|
|
1363
|
+
* ⚠️ STILL A UNION, NOT A PER-MODEL CHECK. GPT Image 2.5 takes five of these
|
|
1317
1364
|
* ten and the other five would be accepted here. The image param surface has
|
|
1318
1365
|
* not been audited (aspect ratios, resolution classes, per-model reference
|
|
1319
1366
|
* caps) — that audit is the named follow-up in
|
|
@@ -1325,9 +1372,35 @@ const IMAGE_ASPECT_RATIOS = aspectRatioUnion(IMAGE_MODELS);
|
|
|
1325
1372
|
// Registry cost-key for an image model+resolution. Mirrors imageCreditKey()
|
|
1326
1373
|
// in slate/src/shared/pricing.ts — MUST byte-match it (the same hard rule as
|
|
1327
1374
|
// videoCostKey): NB2/NB Pro price per resolution, FLUX.2 Max prices per
|
|
1328
|
-
// resolution (1k is the bare key), NB2 Lite/Seedream are flat, GPT Image 2 is
|
|
1329
|
-
// quality × resolution-class
|
|
1330
|
-
|
|
1375
|
+
// resolution (1k is the bare key), NB2 Lite/Seedream are flat, GPT Image 2.5 is
|
|
1376
|
+
// quality × resolution-class over ALL FIVE exposed tiers.
|
|
1377
|
+
//
|
|
1378
|
+
// 🚨 THE KEY SUFFIX IS `med` WHILE THE WIRE VALUE IS `medium` — the one
|
|
1379
|
+
// tier name that differs between fal's enum and our cost keys, inherited from
|
|
1380
|
+
// GPT Image 2 (`gpt-image-2-med-4k`). `gptKeyTier` is that whole translation,
|
|
1381
|
+
// and it must stay byte-identical to `gptKeyTier` in
|
|
1382
|
+
// slate/src/shared/pricing.ts — pricing-consistency-check.mjs enforces it.
|
|
1383
|
+
function gptKeyTier(quality) {
|
|
1384
|
+
return quality === 'medium' ? 'med' : quality;
|
|
1385
|
+
}
|
|
1386
|
+
// 🚨 THE ASPECT RATIO IS PART OF THE PRICE ON GPT IMAGE. fal bills image
|
|
1387
|
+
// OUTPUT TOKENS and the count tracks the frame's SHAPE — metered 2026-09-09,
|
|
1388
|
+
// 4:3/3:4 cost ~4/3 of the class rate and 1:1 ~16/9 of it. Quoting one price
|
|
1389
|
+
// for every aspect sold 1:1 below cost at `high` and above. Byte-identical to
|
|
1390
|
+
// `gptKeyAspect` in slate/src/shared/pricing.ts; pricing-consistency-check
|
|
1391
|
+
// proves it key by key. 16:9 and 9:16 keep the bare key they always had.
|
|
1392
|
+
function gptKeyAspect(aspectRatio) {
|
|
1393
|
+
if (aspectRatio === '1:1')
|
|
1394
|
+
return '-sq';
|
|
1395
|
+
if (aspectRatio === '4:3' || aspectRatio === '3:4')
|
|
1396
|
+
return '-43';
|
|
1397
|
+
return '';
|
|
1398
|
+
}
|
|
1399
|
+
// EXPORTED for the same reason `videoCostKey` is: `pricing-consistency-check.mjs`
|
|
1400
|
+
// imports it and asserts, key by key, that it equals the desktop's
|
|
1401
|
+
// `imageCreditKey`. Before that check existed the byte-match was a comment and
|
|
1402
|
+
// a hope — the two are in different repos and nothing compared them.
|
|
1403
|
+
export function imageCostKey(model, resolution, quality = 'high', aspectRatio) {
|
|
1331
1404
|
if (model === 'flux-2-max')
|
|
1332
1405
|
return resolution === '1k' ? 'flux-2-max' : `flux-2-max-${resolution}`;
|
|
1333
1406
|
if (model === 'seedream-5-lite')
|
|
@@ -1336,8 +1409,8 @@ function imageCostKey(model, resolution, quality = 'medium') {
|
|
|
1336
1409
|
return 'nano-banana-2-lite';
|
|
1337
1410
|
if (model === 'nano-banana-pro')
|
|
1338
1411
|
return `nano-banana-pro-${resolution}`;
|
|
1339
|
-
if (model
|
|
1340
|
-
return
|
|
1412
|
+
if (isGptImageModel(model))
|
|
1413
|
+
return `${model}-${gptKeyTier(quality)}-${resolution}${gptKeyAspect(aspectRatio)}`;
|
|
1341
1414
|
return `nano-banana-2-${resolution}`;
|
|
1342
1415
|
}
|
|
1343
1416
|
export const generateImage = {
|
|
@@ -1349,9 +1422,9 @@ export const generateImage = {
|
|
|
1349
1422
|
// (it still described nano-banana-2-lite by a capability the param owns).
|
|
1350
1423
|
`${describeRouting('image')}\n` +
|
|
1351
1424
|
'Full table: the slates-model-selection skill. ' +
|
|
1352
|
-
'Pass projectId to save into a Slates project (
|
|
1425
|
+
'Pass projectId to save into a Slates project (asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). ' +
|
|
1353
1426
|
CONFIRM_GATE_SENTENCE +
|
|
1354
|
-
' MCP/CLI generation always charges credits. No
|
|
1427
|
+
' MCP/CLI generation always charges credits. No skills installed? Call slates_get_prompting_guide with the model\'s topic and \'slates-cost-discipline\' first. ' +
|
|
1355
1428
|
// GENERATED from the skill file's own never-use list -- the one piece of
|
|
1356
1429
|
// prompting doctrine that is ALWAYS in context, because the agent has
|
|
1357
1430
|
// demonstrably skipped the call that would have taught it.
|
|
@@ -1360,12 +1433,13 @@ export const generateImage = {
|
|
|
1360
1433
|
prompt: z.string().min(1).max(4000),
|
|
1361
1434
|
model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
|
|
1362
1435
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Renderer refreshes live. Required for every model except nano-banana-2.'),
|
|
1363
|
-
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('
|
|
1364
|
-
quality: z.enum(['medium', 'high']).optional().describe('
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1436
|
+
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('1k drafts, 2k hero, 4k final. nano-banana-2-lite: 1k only. GPT Image classes 1024²/1080p/1440p/2160p. Never default this.'),
|
|
1437
|
+
quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image only. UNEVEN ladder: max=4× high, xhigh~1.8×. medium drafts; default high.'),
|
|
1438
|
+
backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
|
|
1439
|
+
aspectRatio: zEnum(IMAGE_ASPECT_RATIOS).optional().describe(`Pick from the use case: cinematic 16:9 · TikTok/Reels 9:16 · IG square 1:1 · ultra-wide 21:9. 1:1 costs most on GPT Image. Per model: ${describeAspectRatios(IMAGE_MODELS)}`),
|
|
1440
|
+
count: z.number().int().min(1).max(10).optional().describe('Up to 10 with projectId; headless caps at 4.'),
|
|
1441
|
+
referenceImageUrls: z.array(z.string().url()).max(14).optional().describe('Headless (no projectId) nano-banana-2 only. With a projectId, upload via slates_upload_reference_image. Label every image role in the prompt.'),
|
|
1442
|
+
referenceAssetIds: z.array(z.string()).max(16).optional().describe("Project assets as references — UUIDs or badge codes (\"IMG-A8\"), resolved at call time. Requires projectId. Caps: GPT Image 16, nano-banana-2 14, FLUX/Seedream lower. Label every reference role in the prompt."),
|
|
1369
1443
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
1370
1444
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
|
|
1371
1445
|
}),
|
|
@@ -1398,13 +1472,24 @@ export const generateImage = {
|
|
|
1398
1472
|
}
|
|
1399
1473
|
const resolution = input.resolution;
|
|
1400
1474
|
const imageModel = input.model ?? 'nano-banana-2';
|
|
1401
|
-
//
|
|
1402
|
-
//
|
|
1403
|
-
|
|
1475
|
+
// 🚨 SEEDREAM HAS A 3k CLASS TOO, AND THIS GUARD USED TO DENY IT.
|
|
1476
|
+
// It read `!== 'gpt-image-2'` and rejected every other model at 3k — but
|
|
1477
|
+
// `seedream-5-lite` declares ['2k','3k','4k'] in the desktop registry and
|
|
1478
|
+
// has a real `seedream-5-lite` cost key, so the op was refusing a request
|
|
1479
|
+
// the desktop would have served. Pre-existing; found by the 2026-09-09
|
|
1480
|
+
// audit, not introduced by the 2.5 swap.
|
|
1481
|
+
//
|
|
1482
|
+
// The ladder itself is owned by MODEL_REGISTRY in slate/src/shared/pricing.ts
|
|
1483
|
+
// and is not readable from here, so THREE_K_IMAGE_MODELS is a MIRROR — declared
|
|
1484
|
+
// and exported below so `pricing-consistency-check.mjs` compares it against
|
|
1485
|
+
// the desktop registry's own `imageResolutions`. It used to be an inline
|
|
1486
|
+
// literal with a comment admitting nothing checked it, which is how it came
|
|
1487
|
+
// to deny `seedream-5-lite` a class the desktop had always served.
|
|
1488
|
+
if (resolution === '3k' && !THREE_K_IMAGE_MODELS.includes(imageModel)) {
|
|
1404
1489
|
return ok({
|
|
1405
1490
|
requires_clarification: true,
|
|
1406
1491
|
missing: ['resolution'],
|
|
1407
|
-
message: `3k (1440p)
|
|
1492
|
+
message: `3k (1440p) exists on ${THREE_K_IMAGE_MODELS.join(', ')} — pick 1k/2k/4k for ${imageModel}.`,
|
|
1408
1493
|
});
|
|
1409
1494
|
}
|
|
1410
1495
|
// Only nano-banana-2 has a headless path — everything else routes through
|
|
@@ -1435,6 +1520,18 @@ export const generateImage = {
|
|
|
1435
1520
|
message: 'background=true routes through the desktop generation pipeline (so slates_get_generation_status can poll it) — pass a projectId, or drop background for a blocking headless run.',
|
|
1436
1521
|
});
|
|
1437
1522
|
}
|
|
1523
|
+
// 🚨 THE HEADLESS PATH ASKS FAL FOR A BATCH, so a provider ceiling binds
|
|
1524
|
+
// here and nowhere else. It is nano-banana-2 only, and Nano Banana's
|
|
1525
|
+
// `num_images` maximum is 4 (fal schema, 2026-09-09). Refused rather than
|
|
1526
|
+
// clamped: a silent clamp would make four images against a request for ten
|
|
1527
|
+
// and read to the caller as a partial failure it should retry.
|
|
1528
|
+
if (!input.projectId && (input.count ?? 1) > 4) {
|
|
1529
|
+
return ok({
|
|
1530
|
+
requires_clarification: true,
|
|
1531
|
+
missing: ['projectId'],
|
|
1532
|
+
message: 'count above 4 needs a projectId. The headless path asks fal for one batch and nano-banana-2 caps a batch at 4; with a projectId the desktop fires them as separate generations and the limit is 10.',
|
|
1533
|
+
});
|
|
1534
|
+
}
|
|
1438
1535
|
let refEcho = '';
|
|
1439
1536
|
if (referenceAssetIds.length > 0) {
|
|
1440
1537
|
await ctx.desktop().requireCapability('image-references', 'reference images on image generation');
|
|
@@ -1451,11 +1548,18 @@ export const generateImage = {
|
|
|
1451
1548
|
// New-roster models need a desktop that knows them — an older desktop's
|
|
1452
1549
|
// allowlist would silently fall back to nano-banana-2 while we quote the
|
|
1453
1550
|
// new model's price.
|
|
1454
|
-
if (input.projectId &&
|
|
1455
|
-
|
|
1551
|
+
if (input.projectId && isGptImageModel(imageModel)) {
|
|
1552
|
+
// 🚨 v3, NOT v2. GPT Image 2.5 landed 2026-09-09 with new ids and a
|
|
1553
|
+
// five-rung ladder; a desktop that only knows v2 has neither, so it would
|
|
1554
|
+
// fall back to nano-banana-2 while this op quotes a 2.5 price — precisely
|
|
1555
|
+
// the failure the gate was built to stop. Reusing v2 reintroduces it.
|
|
1556
|
+
await ctx.desktop().requireCapability('image-models-v3', `${imageModel} generation`);
|
|
1557
|
+
}
|
|
1558
|
+
else if (input.projectId &&
|
|
1559
|
+
(imageModel === 'nano-banana-pro' || imageModel === 'nano-banana-2-lite')) {
|
|
1456
1560
|
await ctx.desktop().requireCapability('image-models-v2', `${imageModel} generation`);
|
|
1457
1561
|
}
|
|
1458
|
-
const costKey = imageCostKey(imageModel, resolution, input.quality ?? '
|
|
1562
|
+
const costKey = imageCostKey(imageModel, resolution, input.quality, input.aspectRatio ?? '1:1');
|
|
1459
1563
|
const cloud = ctx.cloud();
|
|
1460
1564
|
const registry = await cloud.get('/api/agent/models');
|
|
1461
1565
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -1524,7 +1628,7 @@ export const generateImage = {
|
|
|
1524
1628
|
resolution,
|
|
1525
1629
|
aspectRatio: input.aspectRatio ?? '1:1',
|
|
1526
1630
|
count: input.count ?? 1,
|
|
1527
|
-
...(imageModel
|
|
1631
|
+
...(isGptImageModel(imageModel) ? { gptQuality: input.quality, gptBackground: input.backgroundMode } : {}),
|
|
1528
1632
|
...(referenceAssetIds.length > 0 ? { referenceAssetIds } : {}),
|
|
1529
1633
|
background: input.background,
|
|
1530
1634
|
});
|
|
@@ -1617,6 +1721,13 @@ export const generateImage = {
|
|
|
1617
1721
|
params: {
|
|
1618
1722
|
prompt: input.prompt,
|
|
1619
1723
|
aspect_ratio: input.aspectRatio ?? '1:1',
|
|
1724
|
+
// 🚨 THE HEADLESS PATH IS THE ONE PLACE WE ASK FAL FOR A BATCH, so it
|
|
1725
|
+
// is the one place a provider's own `num_images` ceiling binds — and
|
|
1726
|
+
// Nano Banana's is 4 (fal schema, read 2026-09-09), against the op's
|
|
1727
|
+
// limit of 10. Everything else fans out through the desktop as N
|
|
1728
|
+
// separate single-image generations, where no batch ceiling exists.
|
|
1729
|
+
// Guarded above rather than clamped here: silently making 4 when 10
|
|
1730
|
+
// were asked for would bill 4 and look like a partial failure.
|
|
1620
1731
|
num_images: input.count ?? 1,
|
|
1621
1732
|
...(hasReferenceImages
|
|
1622
1733
|
? { image_urls: input.referenceImageUrls }
|
|
@@ -1687,15 +1798,16 @@ async function pollProxyJob(cloud, jobId, options = {}) {
|
|
|
1687
1798
|
export const editImage = {
|
|
1688
1799
|
id: 'slates_edit_image',
|
|
1689
1800
|
billable: true,
|
|
1690
|
-
description: 'Surgically edit an
|
|
1801
|
+
description: 'Surgically edit an image asset with a text instruction (e.g. \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The result is a NEW asset (prompt prefixed \'[Edit]\'); the source is untouched. Default model nano-banana-2 (only model that also accepts referenceAssetIds); flux-2-max / seedream-5-lite use their own edit endpoints and ignore references. Before first use call slates_get_prompting_guide with topic \'slates-edit-and-iterate\'.',
|
|
1691
1802
|
input: z.object({
|
|
1692
1803
|
projectId: z.string().uuid(),
|
|
1693
|
-
sourceAssetId: z.string().uuid().describe('Image asset to edit. Must
|
|
1694
|
-
prompt: z.string().min(1).max(4000).describe('The
|
|
1695
|
-
editModel: z.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2', 'flux-2-max', 'seedream-5-lite']).optional(),
|
|
1696
|
-
referenceAssetIds: z.array(z.string().uuid()).max(13).optional().describe('Nano-Banana
|
|
1697
|
-
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('3k
|
|
1698
|
-
quality: z.enum(['medium', 'high']).optional().describe('
|
|
1804
|
+
sourceAssetId: z.string().uuid().describe('Image asset to edit. Must exist in the project.'),
|
|
1805
|
+
prompt: z.string().min(1).max(4000).describe('The change, not the whole image.'),
|
|
1806
|
+
editModel: z.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2-5-flare', 'gpt-image-2-5-sunburst', 'flux-2-max', 'seedream-5-lite']).optional(),
|
|
1807
|
+
referenceAssetIds: z.array(z.string().uuid()).max(13).optional().describe('Nano-Banana only (NB Pro 13, NB2 Lite 3).'),
|
|
1808
|
+
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('3k = GPT Image/seedream-5-lite; nano-banana-2-lite is 1k only.'),
|
|
1809
|
+
quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
|
|
1810
|
+
backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
|
|
1699
1811
|
aspectRatio: z.string().optional(),
|
|
1700
1812
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
|
|
1701
1813
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
@@ -1708,14 +1820,22 @@ export const editImage = {
|
|
|
1708
1820
|
}
|
|
1709
1821
|
const editModel = input.editModel ?? 'nano-banana-2';
|
|
1710
1822
|
const resolution = input.resolution ?? (editModel === 'nano-banana-2-lite' ? '1k' : '2k');
|
|
1711
|
-
if ((editModel
|
|
1823
|
+
if (isGptImageModel(editModel)) {
|
|
1824
|
+
// 🚨 v3, LIKE generateImage — this site was missed once already.
|
|
1825
|
+
// A pre-2.5 desktop advertises v2, so gating the 2.5 seats on v2 lets it
|
|
1826
|
+
// through; `isFalImageModel` is then false for these ids on that build and
|
|
1827
|
+
// handleEditImage falls through to the nano-banana path — NB2 output billed
|
|
1828
|
+
// at a quoted 2.5 price. Exactly the bug the gate exists to stop.
|
|
1829
|
+
await desktop.requireCapability('image-models-v3', `${editModel} editing`);
|
|
1830
|
+
}
|
|
1831
|
+
else if (editModel === 'nano-banana-pro' || editModel === 'nano-banana-2-lite') {
|
|
1712
1832
|
await desktop.requireCapability('image-models-v2', `${editModel} editing`);
|
|
1713
1833
|
}
|
|
1714
|
-
// Nano-Banana family + GPT Image 2 edits charge the same key as gen;
|
|
1834
|
+
// Nano-Banana family + GPT Image 2.5 edits charge the same key as gen;
|
|
1715
1835
|
// FLUX / Seedream route to dedicated edit endpoints priced under '-edit' keys.
|
|
1716
1836
|
const costKey = editModel === 'flux-2-max' || editModel === 'seedream-5-lite'
|
|
1717
1837
|
? `${imageCostKey(editModel, resolution)}-edit`
|
|
1718
|
-
: imageCostKey(editModel, resolution, input.quality
|
|
1838
|
+
: imageCostKey(editModel, resolution, input.quality, input.aspectRatio);
|
|
1719
1839
|
const cloud = ctx.cloud();
|
|
1720
1840
|
const registry = await cloud.get('/api/agent/models');
|
|
1721
1841
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -1742,7 +1862,7 @@ export const editImage = {
|
|
|
1742
1862
|
editModel,
|
|
1743
1863
|
referenceAssetIds: input.referenceAssetIds,
|
|
1744
1864
|
resolution,
|
|
1745
|
-
...(editModel
|
|
1865
|
+
...(isGptImageModel(editModel) ? { gptQuality: input.quality, gptBackground: input.backgroundMode } : {}),
|
|
1746
1866
|
aspectRatio: input.aspectRatio,
|
|
1747
1867
|
background: input.background,
|
|
1748
1868
|
});
|
|
@@ -2727,14 +2847,68 @@ export const generateVideo = {
|
|
|
2727
2847
|
};
|
|
2728
2848
|
},
|
|
2729
2849
|
};
|
|
2850
|
+
// ── The preset voice shelf ──────────────────────────────────────
|
|
2851
|
+
/**
|
|
2852
|
+
* AGENT PARITY for the voice picker. The desktop browses stock voices by
|
|
2853
|
+
* gender, accent and age and plays each one; until this op the agent could
|
|
2854
|
+
* only pass a `voiceId` it had no way to discover. Disk reads on the desktop,
|
|
2855
|
+
* never a vendor call — browsing is free on every surface.
|
|
2856
|
+
*/
|
|
2857
|
+
export const listVoices = {
|
|
2858
|
+
id: 'slates_list_voices',
|
|
2859
|
+
description: `Browse ${TTS_MODEL} preset voices. Pass a returned voiceId to slates_generate_audio. Filters AND together.`,
|
|
2860
|
+
input: z.object({
|
|
2861
|
+
gender: z.string().optional().describe('male | female'),
|
|
2862
|
+
accent: z.string().optional().describe('Region ("GB") or languageCode ("en-GB"); available accents come from the shelf.'),
|
|
2863
|
+
age: z.string().optional().describe('young | middle_aged | elderly'),
|
|
2864
|
+
query: z.string().optional().describe('Free text over name, description, tags.'),
|
|
2865
|
+
}),
|
|
2866
|
+
async run(input, ctx) {
|
|
2867
|
+
const desktop = ctx.desktop();
|
|
2868
|
+
await desktop.requireCapability('voices', 'the preset voice shelf');
|
|
2869
|
+
const shelf = await desktop.get('/agent/voices');
|
|
2870
|
+
if (!shelf.available) {
|
|
2871
|
+
return ok({ available: false, voices: [], message: 'This desktop build shipped without the preset shelf.' });
|
|
2872
|
+
}
|
|
2873
|
+
const terms = (input.query ?? '').toLowerCase().split(/\s+/).filter(Boolean);
|
|
2874
|
+
const region = input.accent?.includes('-') ? input.accent.split('-')[1] : input.accent;
|
|
2875
|
+
const voices = shelf.voices
|
|
2876
|
+
.filter((v) => !input.gender || v.gender === input.gender)
|
|
2877
|
+
.filter((v) => !region || v.languageCode.split('-')[1]?.toUpperCase() === region.toUpperCase())
|
|
2878
|
+
.filter((v) => !input.age || v.ageGroup === input.age)
|
|
2879
|
+
.filter((v) => {
|
|
2880
|
+
if (terms.length === 0)
|
|
2881
|
+
return true;
|
|
2882
|
+
const hay = [v.displayName, v.description, v.gender, v.ageGroup, v.languageCode, ...(v.tags ?? [])]
|
|
2883
|
+
.join(' ')
|
|
2884
|
+
.toLowerCase();
|
|
2885
|
+
return terms.every((t) => hay.includes(t));
|
|
2886
|
+
})
|
|
2887
|
+
.map(({ voiceId, displayName, description, tags, gender, ageGroup, languageCode }) => ({
|
|
2888
|
+
voiceId,
|
|
2889
|
+
displayName,
|
|
2890
|
+
description,
|
|
2891
|
+
tags,
|
|
2892
|
+
gender,
|
|
2893
|
+
ageGroup,
|
|
2894
|
+
languageCode,
|
|
2895
|
+
}));
|
|
2896
|
+
return ok({
|
|
2897
|
+
available: true,
|
|
2898
|
+
line: shelf.line,
|
|
2899
|
+
count: voices.length,
|
|
2900
|
+
voices,
|
|
2901
|
+
next: `Pass a voiceId to slates_generate_audio (model ${TTS_MODEL}) as voiceId. To keep one on a character for reuse, generate a clip with it and set slates_update_character voiceAssetId.`,
|
|
2902
|
+
});
|
|
2903
|
+
},
|
|
2904
|
+
};
|
|
2730
2905
|
// ── Generate audio ──────────────────────────────────────────────
|
|
2731
2906
|
export const generateAudio = {
|
|
2732
2907
|
id: 'slates_generate_audio',
|
|
2733
2908
|
billable: true,
|
|
2734
|
-
description:
|
|
2735
|
-
'
|
|
2736
|
-
'
|
|
2737
|
-
'projectId is REQUIRED (no headless path). ' +
|
|
2909
|
+
description: `Generate project audio using credits. Choose the surface via the model routing below. ` +
|
|
2910
|
+
'Read slates-cost-discipline and the matching prompting skill first (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). ' +
|
|
2911
|
+
'Seed Audio bills the requested duration, which is appended to the prompt regardless of output length. Kling "SFX:" / "Ambient noise:" syntax does not transfer. ' +
|
|
2738
2912
|
CONFIRM_GATE_SENTENCE +
|
|
2739
2913
|
' No skill files installed? Call slates_get_prompting_guide first.',
|
|
2740
2914
|
input: z.object({
|
|
@@ -2753,25 +2927,27 @@ export const generateAudio = {
|
|
|
2753
2927
|
durationSeconds: z
|
|
2754
2928
|
.number()
|
|
2755
2929
|
.optional()
|
|
2756
|
-
.describe(
|
|
2930
|
+
.describe(`seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (default ${SEED_AUDIO_DEFAULT_SECONDS}) — ⚠️ THIS IS THE BILL: appended to the prompt and charged whatever comes back. eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS} (default ${ELEVEN_SFX_DEFAULT_SECONDS}) — always sent explicitly so the per-second charge is deterministic. Not for ${TTS_MODEL}.`),
|
|
2757
2931
|
voice: z
|
|
2758
2932
|
.string()
|
|
2759
2933
|
.optional()
|
|
2760
|
-
.describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself
|
|
2934
|
+
.describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself. Agent-facing only.'),
|
|
2761
2935
|
voiceId: z
|
|
2762
2936
|
.string()
|
|
2763
2937
|
.optional()
|
|
2764
|
-
.describe(
|
|
2938
|
+
.describe(`${TTS_MODEL} — a preset voiceId from slates_list_voices, not a character or asset id. Exactly one voice source is required.`),
|
|
2765
2939
|
voiceReferenceAssetId: z
|
|
2766
2940
|
.string()
|
|
2767
2941
|
.optional()
|
|
2768
|
-
.describe(
|
|
2942
|
+
.describe(`${TTS_MODEL} — clone this AUDIO asset's voice for the take (${TTS_VOICE_CLONE.minSeconds}-${TTS_VOICE_CLONE.maxSeconds}s, one clean speaker). To speak AS a character pass its voiceAssetId. Cloning: ${TTS_VOICE_CLONE.clonesPerMinute} new voices/min across all of Slates; a burst waits.`),
|
|
2769
2943
|
voiceDescription: z
|
|
2770
2944
|
.string()
|
|
2945
|
+
.min(TTS_VOICE_CLONE.designPromptChars.min)
|
|
2946
|
+
.max(TTS_VOICE_CLONE.designPromptChars.max)
|
|
2771
2947
|
.optional()
|
|
2772
|
-
.describe(
|
|
2948
|
+
.describe(`${TTS_MODEL} — a voice from words (${TTS_VOICE_CLONE.designPromptChars.min}-${TTS_VOICE_CLONE.designPromptChars.max} chars) for a character with no recording; keep it via slates_update_character voiceAssetId.`),
|
|
2773
2949
|
speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
|
|
2774
|
-
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline
|
|
2950
|
+
volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline fader for mix decisions.'),
|
|
2775
2951
|
pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
|
|
2776
2952
|
multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
|
|
2777
2953
|
loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
|
|
@@ -2780,7 +2956,7 @@ export const generateAudio = {
|
|
|
2780
2956
|
.array(z.string())
|
|
2781
2957
|
.max(3)
|
|
2782
2958
|
.optional()
|
|
2783
|
-
.describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId
|
|
2959
|
+
.describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId.'),
|
|
2784
2960
|
imageReferenceAssetId: z
|
|
2785
2961
|
.string()
|
|
2786
2962
|
.optional()
|
|
@@ -2823,7 +2999,7 @@ export const generateAudio = {
|
|
|
2823
2999
|
return ok({
|
|
2824
3000
|
requires_clarification: true,
|
|
2825
3001
|
missing: ['voiceId'],
|
|
2826
|
-
message: `${TTS_MODEL} needs a voice
|
|
3002
|
+
message: `${TTS_MODEL} needs a voice — exactly one of three. Speaking AS a character: pass its voiceAssetId (slates_list_characters) as voiceReferenceAssetId. A stock voice: slates_list_voices lists presets by gender, accent and age; pass one's voiceId. No recording of the voice: voiceDescription (words), or voiceReferenceAssetId with any clean clip of one speaker. A voice worth reusing can be kept on a character with slates_update_character, but nothing requires that — ask the user which they want only when the request does not say.`,
|
|
2827
3003
|
});
|
|
2828
3004
|
}
|
|
2829
3005
|
if (voiceSources.length > 1) {
|
|
@@ -3725,17 +3901,33 @@ export const setFolderCover = {
|
|
|
3725
3901
|
};
|
|
3726
3902
|
export const updateCharacter = {
|
|
3727
3903
|
id: 'slates_update_character',
|
|
3728
|
-
description: 'Update a character\'s name, description, or
|
|
3904
|
+
description: 'Update a character\'s name, description, style, or voice. Use slates_set_character_identity_asset for its canonical image.',
|
|
3729
3905
|
input: z.object({
|
|
3730
3906
|
characterId: z.string().uuid(),
|
|
3731
3907
|
name: z.string().min(1).max(120).optional(),
|
|
3732
3908
|
description: z.string().optional(),
|
|
3733
3909
|
style: z.string().max(200).optional().describe("Art style. Omit to inherit the reference's style (the default). Canonical styles: photoreal, anime, painterly, 3d-render, comic. Or pass any free-text instruction, e.g. 'turn this into a real person'."),
|
|
3910
|
+
// Agent parity for the character card's voice slot: the desktop route has
|
|
3911
|
+
// taken this since 2026-08-28; the op never exposed it, so an agent could
|
|
3912
|
+
// render a voice and had no way to keep it on the character.
|
|
3913
|
+
voiceAssetId: z
|
|
3914
|
+
.string()
|
|
3915
|
+
.uuid()
|
|
3916
|
+
.nullable()
|
|
3917
|
+
.optional()
|
|
3918
|
+
.describe("The AUDIO asset that is this character's voice (what inworld-tts-2 clones for its lines); null detaches, the clip stays."),
|
|
3734
3919
|
}),
|
|
3735
3920
|
async run(input, ctx) {
|
|
3736
3921
|
return ok(await ctx.desktop().post('/agent/characters/update', {
|
|
3737
3922
|
id: input.characterId,
|
|
3738
|
-
data: {
|
|
3923
|
+
data: {
|
|
3924
|
+
name: input.name,
|
|
3925
|
+
description: input.description,
|
|
3926
|
+
style: input.style,
|
|
3927
|
+
// Sent only when given: the route treats presence as intent, and an
|
|
3928
|
+
// explicit null is the detach.
|
|
3929
|
+
...(input.voiceAssetId !== undefined ? { voiceAssetId: input.voiceAssetId } : {}),
|
|
3930
|
+
},
|
|
3739
3931
|
}));
|
|
3740
3932
|
},
|
|
3741
3933
|
};
|
|
@@ -4021,16 +4213,22 @@ const SHOT_ASPECT_RATIOS = [...new Set([...VIDEO_ASPECT_RATIOS, ...IMAGE_ASPECT_
|
|
|
4021
4213
|
function shotParamsShape(described) {
|
|
4022
4214
|
const d = (node, text) => (described ? node.describe(text) : node);
|
|
4023
4215
|
return {
|
|
4024
|
-
aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the
|
|
4025
|
-
duration: d(z.number().int().min(1).max(360).optional(), 'Seconds
|
|
4026
|
-
videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the
|
|
4216
|
+
aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the model; see slates_generate_video.'),
|
|
4217
|
+
duration: d(z.number().int().min(1).max(360).optional(), 'Seconds for video or duration-based audio; TTS uses text length.'),
|
|
4218
|
+
videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the model when the Shot is saved.'),
|
|
4027
4219
|
imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
|
|
4028
|
-
gptQuality: d(z.enum(['medium', 'high']).optional(), '
|
|
4029
|
-
|
|
4220
|
+
gptQuality: d(z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional(), 'GPT Image 2.5 only.'),
|
|
4221
|
+
gptBackground: d(z.enum(['auto', 'transparent', 'opaque']).optional(), 'GPT Image only.'),
|
|
4222
|
+
imageQuantity: d(z.number().int().min(1).max(10).optional(), 'Image models only.'),
|
|
4030
4223
|
negativePrompt: z.string().optional(),
|
|
4031
4224
|
sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
|
|
4032
|
-
seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider
|
|
4225
|
+
seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider, ~45% more."),
|
|
4033
4226
|
audioDurationSeconds: d(z.number().int().min(1).max(120).optional(), 'Audio lane. On seed-audio the requested duration IS the bill.'),
|
|
4227
|
+
// The TTS voice — the same three fields slates_generate_audio takes, so a
|
|
4228
|
+
// Shot is the audio call, serialized. Exactly one of them, enforced at fire.
|
|
4229
|
+
voiceId: d(z.string().optional(), `${TTS_MODEL}: preset voiceId (slates_list_voices).`),
|
|
4230
|
+
voiceReferenceAssetId: d(z.string().optional(), `${TTS_MODEL}: audio asset to clone.`),
|
|
4231
|
+
voiceDescription: d(z.string().optional(), `${TTS_MODEL}: the voice in words.`),
|
|
4034
4232
|
};
|
|
4035
4233
|
}
|
|
4036
4234
|
const shotParamsSchema = z.object(shotParamsShape(true)).optional();
|
|
@@ -4047,10 +4245,7 @@ const shotParamsSchemaTerse = z
|
|
|
4047
4245
|
* ship a field with no explanation — the same failure as a column nothing
|
|
4048
4246
|
* renders.
|
|
4049
4247
|
*
|
|
4050
|
-
*
|
|
4051
|
-
* surface; the prompt is the only thing the request carries. The one exception
|
|
4052
|
-
* is pre-existing: a multiShotSegment still prepends its own camera and
|
|
4053
|
-
* shotSize to its own segment prompt.
|
|
4248
|
+
* Script fields supply prompt prose when no authored prompt exists (shot-spec.ts).
|
|
4054
4249
|
*/
|
|
4055
4250
|
function shotScriptShape(described) {
|
|
4056
4251
|
const text = Object.fromEntries(SCRIPT_TEXT_FIELDS.map((field) => [
|
|
@@ -4151,6 +4346,8 @@ function shotRefInputs(input) {
|
|
|
4151
4346
|
out.push({ ref: input.firstFrameAssetId, role: 'first frame' });
|
|
4152
4347
|
if (input.lastFrameAssetId)
|
|
4153
4348
|
out.push({ ref: input.lastFrameAssetId, role: 'last frame' });
|
|
4349
|
+
if (input.params?.voiceReferenceAssetId)
|
|
4350
|
+
out.push({ ref: input.params.voiceReferenceAssetId, role: 'voice reference' });
|
|
4154
4351
|
return out;
|
|
4155
4352
|
}
|
|
4156
4353
|
/**
|
|
@@ -4187,7 +4384,10 @@ async function buildShotSpecInput(ctx, projectId, input) {
|
|
|
4187
4384
|
// later diverges from it and the desktop card says so — the prompt is
|
|
4188
4385
|
// never rewritten (that is prompt enhancement, deleted 2026-08-01).
|
|
4189
4386
|
authoredFor: input.model ?? null,
|
|
4190
|
-
params:
|
|
4387
|
+
params: {
|
|
4388
|
+
...shotParamsPatch(input.params),
|
|
4389
|
+
...(input.params?.voiceReferenceAssetId ? { voiceReferenceAssetId: rid(input.params.voiceReferenceAssetId) } : {}),
|
|
4390
|
+
},
|
|
4191
4391
|
mentions: {
|
|
4192
4392
|
characterIds: input.characterIds ?? [],
|
|
4193
4393
|
environmentIds: input.environmentIds ?? [],
|
|
@@ -4240,12 +4440,12 @@ function shotCostKey(detail) {
|
|
|
4240
4440
|
// has none, so it falls back to the raw ones and is announced as a floor.
|
|
4241
4441
|
const fires = detail.firesWith;
|
|
4242
4442
|
if (AUDIO_MODELS.includes(model)) {
|
|
4243
|
-
|
|
4244
|
-
|
|
4245
|
-
|
|
4246
|
-
|
|
4247
|
-
|
|
4248
|
-
|
|
4443
|
+
if (model === TTS_MODEL) {
|
|
4444
|
+
const text = detail.composedPrompt ?? (detail.rawPrompt.trim() || detail.line?.trim() || '');
|
|
4445
|
+
if (!text || text.length > TTS_MAX_CHARACTERS)
|
|
4446
|
+
return null;
|
|
4447
|
+
return audioCostKey({ model, characters: text.length });
|
|
4448
|
+
}
|
|
4249
4449
|
const seconds = fires?.audioDurationSeconds ?? p.audioDurationSeconds;
|
|
4250
4450
|
if (!seconds)
|
|
4251
4451
|
return null;
|
|
@@ -4275,7 +4475,15 @@ function shotCostKey(detail) {
|
|
|
4275
4475
|
}
|
|
4276
4476
|
if (IMAGE_MODELS.includes(model)) {
|
|
4277
4477
|
return imageCostKey(model, (fires?.imageResolution ?? p.imageResolution) ??
|
|
4278
|
-
(model === 'nano-banana-2-lite' ? '1k' : '2k'),
|
|
4478
|
+
(model === 'nano-banana-2-lite' ? '1k' : '2k'),
|
|
4479
|
+
// No inline default — see the note at the estimate op. Firing a Shot with
|
|
4480
|
+
// no stored tier lands on `high` via the desktop's `normalizeGptQuality`,
|
|
4481
|
+
// and `imageCostKey`'s parameter default is pinned to match it.
|
|
4482
|
+
p.gptQuality,
|
|
4483
|
+
// The aspect the Shot will fire at — on GPT Image it moves the key, so
|
|
4484
|
+
// quoting without it under-prices every square Shot. `fires` carries
|
|
4485
|
+
// only the clamped resolution/duration axes, never the aspect.
|
|
4486
|
+
p.aspectRatio);
|
|
4279
4487
|
}
|
|
4280
4488
|
return null;
|
|
4281
4489
|
}
|
|
@@ -4453,8 +4661,17 @@ export const duplicateShot = {
|
|
|
4453
4661
|
spec.prompt = input.prompt;
|
|
4454
4662
|
if (input.model !== undefined)
|
|
4455
4663
|
spec.model = input.model;
|
|
4456
|
-
if (input.params !== undefined)
|
|
4457
|
-
|
|
4664
|
+
if (input.params !== undefined) {
|
|
4665
|
+
const voiceRef = input.params.voiceReferenceAssetId;
|
|
4666
|
+
if (voiceRef && !UUID_RE.test(voiceRef)) {
|
|
4667
|
+
const { shot } = await desktop.get('/agent/shots/get', { id: input.shotId });
|
|
4668
|
+
const built = await buildShotSpecInput(ctx, shot.projectId, { params: input.params });
|
|
4669
|
+
spec.params = built.spec.params;
|
|
4670
|
+
}
|
|
4671
|
+
else {
|
|
4672
|
+
spec.params = shotParamsPatch(input.params);
|
|
4673
|
+
}
|
|
4674
|
+
}
|
|
4458
4675
|
const r = await desktop.post('/agent/shots/duplicate', {
|
|
4459
4676
|
id: input.shotId,
|
|
4460
4677
|
name: input.name,
|
|
@@ -4715,7 +4932,7 @@ export const generateFromShots = {
|
|
|
4715
4932
|
// something to try again spends credits before anyone notices.
|
|
4716
4933
|
`\n${failedLines.join('\n')}\nThese were NOT retried. Read each error, fix the Shot, and re-fire only what you meant to.`
|
|
4717
4934
|
: '') +
|
|
4718
|
-
` ${
|
|
4935
|
+
` ${BACKGROUND_REVIEW_POINTER}`);
|
|
4719
4936
|
},
|
|
4720
4937
|
};
|
|
4721
4938
|
function resolveGuideTopic(topic) {
|
|
@@ -4763,7 +4980,7 @@ function resolveGuideTopic(topic) {
|
|
|
4763
4980
|
if (t.startsWith('nano-banana'))
|
|
4764
4981
|
return 'slates-prompting-nano-banana-2';
|
|
4765
4982
|
if (t.startsWith('gpt-image') || t.startsWith('gpt image'))
|
|
4766
|
-
return 'slates-prompting-gpt-image-2';
|
|
4983
|
+
return 'slates-prompting-gpt-image-2-5';
|
|
4767
4984
|
if (t.startsWith('flux'))
|
|
4768
4985
|
return 'slates-prompting-flux-2-max';
|
|
4769
4986
|
if (t.startsWith('seedream'))
|
|
@@ -4855,7 +5072,7 @@ function resolveGuideTopic(topic) {
|
|
|
4855
5072
|
* The guide index, GENERATED from SKILLS.
|
|
4856
5073
|
*
|
|
4857
5074
|
* The list here was hand-typed and had drifted to 25 of 32 names — the prompting
|
|
4858
|
-
* guides for GPT Image
|
|
5075
|
+
* guides for GPT Image, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
|
|
4859
5076
|
* all missing, so an agent reading this description could not learn they exist.
|
|
4860
5077
|
* A hand-typed index of a generated corpus is a stale index; it is only a matter
|
|
4861
5078
|
* of when.
|
|
@@ -4874,15 +5091,16 @@ function describeGuideTopics() {
|
|
|
4874
5091
|
}
|
|
4875
5092
|
export const getPromptingGuide = {
|
|
4876
5093
|
id: 'slates_get_prompting_guide',
|
|
4877
|
-
description:
|
|
4878
|
-
|
|
4879
|
-
|
|
4880
|
-
|
|
4881
|
-
|
|
4882
|
-
|
|
4883
|
-
|
|
4884
|
-
|
|
5094
|
+
description: 'For app help and exact UI instructions use topic "app-manual" with a query such as "voice recording". This returns the canonical product manual, shared by every agent surface. ' +
|
|
5095
|
+
// 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
|
|
5096
|
+
// MEASURED at 13% compliance before and after the enforcement work — pointer
|
|
5097
|
+
// prose is the shape that does not move the agent. What replaced it is
|
|
5098
|
+
// structural: the never-use list rides the generate ops' descriptions and
|
|
5099
|
+
// the craft card rides the estimate result, so the facts arrive whether or
|
|
5100
|
+
// not this op is ever called.
|
|
5101
|
+
"Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
|
|
4885
5102
|
input: z.object({
|
|
5103
|
+
query: z.string().max(200).optional().describe('For app-manual: keywords to retrieve relevant UI sections. Omit for the entire manual.'),
|
|
4886
5104
|
topic: z
|
|
4887
5105
|
.string()
|
|
4888
5106
|
.min(1)
|
|
@@ -4890,6 +5108,10 @@ export const getPromptingGuide = {
|
|
|
4890
5108
|
depth: z.enum(['card', 'full']).optional().describe('"card" returns just the levers block (a few hundred words — the same card slates_estimate_generation_cost already attached, so usually redundant). "full" (default) returns the whole guide, up to several thousand words.'),
|
|
4891
5109
|
}),
|
|
4892
5110
|
async run(input) {
|
|
5111
|
+
if (input.topic.trim().toLowerCase() === 'app-manual') {
|
|
5112
|
+
const content = appManualSections(input.query);
|
|
5113
|
+
return { text: content, data: { topic: 'app-manual', bytes: Buffer.byteLength(content, 'utf8') } };
|
|
5114
|
+
}
|
|
4893
5115
|
const resolved = resolveGuideTopic(input.topic);
|
|
4894
5116
|
const content = resolved ? SKILLS[resolved] : undefined;
|
|
4895
5117
|
if (!resolved || content === undefined) {
|
|
@@ -5124,6 +5346,7 @@ export const ALL_OPERATIONS = [
|
|
|
5124
5346
|
generateImage,
|
|
5125
5347
|
generateVideo,
|
|
5126
5348
|
generateAudio,
|
|
5349
|
+
listVoices,
|
|
5127
5350
|
generateLipSync,
|
|
5128
5351
|
generateMotionTransfer,
|
|
5129
5352
|
editVideo,
|