@slatesvideo/shared 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/api-url.d.ts +5 -3
  2. package/dist/api-url.js +5 -3
  3. package/dist/index.d.ts +1 -0
  4. package/dist/index.js +1 -0
  5. package/dist/manual/content.d.ts +2 -0
  6. package/dist/manual/content.js +3 -0
  7. package/dist/manual/index.d.ts +5 -0
  8. package/dist/manual/index.js +20 -0
  9. package/dist/operations/index.d.ts +67 -17
  10. package/dist/operations/index.js +310 -87
  11. package/dist/prompts/agent-doctrine.js +1 -0
  12. package/dist/prompts/character-sheet.js +10 -0
  13. package/dist/prompts/model-capabilities.js +52 -3
  14. package/dist/prompts/model-facts.js +12 -4
  15. package/dist/prompts/prompting-tips.js +9 -3
  16. package/dist/prompts/reference-composer.js +14 -0
  17. package/dist/prompts/shot-spec.d.ts +30 -10
  18. package/dist/prompts/shot-spec.js +41 -9
  19. package/dist/skills/content.js +12 -12
  20. package/exports/slates-prompt-builder/generated/reference-character.md +1 -1
  21. package/exports/slates-prompt-builder/generated/reference-seedance.md +3 -1
  22. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +10 -10
  23. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  24. package/package.json +1 -1
  25. package/skills/slates-character-identity.md +1 -1
  26. package/skills/slates-model-selection.md +10 -7
  27. package/skills/slates-prompting-elevenlabs.md +1 -1
  28. package/skills/slates-prompting-gpt-image-2-5.md +183 -0
  29. package/skills/slates-prompting-inworld-tts.md +174 -166
  30. package/skills/slates-prompting-lip-sync.md +1 -1
  31. package/skills/slates-prompting-nano-banana-2.md +1 -1
  32. package/skills/slates-prompting-seed-audio.md +1 -1
  33. package/skills/slates-prompting-seedance-2-5.md +3 -1
  34. package/skills/slates-prompting-seedance.md +3 -1
  35. package/skills/slates-ugc-influencer-ad.md +5 -3
  36. package/skills/slates-vision-feedback-loop.md +2 -2
  37. package/skills/slates-prompting-gpt-image-2.md +0 -109
@@ -14,6 +14,7 @@ import { SlatesCloudClient } from '../clients/cloud.js';
14
14
  import { SlatesDesktopClient } from '../clients/desktop.js';
15
15
  import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
16
16
  import { SKILLS } from '../skills/content.js';
17
+ import { appManualSections } from '../manual/index.js';
17
18
  // Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
18
19
  // "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
19
20
  import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
@@ -183,6 +184,15 @@ export const TTS_MAX_CHARACTERS = (() => {
183
184
  return max;
184
185
  })();
185
186
  export const TTS_BUCKET_COUNT = TTS_MAX_CHARACTERS / TTS_BUCKET_CHARS; // 8
187
+ /** The seat's cloning spec — reference-clip bounds, the design-prompt bounds
188
+ * and the workspace-wide clone rate — read from the SSOT for the same reason
189
+ * as the cap: every number in it was MEASURED and lives in exactly one row. */
190
+ export const TTS_VOICE_CLONE = (() => {
191
+ const spec = getModelCapability(TTS_MODEL)?.voiceClone;
192
+ if (!spec)
193
+ throw new Error(`MODEL_CAPABILITIES['${TTS_MODEL}'] must declare voiceClone`);
194
+ return spec;
195
+ })();
186
196
  function creditCost(m) {
187
197
  if (!m)
188
198
  return 0;
@@ -201,7 +211,7 @@ function creditsFromDollars(dollars) {
201
211
  // Shared describe-text for the background flag on every generate_* op. ONE
202
212
  // sentence: it is repeated verbatim on seven ops, so every word costs seven
203
213
  // times, and `slates_get_generation_status` explains the polling itself.
204
- const BACKGROUND_DESCRIBE = 'Return generationId(s) immediately instead of blocking; poll slates_get_generation_status. Recommended for video.';
214
+ const BACKGROUND_DESCRIBE = 'Return generationId(s) now instead of blocking; poll slates_get_generation_status. Recommended for video.';
205
215
  // ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
206
216
  //
207
217
  // QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
@@ -219,7 +229,7 @@ const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it
219
229
  const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
220
230
  'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
221
231
  const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
222
- 'slates_get_asset_video_frames for video.';
232
+ 'slates_get_asset_video_frames for video. For audio, audition the saved file; metadata alone does not establish voice similarity or delivery quality.';
223
233
  // The image saved, but reading it back off disk failed (best-effort fetch). The
224
234
  // agent has an asset and NO pixels, which is the one state where a quality
225
235
  // claim would be pure invention — so this branch has to say so rather than
@@ -492,8 +502,9 @@ export const estimateGenerationCost = {
492
502
  duration: z.number().int().min(1).max(360).optional().describe(`Seconds; cost scales linearly. Required with a video or PER-SECOND audio base id. Per-model windows: see slates_generate_video's duration. Audio: seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (⚠️ the requested duration IS the bill), eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. ⛔ NOT for ${TTS_MODEL} — pass \`characters\`.`),
493
503
  characters: z.number().int().min(1).max(TTS_MAX_CHARACTERS).optional().describe(`${TTS_MODEL} only — the LENGTH OF THE TEXT to speak (${TTS_BUCKET_CHARS}-char buckets).`),
494
504
  videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe('Video only. Omitted, each model quotes at its own default. Per-model ladders: see slates_generate_video\'s videoResolution.'),
495
- resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
496
- quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
505
+ resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k: GPT Image/seedream-5-lite).'),
506
+ quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
507
+ aspectRatio: z.string().optional().describe('Image only. 1:1/4:3/3:4 cost more than 16:9.'),
497
508
  sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
498
509
  seedanceFace: z.boolean().optional().describe('Seedance AI-face route (pricier key).'),
499
510
  seedanceRealFace: z.boolean().optional().describe('Seedance consented real-face route (premium key).'),
@@ -504,15 +515,21 @@ export const estimateGenerationCost = {
504
515
  const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
505
516
  // 1) exact registry cost key
506
517
  let key = byKey.has(input.model) ? input.model : null;
507
- // 2) image base id + resolution (+ quality for gpt-image-2)
518
+ // 2) image base id + resolution (+ quality for GPT Image 2.5)
508
519
  if (!key) {
509
520
  // IMAGE_MODELS, never a second hand-typed copy: this list is declared
510
521
  // below (a runtime read, so no temporal-dead-zone hazard) and is the same
511
522
  // enum `slates_generate_image` accepts. Two copies is how the estimate op
512
523
  // would quietly stop pricing the seventh image model.
513
524
  const img = IMAGE_MODELS.find((m) => m === input.model);
525
+ // ⚙ NO INLINE DEFAULT. `imageCostKey`'s own parameter default is the ONE
526
+ // home for the fallback tier, and `pricing-consistency-check.mjs` pins it
527
+ // to the desktop's. This line read `?? 'medium'` while every generate path
528
+ // billed `high`, so the op the doctrine tells agents to call before every
529
+ // generation quoted 1 cr for a 2 cr job. Four separate copies of one
530
+ // default is what made that possible; there are now none.
514
531
  if (img)
515
- key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality ?? 'medium');
532
+ key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality, input.aspectRatio);
516
533
  }
517
534
  // 2a) audio base id → seconds. Both surfaces bill per second, so a
518
535
  // duration is always required. Runs BEFORE the video resolver: it is
@@ -632,7 +649,7 @@ export const estimateGenerationCost = {
632
649
  if (key == null || perCredits == null) {
633
650
  // Every id in the error comes from the SSOT arrays. The image half was
634
651
  // hand-typed and named three of six, so an agent that mis-spelled
635
- // `gpt-image-2` was told the model did not exist.
652
+ // `gpt-image-2-5-flare` was told the model did not exist.
636
653
  throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | ${IMAGE_MODELS.join(' | ')}) plus duration/resolution params, or use slates_list_available_models with a filter.`);
637
654
  }
638
655
  const qty = input.quantity ?? 1;
@@ -1113,11 +1130,23 @@ export const generateCharacterIdentity = {
1113
1130
  baseAssetId: z.string().uuid().describe('The base portrait asset the identity is generated from.'),
1114
1131
  userNotes: z.string().optional().describe('Extra instruction, e.g. "use the woman on the left".'),
1115
1132
  model: z
1116
- .enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2'])
1133
+ .enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2-5-flare', 'gpt-image-2-5-sunburst'])
1117
1134
  .optional()
1118
1135
  .describe('Image model for the sheet. Omit for the default (nano-banana-2). Exists so the layout-vs-face tradeoff can be tested with comparison gens — do not switch without a receipt.'),
1119
1136
  }),
1120
1137
  async run(input, ctx) {
1138
+ // 🚨 THE ROSTER GATE, WHICH THIS OP NEVER HAD. Its `model` enum offers
1139
+ // seats an older desktop does not know, and the handler resolves an unknown
1140
+ // id by falling back to the Google/Banana path — so the op would quote one
1141
+ // model's price and generate another, silently. Same failure `generateImage`
1142
+ // and `editImage` gate against; this op was simply missed when the gate was
1143
+ // introduced, and the 2.5 swap is what makes it reachable in practice.
1144
+ if (isGptImageModel(input.model)) {
1145
+ await ctx.desktop().requireCapability('image-models-v3', `${input.model} character identity`);
1146
+ }
1147
+ else if (input.model === 'nano-banana-pro' || input.model === 'nano-banana-2-lite') {
1148
+ await ctx.desktop().requireCapability('image-models-v2', `${input.model} character identity`);
1149
+ }
1121
1150
  return ok(await ctx.desktop().post('/agent/characters/generate-identity', {
1122
1151
  characterId: input.characterId,
1123
1152
  projectId: input.projectId,
@@ -1299,21 +1328,39 @@ async function previewAssets(ctx, refs) {
1299
1328
  return out;
1300
1329
  }
1301
1330
  /** The exact `model` ids `slates_generate_image` accepts. */
1331
+ /**
1332
+ * The image models whose ladder includes the 3k (1440p) class — a MIRROR of
1333
+ * `imageResolutions` in slate's MODEL_REGISTRY, which this package cannot read.
1334
+ * Exported so `pricing-consistency-check.mjs` can prove the mirror still
1335
+ * matches; without that proof a model that gains 3k in the registry just goes
1336
+ * quietly unreachable through the op.
1337
+ */
1338
+ export const THREE_K_IMAGE_MODELS = [
1339
+ 'gpt-image-2-5-flare',
1340
+ 'gpt-image-2-5-sunburst',
1341
+ 'seedream-5-lite',
1342
+ ];
1302
1343
  export const IMAGE_MODELS = [
1303
1344
  'nano-banana-2',
1304
1345
  'nano-banana-2-lite',
1305
1346
  'nano-banana-pro',
1306
- 'gpt-image-2',
1347
+ 'gpt-image-2-5-flare',
1348
+ 'gpt-image-2-5-sunburst',
1307
1349
  'flux-2-max',
1308
1350
  'seedream-5-lite',
1309
1351
  ];
1352
+ /** True for either GPT Image 2.5 seat. Flare and Sunburst differ in latency
1353
+ * and routing advice, never in pricing shape or param surface. */
1354
+ export function isGptImageModel(model) {
1355
+ return model === 'gpt-image-2-5-flare' || model === 'gpt-image-2-5-sunburst';
1356
+ }
1310
1357
  /**
1311
1358
  * The aspect ratios an image generation can carry — the UNION over what these
1312
1359
  * models declare in `MODEL_CAPABILITIES`, generated so the enum cannot hold a
1313
1360
  * value no model accepts. It used to be a hand-typed eleven-value list whose
1314
1361
  * `9:21` exists in ZERO models; that phantom is gone by construction.
1315
1362
  *
1316
- * ⚠️ STILL A UNION, NOT A PER-MODEL CHECK. `gpt-image-2` takes five of these
1363
+ * ⚠️ STILL A UNION, NOT A PER-MODEL CHECK. GPT Image 2.5 takes five of these
1317
1364
  * ten and the other five would be accepted here. The image param surface has
1318
1365
  * not been audited (aspect ratios, resolution classes, per-model reference
1319
1366
  * caps) — that audit is the named follow-up in
@@ -1325,9 +1372,35 @@ const IMAGE_ASPECT_RATIOS = aspectRatioUnion(IMAGE_MODELS);
1325
1372
  // Registry cost-key for an image model+resolution. Mirrors imageCreditKey()
1326
1373
  // in slate/src/shared/pricing.ts — MUST byte-match it (the same hard rule as
1327
1374
  // videoCostKey): NB2/NB Pro price per resolution, FLUX.2 Max prices per
1328
- // resolution (1k is the bare key), NB2 Lite/Seedream are flat, GPT Image 2 is
1329
- // quality × resolution-class (med|high; low deliberately not exposed).
1330
- function imageCostKey(model, resolution, quality = 'medium') {
1375
+ // resolution (1k is the bare key), NB2 Lite/Seedream are flat, GPT Image 2.5 is
1376
+ // quality × resolution-class over ALL FIVE exposed tiers.
1377
+ //
1378
+ // 🚨 THE KEY SUFFIX IS `med` WHILE THE WIRE VALUE IS `medium` — the one
1379
+ // tier name that differs between fal's enum and our cost keys, inherited from
1380
+ // GPT Image 2 (`gpt-image-2-med-4k`). `gptKeyTier` is that whole translation,
1381
+ // and it must stay byte-identical to `gptKeyTier` in
1382
+ // slate/src/shared/pricing.ts — pricing-consistency-check.mjs enforces it.
1383
+ function gptKeyTier(quality) {
1384
+ return quality === 'medium' ? 'med' : quality;
1385
+ }
1386
+ // 🚨 THE ASPECT RATIO IS PART OF THE PRICE ON GPT IMAGE. fal bills image
1387
+ // OUTPUT TOKENS and the count tracks the frame's SHAPE — metered 2026-09-09,
1388
+ // 4:3/3:4 cost ~4/3 of the class rate and 1:1 ~16/9 of it. Quoting one price
1389
+ // for every aspect sold 1:1 below cost at `high` and above. Byte-identical to
1390
+ // `gptKeyAspect` in slate/src/shared/pricing.ts; pricing-consistency-check
1391
+ // proves it key by key. 16:9 and 9:16 keep the bare key they always had.
1392
+ function gptKeyAspect(aspectRatio) {
1393
+ if (aspectRatio === '1:1')
1394
+ return '-sq';
1395
+ if (aspectRatio === '4:3' || aspectRatio === '3:4')
1396
+ return '-43';
1397
+ return '';
1398
+ }
1399
+ // EXPORTED for the same reason `videoCostKey` is: `pricing-consistency-check.mjs`
1400
+ // imports it and asserts, key by key, that it equals the desktop's
1401
+ // `imageCreditKey`. Before that check existed the byte-match was a comment and
1402
+ // a hope — the two are in different repos and nothing compared them.
1403
+ export function imageCostKey(model, resolution, quality = 'high', aspectRatio) {
1331
1404
  if (model === 'flux-2-max')
1332
1405
  return resolution === '1k' ? 'flux-2-max' : `flux-2-max-${resolution}`;
1333
1406
  if (model === 'seedream-5-lite')
@@ -1336,8 +1409,8 @@ function imageCostKey(model, resolution, quality = 'medium') {
1336
1409
  return 'nano-banana-2-lite';
1337
1410
  if (model === 'nano-banana-pro')
1338
1411
  return `nano-banana-pro-${resolution}`;
1339
- if (model === 'gpt-image-2')
1340
- return `gpt-image-2-${quality === 'high' ? 'high' : 'med'}-${resolution}`;
1412
+ if (isGptImageModel(model))
1413
+ return `${model}-${gptKeyTier(quality)}-${resolution}${gptKeyAspect(aspectRatio)}`;
1341
1414
  return `nano-banana-2-${resolution}`;
1342
1415
  }
1343
1416
  export const generateImage = {
@@ -1349,9 +1422,9 @@ export const generateImage = {
1349
1422
  // (it still described nano-banana-2-lite by a capability the param owns).
1350
1423
  `${describeRouting('image')}\n` +
1351
1424
  'Full table: the slates-model-selection skill. ' +
1352
- 'Pass projectId to save into a Slates project (recommended — asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). ' +
1425
+ 'Pass projectId to save into a Slates project (asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). ' +
1353
1426
  CONFIRM_GATE_SENTENCE +
1354
- ' MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use. ' +
1427
+ ' MCP/CLI generation always charges credits. No skills installed? Call slates_get_prompting_guide with the model\'s topic and \'slates-cost-discipline\' first. ' +
1355
1428
  // GENERATED from the skill file's own never-use list -- the one piece of
1356
1429
  // prompting doctrine that is ALWAYS in context, because the agent has
1357
1430
  // demonstrably skipped the call that would have taught it.
@@ -1360,12 +1433,13 @@ export const generateImage = {
1360
1433
  prompt: z.string().min(1).max(4000),
1361
1434
  model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
1362
1435
  projectId: z.string().uuid().optional().describe('Save into this Slates project. Renderer refreshes live. Required for every model except nano-banana-2.'),
1363
- resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Pick deliberately: 1k drafts, 2k hero shots, 4k print/final. nano-banana-2-lite is 1k-only. gpt-image-2 classes: 1k=1024², 2k=1080p, 3k=1440p, 4k=2160p (3k is gpt-image-2 only). Never default this.'),
1364
- quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only. medium (default) = sharp text, fast, the value seat; high = max text precision + reasoning at ~4× the price. Ignored by other models.'),
1365
- aspectRatio: zEnum(IMAGE_ASPECT_RATIOS).optional().describe(`Pick deliberately from the use case. Cinematic → 16:9. TikTok/Reels/Story → 9:16. IG square → 1:1. Ultra-wide → 21:9. Ask the user when ambiguous. Per model: ${describeAspectRatios(IMAGE_MODELS)}`),
1366
- count: z.number().int().min(1).max(4).optional(),
1367
- referenceImageUrls: z.array(z.string().url()).max(14).optional().describe('Headless (no-projectId) nano-banana-2 only: up to 14 ref URLs. For projectId runs, upload refs with slates_upload_reference_image first. Always label each image\'s role in the prompt text.'),
1368
- referenceAssetIds: z.array(z.string()).max(14).optional().describe("Project assets to use as reference/ingredient images — asset UUIDs or badge codes (\"IMG-A8\"); codes resolve against the project at call time. Requires projectId. For nano-banana-2 up to 14 refs; FLUX/Seedream route to their edit endpoints with lower per-model caps. Label each reference's role in the prompt text."),
1436
+ resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('1k drafts, 2k hero, 4k final. nano-banana-2-lite: 1k only. GPT Image classes 1024²/1080p/1440p/2160p. Never default this.'),
1437
+ quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image only. UNEVEN ladder: max=4× high, xhigh~1.8×. medium drafts; default high.'),
1438
+ backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
1439
+ aspectRatio: zEnum(IMAGE_ASPECT_RATIOS).optional().describe(`Pick from the use case: cinematic 16:9 · TikTok/Reels 9:16 · IG square 1:1 · ultra-wide 21:9. 1:1 costs most on GPT Image. Per model: ${describeAspectRatios(IMAGE_MODELS)}`),
1440
+ count: z.number().int().min(1).max(10).optional().describe('Up to 10 with projectId; headless caps at 4.'),
1441
+ referenceImageUrls: z.array(z.string().url()).max(14).optional().describe('Headless (no projectId) nano-banana-2 only. With a projectId, upload via slates_upload_reference_image. Label every image role in the prompt.'),
1442
+ referenceAssetIds: z.array(z.string()).max(16).optional().describe("Project assets as references — UUIDs or badge codes (\"IMG-A8\"), resolved at call time. Requires projectId. Caps: GPT Image 16, nano-banana-2 14, FLUX/Seedream lower. Label every reference role in the prompt."),
1369
1443
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
1370
1444
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
1371
1445
  }),
@@ -1398,13 +1472,24 @@ export const generateImage = {
1398
1472
  }
1399
1473
  const resolution = input.resolution;
1400
1474
  const imageModel = input.model ?? 'nano-banana-2';
1401
- // The '3k' class exists only on gpt-image-2 (2560×1440); other models
1402
- // would mis-key the registry lookup.
1403
- if (resolution === '3k' && imageModel !== 'gpt-image-2') {
1475
+ // 🚨 SEEDREAM HAS A 3k CLASS TOO, AND THIS GUARD USED TO DENY IT.
1476
+ // It read `!== 'gpt-image-2'` and rejected every other model at 3k — but
1477
+ // `seedream-5-lite` declares ['2k','3k','4k'] in the desktop registry and
1478
+ // has a real `seedream-5-lite` cost key, so the op was refusing a request
1479
+ // the desktop would have served. Pre-existing; found by the 2026-09-09
1480
+ // audit, not introduced by the 2.5 swap.
1481
+ //
1482
+ // The ladder itself is owned by MODEL_REGISTRY in slate/src/shared/pricing.ts
1483
+ // and is not readable from here, so THREE_K_IMAGE_MODELS is a MIRROR — declared
1484
+ // and exported below so `pricing-consistency-check.mjs` compares it against
1485
+ // the desktop registry's own `imageResolutions`. It used to be an inline
1486
+ // literal with a comment admitting nothing checked it, which is how it came
1487
+ // to deny `seedream-5-lite` a class the desktop had always served.
1488
+ if (resolution === '3k' && !THREE_K_IMAGE_MODELS.includes(imageModel)) {
1404
1489
  return ok({
1405
1490
  requires_clarification: true,
1406
1491
  missing: ['resolution'],
1407
- message: `3k (1440p) is a gpt-image-2 resolution class — pick 1k/2k/4k for ${imageModel}.`,
1492
+ message: `3k (1440p) exists on ${THREE_K_IMAGE_MODELS.join(', ')} — pick 1k/2k/4k for ${imageModel}.`,
1408
1493
  });
1409
1494
  }
1410
1495
  // Only nano-banana-2 has a headless path — everything else routes through
@@ -1435,6 +1520,18 @@ export const generateImage = {
1435
1520
  message: 'background=true routes through the desktop generation pipeline (so slates_get_generation_status can poll it) — pass a projectId, or drop background for a blocking headless run.',
1436
1521
  });
1437
1522
  }
1523
+ // 🚨 THE HEADLESS PATH ASKS FAL FOR A BATCH, so a provider ceiling binds
1524
+ // here and nowhere else. It is nano-banana-2 only, and Nano Banana's
1525
+ // `num_images` maximum is 4 (fal schema, 2026-09-09). Refused rather than
1526
+ // clamped: a silent clamp would make four images against a request for ten
1527
+ // and read to the caller as a partial failure it should retry.
1528
+ if (!input.projectId && (input.count ?? 1) > 4) {
1529
+ return ok({
1530
+ requires_clarification: true,
1531
+ missing: ['projectId'],
1532
+ message: 'count above 4 needs a projectId. The headless path asks fal for one batch and nano-banana-2 caps a batch at 4; with a projectId the desktop fires them as separate generations and the limit is 10.',
1533
+ });
1534
+ }
1438
1535
  let refEcho = '';
1439
1536
  if (referenceAssetIds.length > 0) {
1440
1537
  await ctx.desktop().requireCapability('image-references', 'reference images on image generation');
@@ -1451,11 +1548,18 @@ export const generateImage = {
1451
1548
  // New-roster models need a desktop that knows them — an older desktop's
1452
1549
  // allowlist would silently fall back to nano-banana-2 while we quote the
1453
1550
  // new model's price.
1454
- if (input.projectId &&
1455
- (imageModel === 'gpt-image-2' || imageModel === 'nano-banana-pro' || imageModel === 'nano-banana-2-lite')) {
1551
+ if (input.projectId && isGptImageModel(imageModel)) {
1552
+ // 🚨 v3, NOT v2. GPT Image 2.5 landed 2026-09-09 with new ids and a
1553
+ // five-rung ladder; a desktop that only knows v2 has neither, so it would
1554
+ // fall back to nano-banana-2 while this op quotes a 2.5 price — precisely
1555
+ // the failure the gate was built to stop. Reusing v2 reintroduces it.
1556
+ await ctx.desktop().requireCapability('image-models-v3', `${imageModel} generation`);
1557
+ }
1558
+ else if (input.projectId &&
1559
+ (imageModel === 'nano-banana-pro' || imageModel === 'nano-banana-2-lite')) {
1456
1560
  await ctx.desktop().requireCapability('image-models-v2', `${imageModel} generation`);
1457
1561
  }
1458
- const costKey = imageCostKey(imageModel, resolution, input.quality ?? 'medium');
1562
+ const costKey = imageCostKey(imageModel, resolution, input.quality, input.aspectRatio ?? '1:1');
1459
1563
  const cloud = ctx.cloud();
1460
1564
  const registry = await cloud.get('/api/agent/models');
1461
1565
  const entry = registry.models.find((m) => m.model === costKey);
@@ -1524,7 +1628,7 @@ export const generateImage = {
1524
1628
  resolution,
1525
1629
  aspectRatio: input.aspectRatio ?? '1:1',
1526
1630
  count: input.count ?? 1,
1527
- ...(imageModel === 'gpt-image-2' ? { gptQuality: input.quality ?? 'medium' } : {}),
1631
+ ...(isGptImageModel(imageModel) ? { gptQuality: input.quality, gptBackground: input.backgroundMode } : {}),
1528
1632
  ...(referenceAssetIds.length > 0 ? { referenceAssetIds } : {}),
1529
1633
  background: input.background,
1530
1634
  });
@@ -1617,6 +1721,13 @@ export const generateImage = {
1617
1721
  params: {
1618
1722
  prompt: input.prompt,
1619
1723
  aspect_ratio: input.aspectRatio ?? '1:1',
1724
+ // 🚨 THE HEADLESS PATH IS THE ONE PLACE WE ASK FAL FOR A BATCH, so it
1725
+ // is the one place a provider's own `num_images` ceiling binds — and
1726
+ // Nano Banana's is 4 (fal schema, read 2026-09-09), against the op's
1727
+ // limit of 10. Everything else fans out through the desktop as N
1728
+ // separate single-image generations, where no batch ceiling exists.
1729
+ // Guarded above rather than clamped here: silently making 4 when 10
1730
+ // were asked for would bill 4 and look like a partial failure.
1620
1731
  num_images: input.count ?? 1,
1621
1732
  ...(hasReferenceImages
1622
1733
  ? { image_urls: input.referenceImageUrls }
@@ -1687,15 +1798,16 @@ async function pollProxyJob(cloud, jobId, options = {}) {
1687
1798
  export const editImage = {
1688
1799
  id: 'slates_edit_image',
1689
1800
  billable: true,
1690
- description: 'Surgically edit an existing image asset with a text instruction (e.g. \'remove the lamppost\', \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The edited result is saved as a NEW asset in the project (prompt prefixed \'[Edit]\'); the source is untouched. Default model nano-banana-2 (only model that also accepts referenceAssetIds); flux-2-max / seedream-5-lite use their own edit endpoints and ignore references. Before first use call slates_get_prompting_guide with topic \'slates-edit-and-iterate\'.',
1801
+ description: 'Surgically edit an image asset with a text instruction (e.g. \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The result is a NEW asset (prompt prefixed \'[Edit]\'); the source is untouched. Default model nano-banana-2 (only model that also accepts referenceAssetIds); flux-2-max / seedream-5-lite use their own edit endpoints and ignore references. Before first use call slates_get_prompting_guide with topic \'slates-edit-and-iterate\'.',
1691
1802
  input: z.object({
1692
1803
  projectId: z.string().uuid(),
1693
- sourceAssetId: z.string().uuid().describe('Image asset to edit. Must already exist in the project.'),
1694
- prompt: z.string().min(1).max(4000).describe('The edit instruction — describe the change, not the whole image.'),
1695
- editModel: z.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2', 'flux-2-max', 'seedream-5-lite']).optional(),
1696
- referenceAssetIds: z.array(z.string().uuid()).max(13).optional().describe('Nano-Banana family only: extra reference images (NB Pro takes up to 13, NB2 Lite up to 3).'),
1697
- resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('3k (1440p class) is gpt-image-2 only; nano-banana-2-lite is 1k only.'),
1698
- quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only — quality tier (default medium).'),
1804
+ sourceAssetId: z.string().uuid().describe('Image asset to edit. Must exist in the project.'),
1805
+ prompt: z.string().min(1).max(4000).describe('The change, not the whole image.'),
1806
+ editModel: z.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2-5-flare', 'gpt-image-2-5-sunburst', 'flux-2-max', 'seedream-5-lite']).optional(),
1807
+ referenceAssetIds: z.array(z.string().uuid()).max(13).optional().describe('Nano-Banana only (NB Pro 13, NB2 Lite 3).'),
1808
+ resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('3k = GPT Image/seedream-5-lite; nano-banana-2-lite is 1k only.'),
1809
+ quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
1810
+ backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
1699
1811
  aspectRatio: z.string().optional(),
1700
1812
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
1701
1813
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
@@ -1708,14 +1820,22 @@ export const editImage = {
1708
1820
  }
1709
1821
  const editModel = input.editModel ?? 'nano-banana-2';
1710
1822
  const resolution = input.resolution ?? (editModel === 'nano-banana-2-lite' ? '1k' : '2k');
1711
- if ((editModel === 'gpt-image-2' || editModel === 'nano-banana-pro' || editModel === 'nano-banana-2-lite')) {
1823
+ if (isGptImageModel(editModel)) {
1824
+ // 🚨 v3, LIKE generateImage — this site was missed once already.
1825
+ // A pre-2.5 desktop advertises v2, so gating the 2.5 seats on v2 lets it
1826
+ // through; `isFalImageModel` is then false for these ids on that build and
1827
+ // handleEditImage falls through to the nano-banana path — NB2 output billed
1828
+ // at a quoted 2.5 price. Exactly the bug the gate exists to stop.
1829
+ await desktop.requireCapability('image-models-v3', `${editModel} editing`);
1830
+ }
1831
+ else if (editModel === 'nano-banana-pro' || editModel === 'nano-banana-2-lite') {
1712
1832
  await desktop.requireCapability('image-models-v2', `${editModel} editing`);
1713
1833
  }
1714
- // Nano-Banana family + GPT Image 2 edits charge the same key as gen;
1834
+ // Nano-Banana family + GPT Image 2.5 edits charge the same key as gen;
1715
1835
  // FLUX / Seedream route to dedicated edit endpoints priced under '-edit' keys.
1716
1836
  const costKey = editModel === 'flux-2-max' || editModel === 'seedream-5-lite'
1717
1837
  ? `${imageCostKey(editModel, resolution)}-edit`
1718
- : imageCostKey(editModel, resolution, input.quality ?? 'medium');
1838
+ : imageCostKey(editModel, resolution, input.quality, input.aspectRatio);
1719
1839
  const cloud = ctx.cloud();
1720
1840
  const registry = await cloud.get('/api/agent/models');
1721
1841
  const entry = registry.models.find((m) => m.model === costKey);
@@ -1742,7 +1862,7 @@ export const editImage = {
1742
1862
  editModel,
1743
1863
  referenceAssetIds: input.referenceAssetIds,
1744
1864
  resolution,
1745
- ...(editModel === 'gpt-image-2' ? { gptQuality: input.quality ?? 'medium' } : {}),
1865
+ ...(isGptImageModel(editModel) ? { gptQuality: input.quality, gptBackground: input.backgroundMode } : {}),
1746
1866
  aspectRatio: input.aspectRatio,
1747
1867
  background: input.background,
1748
1868
  });
@@ -2727,14 +2847,68 @@ export const generateVideo = {
2727
2847
  };
2728
2848
  },
2729
2849
  };
2850
+ // ── The preset voice shelf ──────────────────────────────────────
2851
+ /**
2852
+ * AGENT PARITY for the voice picker. The desktop browses stock voices by
2853
+ * gender, accent and age and plays each one; until this op the agent could
2854
+ * only pass a `voiceId` it had no way to discover. Disk reads on the desktop,
2855
+ * never a vendor call — browsing is free on every surface.
2856
+ */
2857
+ export const listVoices = {
2858
+ id: 'slates_list_voices',
2859
+ description: `Browse ${TTS_MODEL} preset voices. Pass a returned voiceId to slates_generate_audio. Filters AND together.`,
2860
+ input: z.object({
2861
+ gender: z.string().optional().describe('male | female'),
2862
+ accent: z.string().optional().describe('Region ("GB") or languageCode ("en-GB"); available accents come from the shelf.'),
2863
+ age: z.string().optional().describe('young | middle_aged | elderly'),
2864
+ query: z.string().optional().describe('Free text over name, description, tags.'),
2865
+ }),
2866
+ async run(input, ctx) {
2867
+ const desktop = ctx.desktop();
2868
+ await desktop.requireCapability('voices', 'the preset voice shelf');
2869
+ const shelf = await desktop.get('/agent/voices');
2870
+ if (!shelf.available) {
2871
+ return ok({ available: false, voices: [], message: 'This desktop build shipped without the preset shelf.' });
2872
+ }
2873
+ const terms = (input.query ?? '').toLowerCase().split(/\s+/).filter(Boolean);
2874
+ const region = input.accent?.includes('-') ? input.accent.split('-')[1] : input.accent;
2875
+ const voices = shelf.voices
2876
+ .filter((v) => !input.gender || v.gender === input.gender)
2877
+ .filter((v) => !region || v.languageCode.split('-')[1]?.toUpperCase() === region.toUpperCase())
2878
+ .filter((v) => !input.age || v.ageGroup === input.age)
2879
+ .filter((v) => {
2880
+ if (terms.length === 0)
2881
+ return true;
2882
+ const hay = [v.displayName, v.description, v.gender, v.ageGroup, v.languageCode, ...(v.tags ?? [])]
2883
+ .join(' ')
2884
+ .toLowerCase();
2885
+ return terms.every((t) => hay.includes(t));
2886
+ })
2887
+ .map(({ voiceId, displayName, description, tags, gender, ageGroup, languageCode }) => ({
2888
+ voiceId,
2889
+ displayName,
2890
+ description,
2891
+ tags,
2892
+ gender,
2893
+ ageGroup,
2894
+ languageCode,
2895
+ }));
2896
+ return ok({
2897
+ available: true,
2898
+ line: shelf.line,
2899
+ count: voices.length,
2900
+ voices,
2901
+ next: `Pass a voiceId to slates_generate_audio (model ${TTS_MODEL}) as voiceId. To keep one on a character for reuse, generate a clip with it and set slates_update_character voiceAssetId.`,
2902
+ });
2903
+ },
2904
+ };
2730
2905
  // ── Generate audio ──────────────────────────────────────────────
2731
2906
  export const generateAudio = {
2732
2907
  id: 'slates_generate_audio',
2733
2908
  billable: true,
2734
- description: 'Generate AUDIO via Slates credits — the third media type, saved as a project asset you can drop on an audio track. Three surfaces: seed-audio (default; a whole audio SCENE — dialogue + SFX + ambience — from one plain sentence, 3-120s), eleven-sfx (ONE effect with an exact 1-22s duration, or a seamless loop), and inworld-tts-2 (one named voice saying one line; the prompt IS the words, billed per character). Which surface for which job: read the slates-model-selection skill. ' +
2735
- '🚨 seed-audio has NO duration parameter — the length you pass is written INTO THE PROMPT and is what the user is BILLED, whatever comes back. Choose it deliberately. ' +
2736
- 'REQUIRED before calling: read slates-cost-discipline and the matching prompting skill (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). Kling\'s "SFX:" / "Ambient noise:" prompt syntax does NOT transfer to seed-audio and makes results worse. ' +
2737
- 'projectId is REQUIRED (no headless path). ' +
2909
+ description: `Generate project audio using credits. Choose the surface via the model routing below. ` +
2910
+ 'Read slates-cost-discipline and the matching prompting skill first (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). ' +
2911
+ 'Seed Audio bills the requested duration, which is appended to the prompt regardless of output length. Kling "SFX:" / "Ambient noise:" syntax does not transfer. ' +
2738
2912
  CONFIRM_GATE_SENTENCE +
2739
2913
  ' No skill files installed? Call slates_get_prompting_guide first.',
2740
2914
  input: z.object({
@@ -2753,25 +2927,27 @@ export const generateAudio = {
2753
2927
  durationSeconds: z
2754
2928
  .number()
2755
2929
  .optional()
2756
- .describe('seed-audio 3-120 (default 15) — ⚠️ THIS IS THE BILL: it is appended to the prompt and charged regardless of the returned length. eleven-sfx 1-22 (default 4) — always sent explicitly so the per-second charge is deterministic.'),
2930
+ .describe(`seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (default ${SEED_AUDIO_DEFAULT_SECONDS}) — ⚠️ THIS IS THE BILL: appended to the prompt and charged whatever comes back. eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS} (default ${ELEVEN_SFX_DEFAULT_SECONDS}) — always sent explicitly so the per-second charge is deterministic. Not for ${TTS_MODEL}.`),
2757
2931
  voice: z
2758
2932
  .string()
2759
2933
  .optional()
2760
- .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself, which is usually right for background dialogue. Agent-facing only: there is no user-facing voice picker.'),
2934
+ .describe('seed-audio only — a preset voice id (e.g. "cedric_en_zh"). Leave unset to let the scene cast itself. Agent-facing only.'),
2761
2935
  voiceId: z
2762
2936
  .string()
2763
2937
  .optional()
2764
- .describe('inworld-tts-2 — the voice to speak in. One of these three is required there.'),
2938
+ .describe(`${TTS_MODEL} — a preset voiceId from slates_list_voices, not a character or asset id. Exactly one voice source is required.`),
2765
2939
  voiceReferenceAssetId: z
2766
2940
  .string()
2767
2941
  .optional()
2768
- .describe('inworld-tts-2 — clone the voice from this AUDIO asset (5-15s of one clean speaker).'),
2942
+ .describe(`${TTS_MODEL} — clone this AUDIO asset's voice for the take (${TTS_VOICE_CLONE.minSeconds}-${TTS_VOICE_CLONE.maxSeconds}s, one clean speaker). To speak AS a character pass its voiceAssetId. Cloning: ${TTS_VOICE_CLONE.clonesPerMinute} new voices/min across all of Slates; a burst waits.`),
2769
2943
  voiceDescription: z
2770
2944
  .string()
2945
+ .min(TTS_VOICE_CLONE.designPromptChars.min)
2946
+ .max(TTS_VOICE_CLONE.designPromptChars.max)
2771
2947
  .optional()
2772
- .describe('inworld-tts-2 — build a voice from this description, for a character with no recording.'),
2948
+ .describe(`${TTS_MODEL} — a voice from words (${TTS_VOICE_CLONE.designPromptChars.min}-${TTS_VOICE_CLONE.designPromptChars.max} chars) for a character with no recording; keep it via slates_update_character voiceAssetId.`),
2773
2949
  speed: z.number().min(0.5).max(2).optional().describe('seed-audio only — 0.5-2.0. Reach for it when dialogue races or drags against picture.'),
2774
- volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline track fader for mix decisions; this is for when the model itself renders a scene too hot or too quiet.'),
2950
+ volume: z.number().min(0.5).max(2).optional().describe('seed-audio only — output gain, 0.5-2.0 (1 = unchanged). Prefer the timeline fader for mix decisions.'),
2775
2951
  pitch: z.number().int().min(-12).max(12).optional().describe('seed-audio only — semitones. Small moves; ±3 is already a lot.'),
2776
2952
  multilingual: z.boolean().optional().describe('seed-audio only — better non-English / mixed-language handling.'),
2777
2953
  loop: z.boolean().optional().describe('eleven-sfx only — produce a seamless loop (rain, engine hum, crowd murmur).'),
@@ -2780,7 +2956,7 @@ export const generateAudio = {
2780
2956
  .array(z.string())
2781
2957
  .max(3)
2782
2958
  .optional()
2783
- .describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId — the API rejects both.'),
2959
+ .describe('seed-audio only — up to 3 AUDIO assets (UUIDs or badge codes like "AUD-S1"), each ≤30s, referenced in the prompt as @Audio1-@Audio3 ("match the room tone of @Audio1"). MUTUALLY EXCLUSIVE with imageReferenceAssetId.'),
2784
2960
  imageReferenceAssetId: z
2785
2961
  .string()
2786
2962
  .optional()
@@ -2823,7 +2999,7 @@ export const generateAudio = {
2823
2999
  return ok({
2824
3000
  requires_clarification: true,
2825
3001
  missing: ['voiceId'],
2826
- message: `${TTS_MODEL} needs a voice. Ask the user WHICH CHARACTER is speaking and pass that character's voice as voiceId — a voice is a field on a character, not a thing to pick at generation time. To make a NEW voice, pass voiceReferenceAssetId (a clip to clone) or voiceDescription (words, for a character with no recording).`,
3002
+ message: `${TTS_MODEL} needs a voice — exactly one of three. Speaking AS a character: pass its voiceAssetId (slates_list_characters) as voiceReferenceAssetId. A stock voice: slates_list_voices lists presets by gender, accent and age; pass one's voiceId. No recording of the voice: voiceDescription (words), or voiceReferenceAssetId with any clean clip of one speaker. A voice worth reusing can be kept on a character with slates_update_character, but nothing requires that — ask the user which they want only when the request does not say.`,
2827
3003
  });
2828
3004
  }
2829
3005
  if (voiceSources.length > 1) {
@@ -3725,17 +3901,33 @@ export const setFolderCover = {
3725
3901
  };
3726
3902
  export const updateCharacter = {
3727
3903
  id: 'slates_update_character',
3728
- description: 'Update a character\'s name, description, or style. Use slates_set_character_identity_asset for its canonical image.',
3904
+ description: 'Update a character\'s name, description, style, or voice. Use slates_set_character_identity_asset for its canonical image.',
3729
3905
  input: z.object({
3730
3906
  characterId: z.string().uuid(),
3731
3907
  name: z.string().min(1).max(120).optional(),
3732
3908
  description: z.string().optional(),
3733
3909
  style: z.string().max(200).optional().describe("Art style. Omit to inherit the reference's style (the default). Canonical styles: photoreal, anime, painterly, 3d-render, comic. Or pass any free-text instruction, e.g. 'turn this into a real person'."),
3910
+ // Agent parity for the character card's voice slot: the desktop route has
3911
+ // taken this since 2026-08-28; the op never exposed it, so an agent could
3912
+ // render a voice and had no way to keep it on the character.
3913
+ voiceAssetId: z
3914
+ .string()
3915
+ .uuid()
3916
+ .nullable()
3917
+ .optional()
3918
+ .describe("The AUDIO asset that is this character's voice (what inworld-tts-2 clones for its lines); null detaches, the clip stays."),
3734
3919
  }),
3735
3920
  async run(input, ctx) {
3736
3921
  return ok(await ctx.desktop().post('/agent/characters/update', {
3737
3922
  id: input.characterId,
3738
- data: { name: input.name, description: input.description, style: input.style },
3923
+ data: {
3924
+ name: input.name,
3925
+ description: input.description,
3926
+ style: input.style,
3927
+ // Sent only when given: the route treats presence as intent, and an
3928
+ // explicit null is the detach.
3929
+ ...(input.voiceAssetId !== undefined ? { voiceAssetId: input.voiceAssetId } : {}),
3930
+ },
3739
3931
  }));
3740
3932
  },
3741
3933
  };
@@ -4021,16 +4213,22 @@ const SHOT_ASPECT_RATIOS = [...new Set([...VIDEO_ASPECT_RATIOS, ...IMAGE_ASPECT_
4021
4213
  function shotParamsShape(described) {
4022
4214
  const d = (node, text) => (described ? node.describe(text) : node);
4023
4215
  return {
4024
- aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the chosen model when the Shot is saved — see slates_generate_video for the per-model sets.'),
4025
- duration: d(z.number().int().min(1).max(360).optional(), 'Seconds — video or audio. Required before a video Shot can be priced or fired; validated against the model when saved.'),
4026
- videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the chosen model when the Shot is saved.'),
4216
+ aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the model; see slates_generate_video.'),
4217
+ duration: d(z.number().int().min(1).max(360).optional(), 'Seconds for video or duration-based audio; TTS uses text length.'),
4218
+ videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the model when the Shot is saved.'),
4027
4219
  imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
4028
- gptQuality: d(z.enum(['medium', 'high']).optional(), 'gpt-image-2 only.'),
4029
- imageQuantity: d(z.number().int().min(1).max(4).optional(), 'Image models only — how many to make per fire.'),
4220
+ gptQuality: d(z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional(), 'GPT Image 2.5 only.'),
4221
+ gptBackground: d(z.enum(['auto', 'transparent', 'opaque']).optional(), 'GPT Image only.'),
4222
+ imageQuantity: d(z.number().int().min(1).max(10).optional(), 'Image models only.'),
4030
4223
  negativePrompt: z.string().optional(),
4031
4224
  sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
4032
- seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider at ~45% more."),
4225
+ seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider, ~45% more."),
4033
4226
  audioDurationSeconds: d(z.number().int().min(1).max(120).optional(), 'Audio lane. On seed-audio the requested duration IS the bill.'),
4227
+ // The TTS voice — the same three fields slates_generate_audio takes, so a
4228
+ // Shot is the audio call, serialized. Exactly one of them, enforced at fire.
4229
+ voiceId: d(z.string().optional(), `${TTS_MODEL}: preset voiceId (slates_list_voices).`),
4230
+ voiceReferenceAssetId: d(z.string().optional(), `${TTS_MODEL}: audio asset to clone.`),
4231
+ voiceDescription: d(z.string().optional(), `${TTS_MODEL}: the voice in words.`),
4034
4232
  };
4035
4233
  }
4036
4234
  const shotParamsSchema = z.object(shotParamsShape(true)).optional();
@@ -4047,10 +4245,7 @@ const shotParamsSchemaTerse = z
4047
4245
  * ship a field with no explanation — the same failure as a column nothing
4048
4246
  * renders.
4049
4247
  *
4050
- * 🚨 AND NONE OF THEM IS SENT TO A MODEL. They are a planning and counting
4051
- * surface; the prompt is the only thing the request carries. The one exception
4052
- * is pre-existing: a multiShotSegment still prepends its own camera and
4053
- * shotSize to its own segment prompt.
4248
+ * Script fields supply prompt prose when no authored prompt exists (shot-spec.ts).
4054
4249
  */
4055
4250
  function shotScriptShape(described) {
4056
4251
  const text = Object.fromEntries(SCRIPT_TEXT_FIELDS.map((field) => [
@@ -4151,6 +4346,8 @@ function shotRefInputs(input) {
4151
4346
  out.push({ ref: input.firstFrameAssetId, role: 'first frame' });
4152
4347
  if (input.lastFrameAssetId)
4153
4348
  out.push({ ref: input.lastFrameAssetId, role: 'last frame' });
4349
+ if (input.params?.voiceReferenceAssetId)
4350
+ out.push({ ref: input.params.voiceReferenceAssetId, role: 'voice reference' });
4154
4351
  return out;
4155
4352
  }
4156
4353
  /**
@@ -4187,7 +4384,10 @@ async function buildShotSpecInput(ctx, projectId, input) {
4187
4384
  // later diverges from it and the desktop card says so — the prompt is
4188
4385
  // never rewritten (that is prompt enhancement, deleted 2026-08-01).
4189
4386
  authoredFor: input.model ?? null,
4190
- params: shotParamsPatch(input.params),
4387
+ params: {
4388
+ ...shotParamsPatch(input.params),
4389
+ ...(input.params?.voiceReferenceAssetId ? { voiceReferenceAssetId: rid(input.params.voiceReferenceAssetId) } : {}),
4390
+ },
4191
4391
  mentions: {
4192
4392
  characterIds: input.characterIds ?? [],
4193
4393
  environmentIds: input.environmentIds ?? [],
@@ -4240,12 +4440,12 @@ function shotCostKey(detail) {
4240
4440
  // has none, so it falls back to the raw ones and is announced as a floor.
4241
4441
  const fires = detail.firesWith;
4242
4442
  if (AUDIO_MODELS.includes(model)) {
4243
- // The TTS seat prices on the TEXT, and a Shot carries no text field — so a
4244
- // Shot cannot be a TTS generation and cannot be quoted as one. Explicit,
4245
- // because the `!seconds` line below would also return null here and that
4246
- // would read as "duration missing" for a surface that has no duration.
4247
- if (model === TTS_MODEL)
4248
- return null;
4443
+ if (model === TTS_MODEL) {
4444
+ const text = detail.composedPrompt ?? (detail.rawPrompt.trim() || detail.line?.trim() || '');
4445
+ if (!text || text.length > TTS_MAX_CHARACTERS)
4446
+ return null;
4447
+ return audioCostKey({ model, characters: text.length });
4448
+ }
4249
4449
  const seconds = fires?.audioDurationSeconds ?? p.audioDurationSeconds;
4250
4450
  if (!seconds)
4251
4451
  return null;
@@ -4275,7 +4475,15 @@ function shotCostKey(detail) {
4275
4475
  }
4276
4476
  if (IMAGE_MODELS.includes(model)) {
4277
4477
  return imageCostKey(model, (fires?.imageResolution ?? p.imageResolution) ??
4278
- (model === 'nano-banana-2-lite' ? '1k' : '2k'), p.gptQuality ?? 'medium');
4478
+ (model === 'nano-banana-2-lite' ? '1k' : '2k'),
4479
+ // No inline default — see the note at the estimate op. Firing a Shot with
4480
+ // no stored tier lands on `high` via the desktop's `normalizeGptQuality`,
4481
+ // and `imageCostKey`'s parameter default is pinned to match it.
4482
+ p.gptQuality,
4483
+ // The aspect the Shot will fire at — on GPT Image it moves the key, so
4484
+ // quoting without it under-prices every square Shot. `fires` carries
4485
+ // only the clamped resolution/duration axes, never the aspect.
4486
+ p.aspectRatio);
4279
4487
  }
4280
4488
  return null;
4281
4489
  }
@@ -4453,8 +4661,17 @@ export const duplicateShot = {
4453
4661
  spec.prompt = input.prompt;
4454
4662
  if (input.model !== undefined)
4455
4663
  spec.model = input.model;
4456
- if (input.params !== undefined)
4457
- spec.params = shotParamsPatch(input.params);
4664
+ if (input.params !== undefined) {
4665
+ const voiceRef = input.params.voiceReferenceAssetId;
4666
+ if (voiceRef && !UUID_RE.test(voiceRef)) {
4667
+ const { shot } = await desktop.get('/agent/shots/get', { id: input.shotId });
4668
+ const built = await buildShotSpecInput(ctx, shot.projectId, { params: input.params });
4669
+ spec.params = built.spec.params;
4670
+ }
4671
+ else {
4672
+ spec.params = shotParamsPatch(input.params);
4673
+ }
4674
+ }
4458
4675
  const r = await desktop.post('/agent/shots/duplicate', {
4459
4676
  id: input.shotId,
4460
4677
  name: input.name,
@@ -4715,7 +4932,7 @@ export const generateFromShots = {
4715
4932
  // something to try again spends credits before anyone notices.
4716
4933
  `\n${failedLines.join('\n')}\nThese were NOT retried. Read each error, fix the Shot, and re-fire only what you meant to.`
4717
4934
  : '') +
4718
- ` ${VIDEO_REVIEW_POINTER}`);
4935
+ ` ${BACKGROUND_REVIEW_POINTER}`);
4719
4936
  },
4720
4937
  };
4721
4938
  function resolveGuideTopic(topic) {
@@ -4763,7 +4980,7 @@ function resolveGuideTopic(topic) {
4763
4980
  if (t.startsWith('nano-banana'))
4764
4981
  return 'slates-prompting-nano-banana-2';
4765
4982
  if (t.startsWith('gpt-image') || t.startsWith('gpt image'))
4766
- return 'slates-prompting-gpt-image-2';
4983
+ return 'slates-prompting-gpt-image-2-5';
4767
4984
  if (t.startsWith('flux'))
4768
4985
  return 'slates-prompting-flux-2-max';
4769
4986
  if (t.startsWith('seedream'))
@@ -4855,7 +5072,7 @@ function resolveGuideTopic(topic) {
4855
5072
  * The guide index, GENERATED from SKILLS.
4856
5073
  *
4857
5074
  * The list here was hand-typed and had drifted to 25 of 32 names — the prompting
4858
- * guides for GPT Image 2, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
5075
+ * guides for GPT Image, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
4859
5076
  * all missing, so an agent reading this description could not learn they exist.
4860
5077
  * A hand-typed index of a generated corpus is a stale index; it is only a matter
4861
5078
  * of when.
@@ -4874,15 +5091,16 @@ function describeGuideTopics() {
4874
5091
  }
4875
5092
  export const getPromptingGuide = {
4876
5093
  id: 'slates_get_prompting_guide',
4877
- description:
4878
- // 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
4879
- // MEASURED at 13% compliance before and after the enforcement work — pointer
4880
- // prose is the shape that does not move the agent. What replaced it is
4881
- // structural: the never-use list rides the generate ops' descriptions and
4882
- // the craft card rides the estimate result, so the facts arrive whether or
4883
- // not this op is ever called.
4884
- "Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
5094
+ description: 'For app help and exact UI instructions use topic "app-manual" with a query such as "voice recording". This returns the canonical product manual, shared by every agent surface. ' +
5095
+ // 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
5096
+ // MEASURED at 13% compliance before and after the enforcement work — pointer
5097
+ // prose is the shape that does not move the agent. What replaced it is
5098
+ // structural: the never-use list rides the generate ops' descriptions and
5099
+ // the craft card rides the estimate result, so the facts arrive whether or
5100
+ // not this op is ever called.
5101
+ "Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
4885
5102
  input: z.object({
5103
+ query: z.string().max(200).optional().describe('For app-manual: keywords to retrieve relevant UI sections. Omit for the entire manual.'),
4886
5104
  topic: z
4887
5105
  .string()
4888
5106
  .min(1)
@@ -4890,6 +5108,10 @@ export const getPromptingGuide = {
4890
5108
  depth: z.enum(['card', 'full']).optional().describe('"card" returns just the levers block (a few hundred words — the same card slates_estimate_generation_cost already attached, so usually redundant). "full" (default) returns the whole guide, up to several thousand words.'),
4891
5109
  }),
4892
5110
  async run(input) {
5111
+ if (input.topic.trim().toLowerCase() === 'app-manual') {
5112
+ const content = appManualSections(input.query);
5113
+ return { text: content, data: { topic: 'app-manual', bytes: Buffer.byteLength(content, 'utf8') } };
5114
+ }
4893
5115
  const resolved = resolveGuideTopic(input.topic);
4894
5116
  const content = resolved ? SKILLS[resolved] : undefined;
4895
5117
  if (!resolved || content === undefined) {
@@ -5124,6 +5346,7 @@ export const ALL_OPERATIONS = [
5124
5346
  generateImage,
5125
5347
  generateVideo,
5126
5348
  generateAudio,
5349
+ listVoices,
5127
5350
  generateLipSync,
5128
5351
  generateMotionTransfer,
5129
5352
  editVideo,