@slatesvideo/shared 0.7.1 → 0.7.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/dist/clients/cloud.d.ts +4 -0
  2. package/dist/clients/cloud.js +11 -3
  3. package/dist/index.d.ts +2 -1
  4. package/dist/index.js +2 -1
  5. package/dist/manual/content.d.ts +1 -1
  6. package/dist/manual/content.js +1 -1
  7. package/dist/manual/index.d.ts +11 -2
  8. package/dist/manual/index.js +178 -14
  9. package/dist/operations/index.d.ts +292 -94
  10. package/dist/operations/index.js +870 -199
  11. package/dist/operations/surface.d.ts +6 -2
  12. package/dist/operations/surface.js +35 -5
  13. package/dist/prompts/agent-doctrine.d.ts +4 -4
  14. package/dist/prompts/agent-doctrine.js +18 -29
  15. package/dist/prompts/generation-policy.d.ts +1 -1
  16. package/dist/prompts/guide-discovery.d.ts +23 -0
  17. package/dist/prompts/guide-discovery.js +39 -0
  18. package/dist/prompts/guide-retrieval.js +1 -1
  19. package/dist/prompts/model-capabilities.d.ts +8 -9
  20. package/dist/prompts/model-capabilities.js +11 -51
  21. package/dist/prompts/model-facts.d.ts +2 -2
  22. package/dist/prompts/model-facts.js +15 -26
  23. package/dist/prompts/partials.generated.js +6 -3
  24. package/dist/prompts/prompting-tips.d.ts +1 -1
  25. package/dist/prompts/prompting-tips.js +21 -63
  26. package/dist/prompts/search-terms.d.ts +3 -0
  27. package/dist/prompts/search-terms.js +24 -0
  28. package/dist/skills/content.js +36 -37
  29. package/dist/skills/metadata.d.ts +7 -0
  30. package/dist/skills/metadata.js +29 -0
  31. package/exports/slates-chatgpt-images/generated/SKILL.md +7 -1
  32. package/exports/slates-chatgpt-images/generated/slates-chatgpt-images.skill +0 -0
  33. package/exports/slates-prompt-builder/generated/SKILL.md +28 -16
  34. package/exports/slates-prompt-builder/generated/reference-character.md +12 -13
  35. package/exports/slates-prompt-builder/generated/reference-content-policy.md +2 -2
  36. package/exports/slates-prompt-builder/generated/reference-gpt-image-2-5.md +191 -0
  37. package/exports/slates-prompt-builder/generated/reference-kling.md +32 -11
  38. package/exports/slates-prompt-builder/generated/reference-nano-banana.md +24 -6
  39. package/exports/slates-prompt-builder/generated/reference-omni-flash.md +65 -0
  40. package/exports/slates-prompt-builder/generated/reference-seedance-2-5.md +362 -0
  41. package/exports/slates-prompt-builder/generated/reference-seedance.md +34 -4
  42. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +77 -23
  43. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  44. package/package.json +2 -1
  45. package/skills/_partials/blender-action-curves.md +24 -0
  46. package/skills/_partials/cinematic-card.md +1 -1
  47. package/skills/_partials/iteration-diagnosis.md +5 -0
  48. package/skills/_partials/model-routing.md +35 -0
  49. package/skills/_partials/seedance-25-timestamps.md +2 -2
  50. package/skills/_partials/still-gate.md +2 -2
  51. package/skills/_partials/thresholds.md +1 -1
  52. package/skills/slates-blocking-to-prompt.md +15 -13
  53. package/skills/slates-camera-language.md +45 -7
  54. package/skills/slates-character-identity.md +8 -6
  55. package/skills/slates-chatgpt-images.md +7 -1
  56. package/skills/slates-cinematic-look.md +1 -1
  57. package/skills/slates-content-policy.md +4 -6
  58. package/skills/slates-cost-discipline.md +18 -12
  59. package/skills/slates-dialogue-blocking.md +6 -6
  60. package/skills/slates-direct-response-ad.md +1 -1
  61. package/skills/slates-edit-and-iterate.md +12 -4
  62. package/skills/slates-model-selection.md +82 -90
  63. package/skills/slates-one-prompt-film.md +1 -1
  64. package/skills/slates-previs-blocking.md +44 -13
  65. package/skills/slates-project-organization.md +2 -2
  66. package/skills/slates-prompting-elevenlabs.md +4 -4
  67. package/skills/slates-prompting-flux-2-max.md +2 -3
  68. package/skills/slates-prompting-gpt-image-2-5.md +2 -2
  69. package/skills/slates-prompting-inworld-tts.md +1 -1
  70. package/skills/slates-prompting-kling-v3.md +11 -9
  71. package/skills/slates-prompting-lip-sync.md +15 -15
  72. package/skills/slates-prompting-ltx-2-5.md +5 -6
  73. package/skills/slates-prompting-minimax-h3.md +11 -11
  74. package/skills/slates-prompting-motion-transfer.md +8 -8
  75. package/skills/slates-prompting-nano-banana-2.md +8 -4
  76. package/skills/slates-prompting-omni-flash.md +9 -9
  77. package/skills/slates-prompting-seed-audio.md +24 -4
  78. package/skills/slates-prompting-seedance-2-5.md +40 -30
  79. package/skills/slates-prompting-seedance.md +4 -4
  80. package/skills/slates-prompting-seedream-5-lite.md +6 -6
  81. package/skills/slates-restyle-from-blocking.md +2 -2
  82. package/skills/slates-script-craft.md +1 -1
  83. package/skills/slates-shot-variety.md +1 -1
  84. package/skills/slates-storyboard-from-script.md +1 -1
  85. package/skills/slates-style-prompting.md +8 -6
  86. package/skills/slates-ugc-influencer-ad.md +1 -1
  87. package/skills/slates-vision-feedback-loop.md +118 -110
  88. package/skills/slates-prompting-veo-3.md +0 -224
@@ -1,5 +1,6 @@
1
1
  import { MAX_IMAGE_VARIATIONS } from '../prompts/generation-policy.js';
2
- import { retrieveGuide, guideSections } from '../prompts/guide-retrieval.js';
2
+ import { discoverGuides, guideCatalog } from '../prompts/guide-discovery.js';
3
+ import { retrieveGuide } from '../prompts/guide-retrieval.js';
3
4
  import { MINIMAX_MAX_REFERENCE, minimaxMaxReferenceTokens, MODEL_CAPABILITIES, GPT_QUALITY_TIERS, DEFAULT_GPT_QUALITY, GPT_BACKGROUNDS } from '../prompts/model-capabilities.js';
4
5
  // Operations layer — the ONE place every Slates agent tool is defined.
5
6
  // Both the MCP server and the CLI register these as their tool / command
@@ -13,11 +14,11 @@ import { MINIMAX_MAX_REFERENCE, minimaxMaxReferenceTokens, MODEL_CAPABILITIES, G
13
14
  // - chooses its transport (cloud vs desktop) internally — callers
14
15
  // don't need to know which side a given op talks to
15
16
  import { z } from 'zod';
16
- import { SlatesCloudClient } from '../clients/cloud.js';
17
+ import { SlatesCloudClient, SlatesCloudHttpError } from '../clients/cloud.js';
17
18
  import { SlatesDesktopClient } from '../clients/desktop.js';
18
19
  import { BlenderBridgeClient, BLENDER_SETUP_HINT, RENDER_TIMEOUT_MS } from '../clients/blender.js';
19
20
  import { SKILLS } from '../skills/content.js';
20
- import { appManualSections } from '../manual/index.js';
21
+ import { appManualIndex, appManualSections } from '../manual/index.js';
21
22
  // Reference-capacity prose is DERIVED, never hand-typed — root CLAUDE.md:
22
23
  // "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
23
24
  import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords,
@@ -59,10 +60,12 @@ import { SHOT_SIZE_BUCKETS, CAMERA_MOVE_BUCKETS, SPEECH_RATE, } from '../prompts
59
60
  // Annotations, tiers and the ONE schema renderer. Declared next door so this
60
61
  // module never hand-sets a hint or a tier per op: `annotate()` derives all four
61
62
  // from the id and the lockstep check re-derives them from the transport verbs.
62
- import { annotate, groupFor, tierFor, toolDefinitions, GROUP_SUMMARY, OPERATION_GROUPS, } from './surface.js';
63
+ import { annotate, groupFor, tierFor, toolDefinitions, searchTools, GROUP_SUMMARY, OPERATION_GROUPS, } from './surface.js';
64
+ // One factory for every default context, so per-connection caches can key on it.
65
+ const defaultCloud = () => new SlatesCloudClient();
63
66
  export function defaultContext() {
64
67
  return {
65
- cloud: () => new SlatesCloudClient(),
68
+ cloud: defaultCloud,
66
69
  desktop: () => new SlatesDesktopClient(),
67
70
  };
68
71
  }
@@ -127,7 +130,10 @@ export const DEVIATION_FACTOR = 1.2;
127
130
  * to find three wordings. Byte-stable (a template over a literal), so the
128
131
  * desktop's prompt-cached prefix is unaffected.
129
132
  */
130
- const CONFIRM_GATE_SENTENCE = `Cost above ${CONFIRM_CREDITS} credits returns requires_confirm — pass confirm=true after explicit user OK.`;
133
+ // The consent half rides every generation tool because a host may drop the
134
+ // server instructions: Codex CLI 0.159.1 passed none to the model (probes
135
+ // 2026-09-30 and 2026-10-02), so a small spend had no approval rule in view.
136
+ const CONFIRM_GATE_SENTENCE = `Show the user the estimate and wait for their OK before any generation, however small. Cost above ${CONFIRM_CREDITS} credits (and, on image and video, any attached reference) also returns requires_confirm — pass confirm=true only to relay that OK.`;
131
137
  // Declared HERE, above every op, because `slates_estimate_generation_cost`
132
138
  // renders them into its `duration` description at MODULE LOAD — a const
133
139
  // declared below the first schema that reads it is a temporal-dead-zone
@@ -216,6 +222,14 @@ function creditsFromDollars(dollars) {
216
222
  // sentence: it is repeated verbatim on seven ops, so every word costs seven
217
223
  // times, and `slates_get_generation_status` explains the polling itself.
218
224
  const BACKGROUND_DESCRIBE = 'Return generationId(s) now instead of blocking; poll slates_get_generation_status. Recommended for video.';
225
+ // `folderId` on the generate and upload ops (1.6.1, decision 17 in second-brain
226
+ // plans/2026-09-30-slates-manual-for-llms-1-6-1-decisions.md). The folder dot in Media means "new pictures go
227
+ // here", the user's own generations and drops follow it, and so does an agent's: the DESKTOP resolves an absent
228
+ // folderId to the folder chosen in the window, so every op and the Studio Agent get it at once.
229
+ const FOLDER_DESCRIBE = "Folder the result lands in (slates_list_folders). Omit it to follow the folder chosen in the user's Slates window, else the project root; null is the root, on purpose.";
230
+ const folderIdField = z.string().uuid().nullable().optional().describe(FOLDER_DESCRIBE);
231
+ /** The `folderId` part of a generate or upload body: present only when the caller named one (null is the root). */
232
+ const folderBody = (folderId) => (folderId === undefined ? {} : { folderId });
219
233
  // ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
220
234
  //
221
235
  // QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
@@ -231,9 +245,9 @@ const BACKGROUND_DESCRIBE = 'Return generationId(s) now instead of blocking; pol
231
245
  const IMAGE_INLINE_REVIEW = 'The image is attached to this result — look at it against the brief before you describe it. ' +
232
246
  'Any claim about how it LOOKS must come from pixels you actually received (slates-vision-feedback-loop).';
233
247
  const VIDEO_REVIEW_POINTER = 'You have NOT seen this clip: call slates_get_asset_video_frames on the asset id above before ' +
234
- 'describing how it looks. A quality claim you cannot point to a tool result for is a REAL NUMBERS ONLY violation.';
248
+ 'describing visible composition or identity. Sampled stills do not establish continuous motion, lip sync or sound; use actual playback through a capable host for those, or report them unreviewed.';
235
249
  const BACKGROUND_REVIEW_POINTER = 'When it completes, look at it before you describe it — slates_get_asset_image for images, ' +
236
- 'slates_get_asset_video_frames for video. For audio, audition the saved file; metadata alone does not establish voice similarity or delivery quality.';
250
+ 'slates_get_asset_video_frames for video. For audio, audition the saved file only through a host that can receive/listen to audio; otherwise report it unreviewed. Metadata does not establish voice similarity or delivery quality.';
237
251
  // The image saved, but reading it back off disk failed (best-effort fetch). The
238
252
  // agent has an asset and NO pixels, which is the one state where a quality
239
253
  // claim would be pure invention — so this branch has to say so rather than
@@ -356,14 +370,95 @@ export const getSelection = {
356
370
  };
357
371
  },
358
372
  };
373
+ // 1.6.1: what is ticked. Mirrors slate's `SELECTION_TARGETS` and `SELECTION_MODES` in `@shared/types/view`
374
+ // (lockstep check 12), which this package cannot import.
375
+ const SELECTION_TARGETS = ['media', 'board'];
376
+ const SELECTION_MODES = ['replace', 'add', 'remove', 'clear', 'all'];
377
+ /**
378
+ * SET what is selected, as the user's clicks do: the cards ticked in Media, the Shots ticked on the Board.
379
+ * Pairs with `slates_get_selection`, which reads it: "tick the shots with no clip, then price them" is this op
380
+ * and then the quote. The desktop applies it through the stores the grids write and answers with what SETTLED.
381
+ */
382
+ export const setSelection = {
383
+ id: 'slates_set_selection',
384
+ description: "Tick or untick cards in Media or Shots on the Board, as clicks, Shift-click, Escape and Ctrl/Cmd+A do: replace makes the ids the whole selection, add and remove change only them, clear empties it, all ticks everything the tab shows (Media: the grid as filtered, folded rounds left out; Board: every Shot on the open board). Media needs its grid showing and the Board needs the Board or Script tab; slates_set_view switches. Ids are asset ids or badge codes (Media) or Shot ids or SHOT-A codes (Board). Answers with the selection as it settled and names any id it could not tick. slates_get_selection reads it back; what is selected is what the user means by 'these'.",
385
+ input: z
386
+ .object({
387
+ surface: z.enum(SELECTION_TARGETS),
388
+ mode: z.enum(SELECTION_MODES),
389
+ ids: z.array(z.string().min(1)).max(500).optional().describe('Asset ids or codes (media) or Shot ids or codes (board). Not used by clear or all.'),
390
+ })
391
+ .strict(),
392
+ async run(input, ctx) {
393
+ await ctx.desktop().requireCapability('selection-set', 'setting what is selected');
394
+ if ((input.mode === 'add' || input.mode === 'remove') && !input.ids?.length)
395
+ throw new Error(`${input.mode} needs ids.`);
396
+ const r = await ctx.desktop().post('/agent/selection', input);
397
+ if (!r.selection)
398
+ return { text: `Nothing was changed (${r.reason ?? 'no answer'}).`, data: { selection: null } };
399
+ const where = r.selection.surface === 'media' ? 'Media' : 'the Board';
400
+ const named = (row) => (row.code ? (row.label ? `${row.code} — ${row.label}` : row.code) : row.id);
401
+ const notes = r.notes?.length ? ` Not applied: ${r.notes.join('; ')}.` : '';
402
+ return {
403
+ text: (r.selection.items.length ? `${r.selection.items.length} selected on ${where}: ${r.selection.items.map(named).join(', ')}.` : `Nothing is selected on ${where}.`) + notes,
404
+ data: { selection: r.selection, notes: r.notes ?? [] },
405
+ };
406
+ },
407
+ };
359
408
  // The view's three lists, mirrored from the desktop's `@shared/types/view`
360
409
  // (`LENSES`, `CUT_SIDES`, `DOCK_SECTIONS`), which this package cannot import.
361
410
  // Lockstep check 12 fails when they differ.
362
411
  const VIEW_LENSES = ['board', 'media', 'script'];
363
412
  const VIEW_CUT_SIDES = ['bottom', 'left', 'right'];
364
413
  const VIEW_DOCK_SECTIONS = ['storyboards', 'library', 'folders', 'pinned'];
414
+ // 1.6.1: what the user is looking at. Mirrors slate's `@shared/types/view` (lockstep check 12).
415
+ const VIEW_BOARD_LEVELS = ['film', 'scenes', 'shot'];
416
+ const VIEW_BOARD_FILTERS = ['all', 'with-clip', 'without-clip'];
417
+ const VIEW_MEDIA_TABS = ['all', 'images', 'videos', 'audio'];
418
+ const VIEW_SETTINGS_PANES = ['account', 'ai', 'storage', 'logs', 'general', 'keys'];
365
419
  /** What each dock section is called on screen. */
366
420
  const DOCK_SECTION_NAMES = { storyboards: 'Boards', library: 'Library', folders: 'Folders', pinned: 'Pinned' };
421
+ /** 1.6.1: what the user is looking at, in the words on screen. Empty parts are left out. */
422
+ const describeLooking = (v) => {
423
+ const parts = [];
424
+ if (!v.projectId)
425
+ parts.push('Home (the project list) is showing.');
426
+ if (v.board?.id) {
427
+ const level = v.board.level === 'film' ? 'Film' : v.board.level === 'shot' ? 'Shot' : 'Scenes';
428
+ const filter = v.board.filter === 'with-clip' ? ', Filter: With a linked clip' : v.board.filter === 'without-clip' ? ', Filter: Without a linked clip' : '';
429
+ parts.push(`The open board is ${v.board.id} (View: ${level}${filter}).`);
430
+ }
431
+ if (v.library?.categoryId)
432
+ parts.push(`A Library page is in Media's place (category ${v.library.categoryId}).`);
433
+ else if (v.media) {
434
+ const narrowing = [
435
+ v.media.unfiledOnly && 'No folder only',
436
+ v.media.search && `search "${v.media.search}"`,
437
+ v.media.favoritesOnly && 'favorites only',
438
+ v.media.linkedOnly && 'only with linked videos',
439
+ ].filter(Boolean);
440
+ parts.push(`Media's tab is ${v.media.tab}${narrowing.length ? `, narrowed to ${narrowing.join(', ')}` : ''}.`);
441
+ }
442
+ if (v.media?.folderId)
443
+ parts.push(`Folder ${v.media.folderId} is chosen: new pictures land in it.`);
444
+ if (v.viewer?.assetId)
445
+ parts.push(`The picture viewer is open on ${v.viewer.assetId}.`);
446
+ if (v.compare?.open)
447
+ parts.push(`Compare is open on ${v.compare.assetIds.length} items.`);
448
+ if (v.animatic?.open)
449
+ parts.push('The animatic is up.');
450
+ if (v.settings?.open)
451
+ parts.push('Settings is open.');
452
+ if (v.composer && !v.composer.open)
453
+ parts.push("The prompt box is hidden (the ' key shows it).");
454
+ if (v.script?.panelShotId)
455
+ parts.push(`The Script page's shot panel is open on Shot ${v.script.panelShotId}.`);
456
+ if (v.layers?.length)
457
+ parts.push(`Open over the page: ${v.layers.join(', ')}.`);
458
+ if (v.notes?.length)
459
+ parts.push(`Not applied: ${v.notes.join('; ')}.`);
460
+ return parts.length ? ' ' + parts.join(' ') : '';
461
+ };
367
462
  const describeView = (v) => {
368
463
  const where = v.cut.full
369
464
  ? 'filling the workspace'
@@ -380,7 +475,8 @@ const describeView = (v) => {
380
475
  return (`The ${v.lens} tab is showing. The timeline is ${where}. ` +
381
476
  `The project navigator is ${v.leftDock.open ? `open (${v.leftDock.width}px)` : 'closed'}, and ${agent}.` +
382
477
  (v.leftDock.folded?.length ? ` Folded in the navigator: ${v.leftDock.folded.map((f) => DOCK_SECTION_NAMES[f] ?? f).join(', ')}.` : '') +
383
- (v.script ? ` The Script page shows ${v.script.details ? 'Words + shots (each shot\'s picture beside its words)' : 'Words (the words alone)'}.` : ''));
478
+ (v.script ? ` The Script page shows ${v.script.details ? 'Words + shots (each shot\'s picture beside its words)' : 'Words (the words alone)'}.` : '') +
479
+ describeLooking(v));
384
480
  };
385
481
  /**
386
482
  * How the Slates window is ARRANGED right now — which lens is showing, where
@@ -389,7 +485,7 @@ const describeView = (v) => {
389
485
  */
390
486
  export const getView = {
391
487
  id: 'slates_get_view',
392
- description: "How the Slates window is arranged: the tab showing (Media, Script or Board), where the timeline sits, which side panels are open, and whether the Script page shows Words or Words + shots.",
488
+ description: "What the user is looking at in Slates: Home or a project, the tab (Media, Script or Board), the open board and its View and Filter, Media's tab, chosen folder and narrowing, a Library page, the picture viewer, Compare, the animatic, Settings, whether the prompt box is hidden, the timeline and the side panels, and any dialog or menu open over the page. Read it before telling the user where to click.",
393
489
  input: z.object({}).strict(),
394
490
  async run(_input, ctx) {
395
491
  await ctx.desktop().requireCapability('view', 'the window layout');
@@ -417,9 +513,10 @@ export const getView = {
417
513
  */
418
514
  export const setView = {
419
515
  id: 'slates_set_view',
420
- description: "Rearrange the Slates window: switch tab (Media, Script or Board), open/close the timeline or park it along the bottom or as a left/right column, open/close or resize the side panels, fold the navigator's sections, and switch the Script page between Words and Words + shots. Only the fields you name change; sizes are clamped by the app and the reply says what it settled on.",
516
+ description: "Change what the user is looking at, as their own clicks would: open a project or go Home, switch tab, open a board, jump to a scene, set the Board's View and Filter, set Media's tab, folder and narrowing, jump to a card in Media, open a Library page, the picture viewer, Compare, the animatic or Settings on a pane, show or hide the prompt box, open or park the timeline or switch the cut on it, open a Shot's panel on Script. Only the fields you name change; the app clamps sizes, answers with what it settled on, and names any field it could not apply. A project change is applied alone (it restores that project's own view). Use it when the user asks you to show or open something, not to answer a question they can see for themselves.",
421
517
  input: z
422
518
  .object({
519
+ projectId: z.string().nullable().optional().describe('Open this project, or null for Home. Applied alone.'),
423
520
  lens: z.enum(VIEW_LENSES).optional().describe('Which tab the centre shows.'),
424
521
  cut: z
425
522
  .object({
@@ -429,8 +526,9 @@ export const setView = {
429
526
  .enum(VIEW_CUT_SIDES)
430
527
  .optional()
431
528
  .describe('Where the timeline is parked. A column suits a wide monitor; picking a side opens the timeline.'),
432
- height: z.number().optional().describe('The bottom band\'s height in px.'),
433
- width: z.number().optional().describe('The side column\'s width in px.'),
529
+ height: z.number().optional().describe("The bottom band's height in px."),
530
+ width: z.number().optional().describe("The side column's width in px."),
531
+ timelineId: z.string().optional().describe('The named cut to show on the timeline (slates_list_timelines).'),
434
532
  })
435
533
  .strict()
436
534
  .optional(),
@@ -456,21 +554,67 @@ export const setView = {
456
554
  .describe('The Studio Agent panel on the right. It cannot be opened while the agent is off in Settings.'),
457
555
  script: z
458
556
  .object({
459
- details: z.boolean().optional().describe('Words + shots (true): each shot\'s picture beside its words on the Script page. Words (false): the words alone.'),
557
+ details: z.boolean().optional().describe("Words + shots (true): each shot's picture beside its words on the Script page. Words (false): the words alone."),
558
+ textScale: z.number().optional().describe("The page's text size; 1 is the default."),
559
+ panelShotId: z.string().nullable().optional().describe("Open this Shot's details panel on the Script page (it binds the Shot to the prompt box, as a click does); null closes it."),
460
560
  })
461
561
  .strict()
462
562
  .optional()
463
563
  .describe('The Script page.'),
564
+ board: z
565
+ .object({
566
+ id: z.string().optional().describe('Open this board (on Board and Script).'),
567
+ sceneId: z.string().optional().describe('Scroll the Board to this scene.'),
568
+ level: z.enum(VIEW_BOARD_LEVELS).optional().describe('View: Film (posters), Scenes (the working card) or Shot (each card open).'),
569
+ cardWidth: z.number().optional().describe('Card width in px, anywhere between the stops.'),
570
+ filter: z.enum(VIEW_BOARD_FILTERS).optional().describe('Filter: every shot, only shots with a linked clip, or only shots without one.'),
571
+ linkedClips: z.boolean().optional().describe('Show the row of linked clips under each shot.'),
572
+ scriptFollowsDrag: z.boolean().optional().describe("The board menu's Script follows a drag."),
573
+ collapsedSceneIds: z.array(z.string()).optional().describe('Every scene to fold; a scene not named unfolds.'),
574
+ })
575
+ .strict()
576
+ .optional(),
577
+ media: z
578
+ .object({
579
+ tab: z.enum(VIEW_MEDIA_TABS).optional(),
580
+ folderId: z.string().nullable().optional().describe('Choose this folder (new pictures land in it), or null for all media.'),
581
+ unfiledOnly: z.boolean().optional().describe('Show only media in no folder.'),
582
+ search: z.string().optional().describe("Media's search box (matches prompts)."),
583
+ favoritesOnly: z.boolean().optional(),
584
+ linkedOnly: z.boolean().optional().describe('Images tab: only pictures with a linked video.'),
585
+ rounds: z.boolean().optional().describe('Group by generation.'),
586
+ cardSize: z.number().optional().describe('Card size in px.'),
587
+ revealAssetId: z.string().optional().describe('Jump to this card in Media, undoing only what hides it (the app says what it changed).'),
588
+ })
589
+ .strict()
590
+ .optional(),
591
+ library: z.object({ categoryId: z.string().nullable() }).strict().optional().describe("Open a Library category's page in Media's place; null goes back to the grid."),
592
+ viewer: z.object({ assetId: z.string().nullable() }).strict().optional().describe('Open the picture viewer on a picture; null closes it.'),
593
+ compare: z
594
+ .object({ open: z.boolean().optional(), assetIds: z.array(z.string()).max(4).optional().describe('The compare set, two to four items.') })
595
+ .strict()
596
+ .optional(),
597
+ animatic: z.object({ open: z.boolean(), fromShotId: z.string().optional() }).strict().optional().describe('Play the board as a rough cut from a Shot (the Board or Script tab must be showing).'),
598
+ settings: z.object({ open: z.boolean(), pane: z.enum(VIEW_SETTINGS_PANES).optional() }).strict().optional().describe('Open Settings, on a pane (ai is AI tools, keys is API keys).'),
599
+ composer: z.object({ open: z.boolean() }).strict().optional().describe('Show or hide the prompt box.'),
464
600
  })
465
601
  .strict(),
466
602
  async run(input, ctx) {
467
603
  await ctx.desktop().requireCapability('view', 'the window layout');
468
604
  if (Object.keys(input).length === 0) {
469
- throw new Error('Name at least one of lens, cut, leftDock, studioAgent or script — there is nothing to change otherwise.');
605
+ throw new Error('Name at least one field to change, e.g. lens, board, media, viewer or settings.');
470
606
  }
607
+ // The 1.6.1 fields need a desktop that knows them; an older one would drop them without a word.
608
+ const LOOKING = ['projectId', 'board', 'media', 'library', 'viewer', 'compare', 'animatic', 'settings', 'composer'];
609
+ const widened = LOOKING.some((k) => input[k] !== undefined) ||
610
+ input.cut?.timelineId !== undefined ||
611
+ input.script?.textScale !== undefined ||
612
+ input.script?.panelShotId !== undefined;
613
+ if (widened)
614
+ await ctx.desktop().requireCapability('view-v2', 'showing and opening things in the window');
471
615
  const r = await ctx.desktop().post('/agent/view', input);
472
616
  if (!r.view) {
473
- return { text: r.reason ? `Nothing was rearranged (${r.reason}).` : 'Nothing was rearranged.', data: { view: null } };
617
+ return { text: r.reason ? `Nothing was changed (${r.reason}).` : 'Nothing was changed.', data: { view: null } };
474
618
  }
475
619
  // A desktop whose view predates folding takes the patch and ignores `folded`.
476
620
  const noFolds = input.leftDock?.folded !== undefined && r.view.leftDock.folded === undefined
@@ -479,6 +623,288 @@ export const setView = {
479
623
  return { text: describeView(r.view) + noFolds, data: { view: r.view } };
480
624
  },
481
625
  };
626
+ /**
627
+ * SHOW AND POINT (1.6.1). The manual's pictures and a ring around a live control, so an agent can
628
+ * answer "where is it?" with the thing itself instead of a paragraph. Neither changes the project or
629
+ * the view. The pictures come from the INSTALLED app (it ships them), so they show the user's version;
630
+ * the public URL is for a client that renders only links. Decisions 13-16 in second-brain
631
+ * plans/2026-09-30-slates-manual-for-llms-1-6-1-decisions.md.
632
+ */
633
+ export const getManualPicture = {
634
+ id: 'slates_get_manual_picture',
635
+ description: 'A picture of a Slates screen from the app manual, with numbered callouts and their legend, taken from the version the user runs. Pass the picture id the manual names (app-manual sections show "Picture `id`"); no id lists them. Show one when the user cannot find something or asks what a screen looks like; never unasked. When the control is on screen, slates_highlight_control points at the real thing instead.',
636
+ input: z
637
+ .object({ id: z.string().min(1).optional().describe('Picture id from the manual, e.g. "window-overview". Omit to list.') })
638
+ .strict(),
639
+ async run(input, ctx) {
640
+ await ctx.desktop().requireCapability('manual-pictures', "the manual's pictures");
641
+ if (!input.id) {
642
+ const r = await ctx.desktop().get('/agent/manual/pictures');
643
+ return ok(r, r.pictures.length ? r.pictures.map((p) => `${p.id}: ${p.title}`).join('\n') : 'This Slates ships no manual pictures.');
644
+ }
645
+ const p = await ctx.desktop().get('/agent/manual/picture', { id: input.id });
646
+ const legend = p.callouts.length ? '\n' + p.callouts.map((c) => `${c.n}. ${c.name} (${c.target})`).join('\n') : '';
647
+ return {
648
+ text: `${p.title} (Slates ${p.appVersion}). Link: ${p.url}${legend}`,
649
+ images: [{ data: p.data, mimeType: p.mimeType }],
650
+ data: { id: p.id, title: p.title, url: p.url, appVersion: p.appVersion, callouts: p.callouts },
651
+ };
652
+ },
653
+ };
654
+ export const highlightControl = {
655
+ id: 'slates_highlight_control',
656
+ description: "Point at a control in the user's Slates window: a grey outline around it for a few seconds, with your caption beside it. It clicks nothing and never changes the view; the user's next click clears it. Target ids are listed at the end of each app-manual surface section (\"controls you can point at\"), or pass list:true to get every id with whether it is on screen now. shown:false means it is not on screen: tell the user the way there, or open that view with slates_set_view if they want you to.",
657
+ input: z
658
+ .object({
659
+ target: z.string().min(1).optional().describe('Control id, e.g. "shell.titlebar.search".'),
660
+ caption: z.string().max(140).optional().describe('A few words shown beside it, e.g. "Click here to open the timeline".'),
661
+ list: z.boolean().optional().describe('Return every target id, its on-screen name, where it is, and whether it is on screen now.'),
662
+ })
663
+ .strict(),
664
+ async run(input, ctx) {
665
+ await ctx.desktop().requireCapability('ui-pointer', 'pointing at controls');
666
+ if (input.list || !input.target) {
667
+ const r = await ctx.desktop().get('/agent/ui/targets');
668
+ const on = r.targets.filter((t) => t.onScreen);
669
+ return ok(r, `${r.targets.length} controls; on screen now: ${on.map((t) => `${t.id} (${t.name})`).join(', ') || 'none reported'}.`);
670
+ }
671
+ const r = await ctx.desktop().post('/agent/ui/highlight', {
672
+ target: input.target,
673
+ caption: input.caption,
674
+ });
675
+ return ok(r, r.shown
676
+ ? `Pointing at ${r.name} (${r.where}) in the user's window.`
677
+ : `${r.name} is not on screen (${r.reason ?? 'not shown'}). It is at: ${r.where}. Tell the user the way there, or open that view with slates_set_view if they ask you to.`);
678
+ },
679
+ };
680
+ const COMPOSER_ROLES = Object.keys(ATTACHMENT_ROLE_DESCRIPTION);
681
+ function describeComposer(r) {
682
+ const lines = [];
683
+ const where = r.bound
684
+ ? `editing Shot ${r.bound.code ?? r.bound.shotId}${r.bound.name ? ` "${r.bound.name}"` : ''} (every change is saved to it)`
685
+ : r.draftFrom
686
+ ? `a draft from ${r.draftFrom.code ?? r.draftFrom.assetId}`
687
+ : 'an unsaved draft';
688
+ lines.push(`Prompt box (${r.open ? 'showing' : 'hidden'}), ${r.lane} lane, ${r.model.label}${r.destination === 'chatgpt' ? ' (ChatGPT)' : ''}; ${where}.`);
689
+ lines.push(`Words (${r.prompt.length}${r.promptMax ? `/${r.promptMax}` : ''}): ${r.prompt ? JSON.stringify(r.prompt) : 'none'}`);
690
+ if (r.params.length)
691
+ lines.push(`Settings: ${r.params.map((p) => `${p.label} ${p.value || '(none)'}`).join(' · ')}`);
692
+ if (r.tiles.length) {
693
+ const tiles = r.tiles.map((t) => `${t.badge} ${t.code ?? t.name}${t.role ? ` (${t.role})` : t.token ? ` (${t.token})` : ''}${t.voice ? ' voice' : ''}${t.sent ? '' : ' NOT SENT'}`);
694
+ lines.push(`References: ${tiles.join(', ')}`);
695
+ }
696
+ const ti = r.toolInputs;
697
+ if (ti && r.tool === 'lip-sync')
698
+ lines.push(`Lip Sync source: ${ti.sourceAssetId ?? 'none'}${ti.sourceType ? ` (${ti.sourceType})` : ''}; Speech Text: ${ti.speechText ? JSON.stringify(ti.speechText) : 'none'}`);
699
+ if (ti && r.tool === 'motion-transfer')
700
+ lines.push(`Motion: ${ti.drivingVideoAssetId ?? 'none'}; Character: ${ti.characterImageAssetId ?? 'none'}`);
701
+ if (r.composedPrompt !== r.prompt)
702
+ lines.push(`Sent as: ${JSON.stringify(r.composedPrompt)}`);
703
+ if (r.unresolved.length)
704
+ lines.push(`Match nothing (sent as plain words): ${r.unresolved.join(', ')}`);
705
+ if (r.dangling.length)
706
+ lines.push(`Point past what is sent: ${r.dangling.join(', ')}`);
707
+ if (r.notice)
708
+ lines.push(`Notice line: ${r.notice.text}`);
709
+ lines.push(`Button: ${r.generate}.${r.running ? ` ${r.running} generating.` : ''}`);
710
+ return lines.join('\n');
711
+ }
712
+ export const getComposer = {
713
+ id: 'slates_get_composer',
714
+ description: "What the prompt box holds right now, as the user sees it: the lane and model, whether it is editing a Shot or a draft, the words, every setting with the values it offers, each reference tile with its number, role and whether the model's limit leaves it out, the exact text the model receives (See what gets sent), mentions that match nothing, the notice line, and the Generate button's price. Read it before telling the user what a press will do, or before staging with slates_set_composer.",
715
+ input: z.object({}).strict(),
716
+ async run(_input, ctx) {
717
+ await ctx.desktop().requireCapability('composer', 'reading the prompt box');
718
+ const r = await ctx.desktop().get('/agent/composer');
719
+ if (!r.report)
720
+ return { text: r.reason ?? 'The prompt box did not answer.', data: { report: null } };
721
+ return { text: describeComposer(r.report), data: r.report };
722
+ },
723
+ };
724
+ export const setComposer = {
725
+ id: 'slates_set_composer',
726
+ description: "Stage a request in the user's prompt box, as their own clicks would, and leave it for them to press Generate: edit a Shot in it or stop, clear or restore the draft, reuse a result's prompt, edit a clip, pick the lane and model, write the words, attach pictures, clips or audio with roles, re-file or remove them, set settings and the voice, fill Lip Sync or Motion Control's inputs. It never generates and never spends; the answer is the box as it settled (the same report as slates_get_composer) plus a line for anything it could not apply and why. Use it when the user wants a setup ready to look at; to generate yourself, quote with slates_estimate_generation_cost and use the generate ops.",
727
+ input: z
728
+ .object({
729
+ projectId: z.string().uuid().optional().describe('The project you expect the box to be in; refused if another is open.'),
730
+ bindShotId: z
731
+ .string()
732
+ .min(1)
733
+ .nullable()
734
+ .optional()
735
+ .describe('Edit this Shot in the box (id or code; every change is saved to it); null stops editing and brings the draft back.'),
736
+ clear: z.boolean().optional().describe('Empty the unsaved draft; restoreDraft undoes it. Not while a Shot is bound.'),
737
+ restoreDraft: z.boolean().optional().describe('Put back the draft that Clear, Reuse prompt or Edit with AI replaced (they swap).'),
738
+ restoreSetup: z.boolean().optional().describe('On a bound Shot, swap back the setup Continue replaced (writes the Shot; again swaps back).'),
739
+ fromAssetId: z.string().min(1).optional().describe("Reuse prompt: this result's recipe replaces the box (id or code)."),
740
+ editSourceAssetId: z.string().min(1).optional().describe('Edit with AI: this clip becomes the canvas and the box starts clean.'),
741
+ lane: z.enum(['image', 'video', 'audio']).optional().describe('The Image / Video / Audio tab; it lands on the last model used there.'),
742
+ model: z
743
+ .string()
744
+ .min(1)
745
+ .optional()
746
+ .describe('A model id (slates_list_available_models), `<id>::face` for a face route, `chatgpt`, `lip-sync` or `motion-transfer`.'),
747
+ prompt: z.string().optional().describe('Replaces the words. @name and #name attach what they name, as typing does.'),
748
+ addMentions: z
749
+ .array(z.string().regex(/^[@#]\S+$/))
750
+ .optional()
751
+ .describe('@name / #name added at the end of the words (slates_list_library gives each item its mention).'),
752
+ attach: z
753
+ .array(z.object({ assetId: z.string().min(1), role: z.enum(COMPOSER_ROLES).optional() }).strict())
754
+ .optional()
755
+ .describe('Pictures, clips or audio from the project (id or code). A picture with no role goes in as a dropped one does; a clip is a video-reference, audio an audio-reference.'),
756
+ detach: z
757
+ .array(z.string().min(1))
758
+ .optional()
759
+ .describe("Take these out of the box: asset ids or codes, or a character's id to drop the voice her @mention brought."),
760
+ setRole: z
761
+ .array(z.object({ assetId: z.string().min(1), role: z.enum(COMPOSER_ROLES) }).strict())
762
+ .optional()
763
+ .describe("Re-file pictures already in the box, as a tile's role menu does."),
764
+ params: z
765
+ .record(z.union([z.string(), z.number(), z.boolean()]))
766
+ .optional()
767
+ .describe('Settings by id, each to a value it offers (slates_get_composer lists both), applied after the model.'),
768
+ voice: z
769
+ .union([
770
+ z.object({ presetId: z.string().min(1) }).strict(),
771
+ z.object({ clipAssetId: z.string().min(1) }).strict(),
772
+ z.object({ description: z.string().min(1) }).strict(),
773
+ ])
774
+ .optional()
775
+ .describe('The voice on a voice model: a preset (slates_list_voices), an audio clip in the project, or described in words.'),
776
+ tool: z
777
+ .object({
778
+ sourceAssetId: z.string().min(1).nullable().optional().describe('Lip Sync: the picture or clip with the face; null clears it.'),
779
+ speechText: z.string().max(120).optional().describe('Lip Sync, Text to speech: the words spoken (Speech Text).'),
780
+ drivingVideoAssetId: z.string().min(1).nullable().optional().describe('Motion Control: the clip whose motion is copied (Motion).'),
781
+ characterImageAssetId: z.string().min(1).nullable().optional().describe('Motion Control: the picture to animate (Character).'),
782
+ })
783
+ .strict()
784
+ .optional()
785
+ .describe('Lip Sync and Motion Control inputs (pick model lip-sync or motion-transfer first). An uploaded audio file is only the user.'),
786
+ })
787
+ .strict(),
788
+ async run(input, ctx) {
789
+ await ctx.desktop().requireCapability('composer', 'staging the prompt box');
790
+ if (Object.keys(input).filter((k) => k !== 'projectId').length === 0) {
791
+ throw new Error('Name at least one change, e.g. model, prompt, attach or params. To read the box, use slates_get_composer.');
792
+ }
793
+ const r = await ctx.desktop().post('/agent/composer', input);
794
+ if (!r.report)
795
+ return { text: `Nothing was staged (${r.reason ?? 'no answer'}).`, data: { report: null } };
796
+ const skipped = r.skipped?.length ? `\nNot applied:\n- ${r.skipped.join('\n- ')}` : '';
797
+ return { text: `${describeComposer(r.report)}${skipped}`, data: { report: r.report, skipped: r.skipped ?? [] } };
798
+ },
799
+ };
800
+ /**
801
+ * AGENT PARITY (1.6.1): what a user could do in the window and an agent could not. Each calls a desktop
802
+ * route that runs the window's own function (`slate/src/main/agent/routes-parity.ts`). Plan step 5b,
803
+ * second-brain plans/2026-09-30-slates-manual-for-llms-1-6-1.md.
804
+ */
805
+ export const reorderFolders = {
806
+ id: 'slates_reorder_folders',
807
+ description: "Reorder the Folders section of the left dock, as dragging a row does. Pass folder ids in the new order; ones you leave out keep their order after them.",
808
+ input: z.object({ projectId: z.string().uuid(), folderIds: z.array(z.string().min(1)).min(1) }).strict(),
809
+ async run(input, ctx) {
810
+ await ctx.desktop().requireCapability('reorder', 'reordering folders and pins');
811
+ return ok(await ctx.desktop().post('/agent/folders/reorder', input));
812
+ },
813
+ };
814
+ export const reorderPins = {
815
+ id: 'slates_reorder_pins',
816
+ description: "Reorder the Pinned section of the left dock, as dragging a pin does. Pass the pinned images (UUIDs or badge codes) in the new order; ones you leave out keep their order after them.",
817
+ input: z.object({ projectId: z.string().uuid(), assetIds: z.array(z.string().min(1)).min(1) }).strict(),
818
+ async run(input, ctx) {
819
+ await ctx.desktop().requireCapability('reorder', 'reordering folders and pins');
820
+ const resolved = await resolveAssetRefs(ctx, input.projectId, input.assetIds);
821
+ return ok(await ctx.desktop().post('/agent/pins/reorder', { projectId: input.projectId, assetIds: input.assetIds.map((r) => resolved.get(r).id) }));
822
+ },
823
+ };
824
+ export const getUsage = {
825
+ id: 'slates_get_usage',
826
+ description: 'What Settings → Account → Usage shows: estimated spend across every project for this month (default), the last 7 days or all time, the image and video counts, video seconds and the models spent on most.',
827
+ input: z.object({ period: z.enum(['week', 'month', 'all']).optional() }).strict(),
828
+ async run(input, ctx) {
829
+ await ctx.desktop().requireCapability('usage', 'usage');
830
+ return ok(await ctx.desktop().get('/agent/usage', { period: input.period ?? 'month' }));
831
+ },
832
+ };
833
+ export const getAppSettings = {
834
+ id: 'slates_get_app_settings',
835
+ description: "The app's own settings an agent may read: the app version, the projects folder, which tab a never-opened project lands on (newProjectLens), whether ChatGPT images is on, and (from Slates 1.6.1) who does the Studio Agent's thinking (studioAgentHost: slates, codex or claude) with each host's model and thinking level ('' is its default). Never keys or sign-in.",
836
+ input: z.object({}).strict(),
837
+ async run(_input, ctx) {
838
+ await ctx.desktop().requireCapability('app-settings', 'app settings');
839
+ return ok(await ctx.desktop().get('/agent/app-settings'));
840
+ },
841
+ };
842
+ const HOST_KEYS = ['studioAgentHost', 'studioAgentCodexModel', 'studioAgentCodexEffort', 'studioAgentClaudeModel', 'studioAgentClaudeEffort'];
843
+ export const setAppSettings = {
844
+ id: 'slates_set_app_settings',
845
+ description: "Change an app setting the user asked you to: newProjectLens (the tab a never-opened project lands on: media, script or board), chatGptImagesEnabled (Settings → AI tools → ChatGPT images), or who does the Studio Agent's thinking: studioAgentHost (slates, codex for the user's own Codex on their ChatGPT plan, claude for their own Claude Code on their Claude plan) and each host's model and thinking level, ids from the host's own list or '' for its default. The projects folder, keys and sign-in are the user's to change.",
846
+ input: z
847
+ .object({
848
+ newProjectLens: z.enum(['media', 'script', 'board']).optional(),
849
+ chatGptImagesEnabled: z.boolean().optional(),
850
+ studioAgentHost: z.enum(['slates', 'codex', 'claude']).optional(),
851
+ studioAgentCodexModel: z.string().max(120).optional(),
852
+ studioAgentCodexEffort: z.string().max(120).optional(),
853
+ studioAgentClaudeModel: z.string().max(120).optional(),
854
+ studioAgentClaudeEffort: z.string().max(120).optional(),
855
+ })
856
+ .strict(),
857
+ async run(input, ctx) {
858
+ await ctx.desktop().requireCapability('app-settings', 'app settings');
859
+ // An older desktop has the route but not these keys: ask it by capability, so it says "update Slates".
860
+ if (HOST_KEYS.some((key) => input[key] !== undefined)) {
861
+ await ctx.desktop().requireCapability('agent-host', "choosing who does the Studio Agent's thinking");
862
+ }
863
+ return ok(await ctx.desktop().post('/agent/app-settings', input));
864
+ },
865
+ };
866
+ export const getAsset = {
867
+ id: 'slates_get_asset',
868
+ description: 'One asset in full (UUID or badge code): its recorded prompt, model, settings and inputs, its source picture and linked clips, and how many clips and board shots use it. slates_list_assets rows are compact; read this before reusing or explaining one card.',
869
+ input: z.object({ projectId: z.string().uuid(), assetId: z.string().min(1) }).strict(),
870
+ async run(input, ctx) {
871
+ await ctx.desktop().requireCapability('asset-detail', 'reading one asset in full');
872
+ const id = (await resolveAssetRefs(ctx, input.projectId, [input.assetId])).get(input.assetId).id;
873
+ return ok(await ctx.desktop().get('/agent/assets/get', { id }));
874
+ },
875
+ };
876
+ export const linkAssetSource = {
877
+ id: 'slates_link_asset_source',
878
+ description: "Link a clip to the picture it came from, as the clip card's Link video to image does (it adds a link; a clip can have several), or pass imageAssetId null to unlink it from every picture. UUIDs or badge codes.",
879
+ input: z
880
+ .object({ projectId: z.string().uuid(), assetId: z.string().min(1).describe('The clip.'), imageAssetId: z.string().min(1).nullable().describe('The picture, or null to unlink.') })
881
+ .strict(),
882
+ async run(input, ctx) {
883
+ await ctx.desktop().requireCapability('asset-link', 'linking a clip to its picture');
884
+ const refs = [input.assetId, ...(input.imageAssetId ? [input.imageAssetId] : [])];
885
+ const resolved = await resolveAssetRefs(ctx, input.projectId, refs);
886
+ return ok(await ctx.desktop().post('/agent/assets/link-source', {
887
+ assetId: resolved.get(input.assetId).id,
888
+ imageAssetId: input.imageAssetId ? resolved.get(input.imageAssetId).id : null,
889
+ }));
890
+ },
891
+ };
892
+ export const extractVideoFrame = {
893
+ id: 'slates_extract_video_frame',
894
+ description: "Save one frame of a clip as a new picture in Media, as the clip card's camera does: at 'first', 'last' (default 'first') or a second. Free. The usual continuity move: extract a clip's last frame and use it as the next clip's first frame.",
895
+ input: z
896
+ .object({
897
+ projectId: z.string().uuid(),
898
+ assetId: z.string().min(1).describe('The clip, UUID or badge code.'),
899
+ at: z.union([z.enum(['first', 'last']), z.number().min(0)]).optional(),
900
+ })
901
+ .strict(),
902
+ async run(input, ctx) {
903
+ await ctx.desktop().requireCapability('frame-extract', 'extracting a frame');
904
+ const id = (await resolveAssetRefs(ctx, input.projectId, [input.assetId])).get(input.assetId).id;
905
+ return ok(await ctx.desktop().post('/agent/assets/extract-frame', { assetId: id, at: input.at ?? 'first' }));
906
+ },
907
+ };
482
908
  export const getMe = {
483
909
  id: 'slates_get_me',
484
910
  description: 'Identity, license tier, and credit balance for the connected Slates account.',
@@ -539,8 +965,8 @@ export const VIDEO_MODELS = [
539
965
  'kling-v3.0-std',
540
966
  'kling-v3.0-pro',
541
967
  'kling-v3.0-omni',
542
- 'veo-3.1-fast',
543
- 'veo-3.1-standard',
968
+ // Veo 3.1 Fast and Standard were retired on 2026-10-02 (Eric). The server
969
+ // keeps their keys for older desktops only; nothing here offers them.
544
970
  'seedance-2',
545
971
  // Seedance 2.5 is the DEFAULT video model (Eric, 2026-09-13): 30s takes, 30
546
972
  // image references, audio-only references, up to 1080p (2026-08-24). 2.0 stays
@@ -565,8 +991,8 @@ export const VIDEO_MODELS = [
565
991
  'minimax-h3-max',
566
992
  'minimax-h3-max-turbo',
567
993
  // LTX-2.5, two seats in one family (2026-08-29). Base is the VOLUME seat —
568
- // cheapest native 1080p second we sell, free native audio at every tier, the
569
- // only row reaching 1440p, and the longest clips in the catalogue (20s).
994
+ // cheapest 1080p second with sound included, free native audio at every tier,
995
+ // the only row reaching 1440p, and clips up to 20s.
570
996
  // Pro is the fidelity seat and is NOT a superset: shorter ladder (no
571
997
  // 1440p/4K), shorter clips (6/8/10 only) and ~1/3 dearer.
572
998
  //
@@ -688,7 +1114,7 @@ function editClipBounds(model) {
688
1114
  const EDIT_VIDEO_RESOLUTIONS = videoResolutionUnion(['seedance-2.5-edit']);
689
1115
  export const estimateGenerationCost = {
690
1116
  id: 'slates_estimate_generation_cost',
691
- description: 'Quote credits before any generate_* op. Accepts the same base model ids and parameters as generation, or an exact registry cost key. Pairs with the confirm gate.',
1117
+ description: 'Quote credits before any generate_* op. Accepts the generation model ids and parameters (edit seats other than Kling, lip-sync and motion-transfer engines need an exact registry cost key), or an exact registry cost key. Pairs with the confirm gate.',
692
1118
  input: z.object({
693
1119
  model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
694
1120
  quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
@@ -706,8 +1132,8 @@ export const estimateGenerationCost = {
706
1132
  videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe('Video only. Omitted, each model quotes at its own default. Per-model ladders: see slates_generate_video\'s videoResolution.'),
707
1133
  resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only. Omit for the model default; pass the same value to generation.'),
708
1134
  quality: z.enum(GPT_QUALITY_TIERS).optional().describe(`GPT Image tier; default ${DEFAULT_GPT_QUALITY}.`),
709
- aspectRatio: z.string().optional().describe('Image only. 1:1/4:3/3:4 cost more than 16:9.'),
710
- sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
1135
+ aspectRatio: z.string().optional().describe('GPT Image only: 1:1, 4:3 and 3:4 cost more than 16:9.'),
1136
+ sound: z.boolean().optional().describe('Kling 3.0: the audio flag changes the cost key. Omitted means sound on, as generation bills it; pass false to price a silent take. Kling 4K keys include audio.'),
711
1137
  seedanceFace: z.boolean().optional().describe('Seedance AI-face route (pricier key).'),
712
1138
  seedanceRealFace: z.boolean().optional().describe('Seedance consented real-face route (premium key).'),
713
1139
  videoRefSeconds: z.number().nonnegative().optional().describe('Combined reference-video seconds, measured from the clips.'),
@@ -1094,7 +1520,7 @@ export const getAssetsBatch = {
1094
1520
  };
1095
1521
  export const getAssetVideoFrames = {
1096
1522
  id: 'slates_get_asset_video_frames',
1097
- description: 'Extract N evenly-spaced keyframes from a video asset and return them inline as base64 JPEGs. This is the "see the video" path — LLMs can\'t consume video natively, so frames are the next best thing. Default 3 frames (start / middle / end). Bump to 5-8 for longer clips or when motion is the whole story. Use this before writing a motion-transfer prompt, a lip-sync refinement, or any iteration on a video clip. Response carries the asset\'s code (e.g. VID-V3) + label — name them when discussing the clip with the user.',
1523
+ description: 'Extract evenly-spaced sampled still frames from a video asset and return them inline as JPEGs. Inspect visible appearance, composition and identity before revising a prompt. Stills do not verify continuous motion, timing, lip sync or sound; actual playback through a capable host is needed for those claims. Default 3 samples; count can request more. The response includes the asset code and label; use them when discussing the clip.',
1098
1524
  input: z.object({
1099
1525
  id: z.string().uuid(),
1100
1526
  count: z.number().int().min(1).max(8).optional().describe('Number of frames to extract. Default 3.'),
@@ -1149,10 +1575,13 @@ export const generateChatGptImage = {
1149
1575
  description: 'Generate an image through the local Codex host using the connected ChatGPT account, then save it in Slates with its exact submitted prompt, reference lineage and measured dimensions. Uses ChatGPT account limits, never Slates credits or a paid API fallback. Check slates_get_chatgpt_status first. Supply a new UUID requestId once per intended generation; reuse it for retries to avoid duplicate generation/import. No explicit image model, quality or size controls are exposed. Prompt may request visual properties without guaranteeing them. Background returns a generationId for slates_get_generation_status.',
1150
1576
  input: z.object({ projectId: z.string().uuid(), requestId: z.string().uuid(), prompt: z.string().min(1),
1151
1577
  aspectRatio: z.enum(CHATGPT_FRAMING_RATIOS).optional().describe('Optional framing request appended verbally to the prompt, not an exact output-size guarantee.'),
1152
- referenceAssetIds: z.array(z.string().min(1)).optional(), background: z.boolean().optional() }),
1578
+ referenceAssetIds: z.array(z.string().min(1)).optional(), background: z.boolean().optional(), folderId: folderIdField }),
1153
1579
  async run(input, ctx) {
1154
1580
  const desktop = ctx.desktop();
1155
1581
  await desktop.requireCapability('chatgpt-image-generation', 'ChatGPT image generation');
1582
+ // A desktop before the folder rule ignores folderId and files at the root.
1583
+ if (input.folderId !== undefined)
1584
+ await desktop.requireCapability('media-folder', 'choosing the folder a result lands in');
1156
1585
  const refs = await resolveAssetRefs(ctx, input.projectId, input.referenceAssetIds ?? []);
1157
1586
  const result = await desktop.post('/agent/generation/chatgpt-image', {
1158
1587
  ...input, referenceAssetIds: (input.referenceAssetIds ?? []).map(id => refs.get(id).id),
@@ -1203,17 +1632,22 @@ export const uploadReferenceImage = {
1203
1632
  .enum(['image', 'video', 'audio'])
1204
1633
  .optional()
1205
1634
  .describe('Asset kind for a filePath import — "image" (default), "video", or "audio". A dataUrl is always an image.'),
1635
+ folderId: folderIdField,
1206
1636
  })
1207
1637
  .refine((d) => !!d.filePath !== !!d.dataUrl, {
1208
1638
  message: 'Pass exactly one of filePath or dataUrl',
1209
1639
  }),
1210
1640
  async run(input, ctx) {
1211
1641
  const desktop = ctx.desktop();
1642
+ // A desktop before the folder rule ignores folderId and files at the root.
1643
+ if (input.folderId !== undefined)
1644
+ await desktop.requireCapability('media-folder', 'choosing the folder a result lands in');
1212
1645
  if (input.filePath) {
1213
1646
  const r = await desktop.post('/agent/assets/upload', {
1214
1647
  projectId: input.projectId,
1215
1648
  filePath: input.filePath,
1216
1649
  type: input.type ?? 'image',
1650
+ ...folderBody(input.folderId),
1217
1651
  });
1218
1652
  return ok(r);
1219
1653
  }
@@ -1223,6 +1657,7 @@ export const uploadReferenceImage = {
1223
1657
  const r = await desktop.post('/agent/assets/upload-base64', {
1224
1658
  projectId: input.projectId,
1225
1659
  dataUrl: input.dataUrl,
1660
+ ...folderBody(input.folderId),
1226
1661
  });
1227
1662
  return ok(r);
1228
1663
  },
@@ -1761,19 +2196,21 @@ const LEGACY_DESKTOP_IMAGE_BATCH = 4;
1761
2196
  export const generateImage = {
1762
2197
  id: 'slates_generate_image',
1763
2198
  billable: true,
1764
- description: 'Generate an image via Slates credits.\n' +
1765
- // GENERATED from MODEL_FACTS — the hand-typed model list that stood here
1766
- // was a third copy of the routing doctrine, and it had already gone stale
1767
- // (it still described nano-banana-2-lite by a capability the param owns).
1768
- `${describeRouting('image')}\n` +
1769
- 'Full table: the slates-model-selection skill. ' +
1770
- 'Pass projectId to save into a Slates project (asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). ' +
2199
+ description:
2200
+ // Rules first: Claude Code keeps only the first 2,048 characters of a tool
2201
+ // description, and the generated roster below runs past that cut.
2202
+ 'Generate an image via Slates credits. Pass projectId to save into a Slates project (asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). You MUST pass aspectRatio (the server returns requires_clarification when missing); resolution defaults to the model\'s own. ' +
1771
2203
  CONFIRM_GATE_SENTENCE +
1772
- ' MCP/CLI generation always charges credits. No skills installed? Call slates_get_prompting_guide with the model\'s topic and \'slates-cost-discipline\' first. ' +
2204
+ ' MCP/CLI generation always charges credits. The estimate returns the model\'s prompting card; load its slates-prompting-* guide with slates_get_prompting_guide for anything the card leaves out. ' +
1773
2205
  // GENERATED from the skill file's own never-use list -- the one piece of
1774
2206
  // prompting doctrine that is ALWAYS in context, because the agent has
1775
2207
  // demonstrably skipped the call that would have taught it.
1776
- describeBannedTokens('image'),
2208
+ describeBannedTokens('image') +
2209
+ // GENERATED from MODEL_FACTS — the hand-typed model list that stood here
2210
+ // was a third copy of the routing doctrine, and it had already gone stale
2211
+ // (it still described nano-banana-2-lite by a capability the param owns).
2212
+ `\n${describeRouting('image')}\n` +
2213
+ 'Full table: the slates-model-selection skill.',
1777
2214
  input: z.object({
1778
2215
  prompt: z.string().min(1).max(4000),
1779
2216
  model: zEnum(IMAGE_MODELS).optional().describe(`Image model. Omitted: ${DEFAULT_IMAGE_MODEL} with projectId, nano-banana-2 (the only headless seat) without. Routing: slates-model-selection skill.`),
@@ -1787,8 +2224,12 @@ export const generateImage = {
1787
2224
  referenceAssetIds: z.array(z.string()).max(16).optional().describe("Project assets as references — UUIDs or badge codes (\"IMG-A8\"), resolved at call time. Requires projectId. Caps: GPT Image 16, nano-banana-2 14, FLUX/Seedream lower. Label every reference role in the prompt."),
1788
2225
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
1789
2226
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
2227
+ folderId: z.string().uuid().nullable().optional().describe(`${FOLDER_DESCRIBE} Needs projectId.`),
1790
2228
  }),
1791
2229
  async run(input, ctx) {
2230
+ // A desktop before the folder rule ignores folderId and files at the root.
2231
+ if (input.folderId !== undefined)
2232
+ await ctx.desktop().requireCapability('media-folder', 'choosing the folder a result lands in');
1792
2233
  // Clarification gate: aspectRatio + resolution must be deliberate.
1793
2234
  // Mirrors the cost confirm gate — defaults silently wasted credits
1794
2235
  // (1:1 when user wanted 16:9, 4k when 1k would have done). The LLM
@@ -1970,6 +2411,7 @@ export const generateImage = {
1970
2411
  count: input.count ?? 1,
1971
2412
  ...(isGptImageModel(imageModel) ? { gptQuality: input.quality ?? DEFAULT_GPT_QUALITY, gptBackground: input.backgroundMode } : {}),
1972
2413
  ...(referenceAssetIds.length > 0 ? { referenceAssetIds } : {}),
2414
+ ...folderBody(input.folderId),
1973
2415
  background: input.background,
1974
2416
  });
1975
2417
  // Partial multi-image failure: the desktop attaches an error message
@@ -2043,7 +2485,7 @@ export const generateImage = {
2043
2485
  };
2044
2486
  }
2045
2487
  // /proxy/generate kicks off a credit-aware job and returns a jobId
2046
- // for fal/Veo (async providers). We poll /proxy/jobs/{jobId} until
2488
+ // for fal (async providers). We poll /proxy/jobs/{jobId} until
2047
2489
  // the status is `completed` or `failed`, then fetch each image URL
2048
2490
  // and inline as base64 so the calling LLM sees the pixels.
2049
2491
  // The fal endpoint differs by mode: bare model id for text-to-image,
@@ -2156,7 +2598,7 @@ const LEGACY_EDIT_REFERENCE_MODELS = ['nano-banana-2', 'nano-banana-2-lite', 'na
2156
2598
  export const editImage = {
2157
2599
  id: 'slates_edit_image',
2158
2600
  billable: true,
2159
- description: 'Surgically edit an image asset with a text instruction (e.g. \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The result is a NEW asset (prompt prefixed \'[Edit]\'); the source is untouched. Default model ' + toolModelFor('image-edit') + ' (the image-edit tool seat, the model the app\'s Edit box opens on). Every model also takes referenceAssetIds, up to its reference cap less one: the source is image 1 of the request. Before first use call slates_get_prompting_guide with topic \'slates-edit-and-iterate\'.',
2601
+ description: 'Surgically edit an image asset with a text instruction (e.g. \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The result is a NEW asset (prompt prefixed \'[Edit]\'); the source is untouched. Default model ' + toolModelFor('image-edit') + ' (the image-edit tool seat, the model the app\'s Edit box opens on). Every model also takes referenceAssetIds, up to its reference cap less one: the source is image 1 of the request. Use slates-edit-and-iterate for missing edit craft; reuse current guidance already in context.',
2160
2602
  input: z.object({
2161
2603
  projectId: z.string().uuid(),
2162
2604
  sourceAssetId: z.string().uuid().describe('Image asset to edit. Must exist in the project.'),
@@ -2282,6 +2724,105 @@ export const editImage = {
2282
2724
  };
2283
2725
  },
2284
2726
  };
2727
+ /**
2728
+ * The picture viewer's EXTRACT (its Cells tab): a 2x2 or 3x3 grid picture, cells cropped out and re-rendered
2729
+ * at full resolution, each as its own new picture. BILLABLE, and priced by the DESKTOP: its quote is the
2730
+ * `imageCost` the viewer's Extract button prints, for this seat, rung and cell count, so the price an agent
2731
+ * is shown cannot differ from the button's. The confirm gate is every billable op's: above CONFIRM_CREDITS
2732
+ * with no `confirm`, the op answers with the price and the exact prompt, and spends nothing.
2733
+ */
2734
+ /** The seats the viewer's Cells box offers. Mirrors slate's `GRID_EXTRACT_MODELS` (`shared/prompts/storyboard-grids.ts`), which this package cannot import. */
2735
+ const GRID_EXTRACT_SEATS = ['nano-banana-2', 'flux-2-max', 'seedream-5-lite'];
2736
+ export const extractGridCells = {
2737
+ id: 'slates_extract_grid_cells',
2738
+ billable: true,
2739
+ description: "Pull cells out of a 2x2 or 3x3 grid picture, as the picture viewer's Cells tab does: each named cell is cropped and re-rendered at full resolution as its own NEW picture (the grid is untouched). Name cells as they read on screen, row number then column letter: 1A top-left, 2C second row third column. Default seat " +
2740
+ toolModelFor('grid-extract') +
2741
+ " (the grid-extract tool seat, what the viewer opens on); resolution and aspect ratio default as the viewer does (the seat's own rung, the ratio nearest the grid's shape). prompt adds words to every cell; on the default seat @name and #look in it attach their pictures, and referenceAssetIds adds pictures, as in the viewer. " +
2742
+ CONFIRM_GATE_SENTENCE +
2743
+ " The confirm answer carries the price the viewer's Extract button prints and the exact prompt that will be sent.",
2744
+ input: z.object({
2745
+ projectId: z.string().uuid(),
2746
+ assetId: z.string().min(1).describe('The grid picture — UUID or badge code. Must be a 2x2 or 3x3 grid.'),
2747
+ cells: z.array(z.string().regex(/^[1-3][A-Ca-c]$/, 'a cell is its row number then column letter, e.g. 1A')).min(1).max(9).describe('Cells to extract, e.g. ["1A","2C"].'),
2748
+ model: zEnum(GRID_EXTRACT_SEATS).optional(),
2749
+ resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe("Default and ceiling are the seat's own, as in the viewer."),
2750
+ aspectRatio: z.string().optional().describe("Output aspect ratio, one the seat offers. Default: the ratio nearest the grid's own shape."),
2751
+ prompt: z.string().max(2500).optional().describe('Words for every cell (what the shot is). On the default seat @name and #look attach their pictures; the crop is image 1.'),
2752
+ referenceAssetIds: z.array(z.string()).max(13).optional().describe('Pictures added beside the crop (default seat only), UUIDs or badge codes.'),
2753
+ confirm: z.boolean().optional().describe('Set true to bypass the confirm gate after explicit user OK.'),
2754
+ background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2755
+ }),
2756
+ async run(input, ctx) {
2757
+ const desktop = ctx.desktop();
2758
+ await desktop.requireCapability('grid-extract', 'extracting grid cells');
2759
+ if (input.background)
2760
+ await desktop.requireCapability('background-generation', 'background generation');
2761
+ const refs = await resolveAssetRefs(ctx, input.projectId, [input.assetId, ...(input.referenceAssetIds ?? [])]);
2762
+ const grid = refs.get(input.assetId);
2763
+ const request = {
2764
+ projectId: input.projectId,
2765
+ assetId: grid.id,
2766
+ cells: input.cells.map((c) => c.toUpperCase()),
2767
+ model: input.model,
2768
+ resolution: input.resolution,
2769
+ aspectRatio: input.aspectRatio,
2770
+ prompt: input.prompt,
2771
+ referenceAssetIds: input.referenceAssetIds?.map((r) => refs.get(r).id),
2772
+ };
2773
+ // The desktop prices it: the viewer's own `imageCost`, for this seat, rung and cell count. The desktop also
2774
+ // refuses what the viewer would (not a grid, a cell the grid lacks, a seat it does not offer) before any spend.
2775
+ const quote = await desktop.get('/agent/generation/extract-quote', { request: JSON.stringify(request) });
2776
+ const gridName = grid.code ?? grid.id;
2777
+ if (quote.credits > CONFIRM_CREDITS && !input.confirm) {
2778
+ return ok({
2779
+ requires_confirm: true,
2780
+ estimated_credits: quote.credits,
2781
+ estimated_cents: quote.credits,
2782
+ cells: quote.cells,
2783
+ model: quote.model,
2784
+ resolution: quote.resolution,
2785
+ aspectRatio: quote.aspectRatio,
2786
+ credits_per_cell: quote.creditsPerCell,
2787
+ prompt_sent: quote.prompt,
2788
+ ...(quote.notes.length ? { notes: quote.notes } : {}),
2789
+ message: `Cost: ${fmtCredits(quote.credits)} to extract ${quote.cells.length} cell${quote.cells.length === 1 ? '' : 's'} (${quote.cells.join(', ')}) of ${gridName} with ${quote.modelLabel} at ${quote.resolution}, ${fmtCredits(quote.creditsPerCell)} each. ` +
2790
+ (quote.notes.length ? `${quote.notes.join('. ')}. ` : '') +
2791
+ `Re-call with confirm=true after the user explicitly OKs the spend. When discussing with the user, refer to the grid by its code (matches the gallery badge).`,
2792
+ });
2793
+ }
2794
+ const result = await desktop.post('/agent/generation/extract-cells', { ...request, background: input.background });
2795
+ if (result.background) {
2796
+ return backgroundSubmitted(`extraction of ${quote.cells.join(', ')} from ${gridName}`, result.generationIds ?? [], {
2797
+ projectId: input.projectId,
2798
+ gridAssetId: grid.id,
2799
+ cells: quote.cells,
2800
+ model: quote.model,
2801
+ cost_credits: quote.credits,
2802
+ });
2803
+ }
2804
+ const assets = result.assets ?? [];
2805
+ if (assets.length === 0)
2806
+ throw new Error(result.error ?? 'Extraction failed');
2807
+ const codes = assets.map((a) => a.code ?? a.id).join(', ');
2808
+ const partial = result.success === false ? ` Only ${assets.length} of ${quote.cells.length} cells came back; the rest failed (slates_list_generations shows why).` : '';
2809
+ return {
2810
+ text: `Extracted ${assets.length} cell${assets.length === 1 ? '' : 's'} (${quote.cells.join(', ')}) of ${gridName} as ${codes} via ${quote.modelLabel} at ${quote.resolution}, ${fmtCredits(quote.credits)}.${partial} ` +
2811
+ BACKGROUND_REVIEW_POINTER,
2812
+ data: {
2813
+ projectId: input.projectId,
2814
+ gridAssetId: grid.id,
2815
+ cells: quote.cells,
2816
+ model: quote.model,
2817
+ resolution: quote.resolution,
2818
+ aspectRatio: quote.aspectRatio,
2819
+ cost_credits: quote.credits,
2820
+ assets,
2821
+ generationIds: result.generationIds,
2822
+ },
2823
+ };
2824
+ },
2825
+ };
2285
2826
  // ── Generate video ──────────────────────────────────────────────
2286
2827
  //
2287
2828
  // VIDEO_MODELS and the capability-derived param vocabulary are declared near the
@@ -2301,10 +2842,9 @@ export const editImage = {
2301
2842
  * same teaching `requires_clarification` shape as every other gate in this op,
2302
2843
  * rather than a raw Zod error the agent has to guess its way out of.
2303
2844
  *
2304
- * `promptMode` matters: Veo's reference-to-video endpoint is 8s only, declared
2305
- * as `duration.modeOverrides.ingredients`. Free reference images with no
2306
- * first/last frame IS ingredients mode — the same condition
2307
- * `buildFalVeoRequest` uses to pick the ref2v endpoint.
2845
+ * `promptMode` matters for any row that declares `duration.modeOverrides.ingredients`
2846
+ * (retired Veo's reference-to-video endpoint was 8s only). Free reference images
2847
+ * with no first/last frame IS ingredients mode.
2308
2848
  */
2309
2849
  function assertVideoCapabilities(input) {
2310
2850
  const freeRefs = (input.ingredientAssetIds?.length ?? 0) +
@@ -2333,7 +2873,6 @@ function assertVideoCapabilities(input) {
2333
2873
  // shape (verified against /api/agent/models):
2334
2874
  // Kling: kling-v3-{standard|pro|omni}-{N}s — note the user-facing
2335
2875
  // model id `kling-v3.0-std` maps to registry key `kling-v3-standard`.
2336
- // Veo: veo-3.1-{fast|standard}[-4k]-{N}s[-audio]
2337
2876
  // Seedance: seedance-2-{res}-{N}s (BytePlus ModelArk, sole provider). The
2338
2877
  // cost key encodes resolution (480p/720p/1080p/4k) — price scales with
2339
2878
  // resolution, so the key MUST carry it or the pre-flight quote is wrong.
@@ -2399,29 +2938,21 @@ export function videoCostKey(input) {
2399
2938
  }
2400
2939
  return `${input.model}${face}-${res}-${input.duration}s`;
2401
2940
  }
2402
- if (input.model.startsWith('veo')) {
2403
- const is4k = input.videoResolution === '4k';
2404
- const audio = input.sound !== false; // default audio on for Veo
2405
- const parts = [input.model];
2406
- if (is4k)
2407
- parts.push('4k');
2408
- parts.push(`${input.duration}s`);
2409
- if (audio)
2410
- parts.push('audio');
2411
- return parts.join('-');
2412
- }
2413
2941
  if (input.model.startsWith('kling-v3.0')) {
2414
2942
  // Mirrors klingCreditKey() in slate/src/shared/pricing.ts. Kling native 4K
2415
2943
  // bills flat-rate keys: std/pro/omni all get a `-4k` tier key, and omni-pro
2416
2944
  // shares kling-v3-omni-4k (the o3/4k endpoint has one flat rate, audio
2417
2945
  // included). At 1080p AUDIO IS A KEY DIMENSION (credits = COGS × markup,
2418
- // locked 2026-07-05): sound → `-audio` variant. Kling sound defaults OFF.
2946
+ // locked 2026-07-05): sound → `-audio` variant. An OMITTED sound is ON: the
2947
+ // desktop agent route sends `sound ?? true` and bills the audio key, so a
2948
+ // quote that read omitted as silent under-quoted every default Kling take
2949
+ // (21 credits quoted, 32 billed at std 5s; found 2026-10-02).
2419
2950
  const tier = KLING_TIER_MAP[input.model] ?? input.model;
2420
2951
  if (input.videoResolution === '4k') {
2421
2952
  const tier4k = tier === 'kling-v3-omni-pro' ? 'kling-v3-omni' : tier;
2422
2953
  return `${tier4k}-4k-${input.duration}s`;
2423
2954
  }
2424
- return `${tier}-${input.duration}s${input.sound === true ? '-audio' : ''}`;
2955
+ return `${tier}-${input.duration}s${input.sound !== false ? '-audio' : ''}`;
2425
2956
  }
2426
2957
  if (input.model === 'omni-flash') {
2427
2958
  // Mirrors omniFlashCreditKey() in slate/src/shared/pricing.ts — flat 720p
@@ -2546,13 +3077,19 @@ function resolveVideoModel(raw) {
2546
3077
  // so a pasted `minimax-h3-768p-10s-ref2` resolves instead of erroring. The
2547
3078
  // number it carries is K (images PAST the free five), so it is converted back
2548
3079
  // to a TOTAL before anything can re-surcharge it.
3080
+ // The free allowance differs per row (H3 five, H3 Max four), so the total is
3081
+ // computed once the model is known, in `withRefTotal`; reading `out.model`
3082
+ // here read the placeholder and gave every row five.
2549
3083
  const ref = /-ref(\d+)\b/.exec(s);
2550
- if (ref) {
2551
- out.referenceImages =
2552
- (MINIMAX_FREE_REF_IMAGES_BY_MODEL[out.model] ?? MINIMAX_FREE_REF_IMAGES) +
2553
- parseInt(ref[1], 10);
3084
+ const paidRefs = ref ? parseInt(ref[1], 10) : null;
3085
+ if (ref)
2554
3086
  s = s.replace(/-ref(\d+)\b/, '');
2555
- }
3087
+ const withRefTotal = (resolved) => {
3088
+ if (paidRefs !== null) {
3089
+ resolved.referenceImages = (MINIMAX_FREE_REF_IMAGES_BY_MODEL[resolved.model] ?? MINIMAX_FREE_REF_IMAGES) + paidRefs;
3090
+ }
3091
+ return resolved;
3092
+ };
2556
3093
  // The RESOLUTION vocabulary is GENERATED from MODEL_CAPABILITIES — the
2557
3094
  // hand-typed list that stood here went stale the day 768p and 2k shipped.
2558
3095
  const resRe = new RegExp(`-(${VIDEO_RESOLUTION_VOCAB.join('|')})\\b`);
@@ -2578,7 +3115,7 @@ function resolveVideoModel(raw) {
2578
3115
  const direct = VIDEO_MODELS.find((m) => m === s);
2579
3116
  if (direct) {
2580
3117
  out.model = direct;
2581
- return out;
3118
+ return withRefTotal(out);
2582
3119
  }
2583
3120
  const aliases = {
2584
3121
  'kling-v3-standard': 'kling-v3.0-std',
@@ -2600,8 +3137,6 @@ function resolveVideoModel(raw) {
2600
3137
  'seedance-2-5': 'seedance-2.5',
2601
3138
  'seedance2.5': 'seedance-2.5',
2602
3139
  seedance: 'seedance-2',
2603
- 'veo-3.1': 'veo-3.1-fast',
2604
- 'veo-3': 'veo-3.1-fast',
2605
3140
  'gemini-omni-flash': 'omni-flash',
2606
3141
  'gemini-omni-flash-preview': 'omni-flash',
2607
3142
  'omni-flash-preview': 'omni-flash',
@@ -2626,7 +3161,7 @@ function resolveVideoModel(raw) {
2626
3161
  };
2627
3162
  if (aliases[s]) {
2628
3163
  out.model = aliases[s];
2629
- return out;
3164
+ return withRefTotal(out);
2630
3165
  }
2631
3166
  return null;
2632
3167
  }
@@ -2682,11 +3217,11 @@ const VIDEO_MODEL_GUIDES = [
2682
3217
  export const generateVideo = {
2683
3218
  id: 'slates_generate_video',
2684
3219
  billable: true,
2685
- description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (' +
3220
+ description: 'Generate video via Slates credits. Choose the model with slates-model-selection. Video models prompt very differently: the estimate returns the chosen model\'s prompting card, and its full guide (' +
2686
3221
  VIDEO_MODEL_GUIDES +
2687
- ') — video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). ' +
3222
+ ') covers modes the card leaves out, via slates_get_prompting_guide. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). ' +
2688
3223
  CONFIRM_GATE_SENTENCE +
2689
- ' Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8"). ' +
3224
+ ' Image-to-video via firstFrameAssetId; first+last frames on every model except Omni Flash; ingredients via ingredientAssetIds (Kling Omni, Seedance, Omni Flash, H3 and H3 Max; Kling Std/Pro only with a first frame). Asset params take UUIDs or badge codes ("IMG-A8"). ' +
2690
3225
  // GENERATED from the skill's own slop-token list. Always in context on both
2691
3226
  // surfaces, so it survives an agent that skips slates_get_prompting_guide.
2692
3227
  describeBannedTokens('video'),
@@ -2706,7 +3241,7 @@ export const generateVideo = {
2706
3241
  `Full table: the slates-model-selection skill. Each model's legal aspect ratios, durations and ` +
2707
3242
  `resolutions are in those params' own descriptions — read them there, not from memory. ` +
2708
3243
  `For per-call cost, call slates_estimate_generation_cost.`),
2709
- projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended — the desktop UI shows a progress card live and the asset appears when complete.'),
3244
+ projectId: z.string().uuid().optional().describe('Save into this Slates project. Required — the desktop UI shows a progress card live and the asset appears when complete.'),
2710
3245
  // 🚨 THESE THREE DESCRIPTIONS ARE GENERATED FROM `MODEL_CAPABILITIES`.
2711
3246
  // Never hand-write a ratio, resolution or duration into them again — every
2712
3247
  // one of the hand-written claims that stood here had drifted, and an
@@ -2723,7 +3258,7 @@ export const generateVideo = {
2723
3258
  // demand, by the one session that needs it. If you are about to explain
2724
3259
  // WHY here, you are writing the skill in the wrong file.
2725
3260
  firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video (UUID or badge code, resolved at call time).'),
2726
- lastFrameAssetId: z.string().optional().describe('Ending frame. Veo and Seedance only; pairs with firstFrameAssetId.'),
3261
+ lastFrameAssetId: z.string().optional().describe('Ending frame; every model except Omni Flash. Pairs with firstFrameAssetId.'),
2727
3262
  ingredientAssetIds: z.array(z.string()).max(30).optional().describe(
2728
3263
  // Caps DERIVED from MODEL_CAPABILITIES — the hand-typed list omitted
2729
3264
  // kling-v3.0-omni-pro entirely and read as if 7 were an Omni Flash-only rule.
@@ -2738,21 +3273,25 @@ export const generateVideo = {
2738
3273
  // The capacity sentences are DERIVED from MODEL_FACTS (see
2739
3274
  // multimodalRefSummary) rather than hand-typed, so a cap change in one
2740
3275
  // place cannot leave a stale number in a description an LLM reads.
2741
- videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS, cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; per-model caps are on audioReferenceAssetIds. Billing switches to the vref key (input+output seconds) — pass videoReferenceSecondsEach. Over the cap is REFUSED, never trimmed.`),
3276
+ videoReferenceAssetIds: z.array(z.string()).optional().describe(`Reference VIDEOS, cited in the prompt as "video 1", "video 2"… in the order given. ${multimodalRefModels().join(' / ')} only; per-model caps are on audioReferenceAssetIds. On Seedance, billing switches to the vref key (input+output seconds) — pass videoReferenceSecondsEach. Over the cap is REFUSED, never trimmed.`),
2742
3277
  videoReferenceSecondsEach: z.array(z.number()).optional().describe('REQUIRED with videoReferenceAssetIds, same order and length: each clip\'s duration in seconds. Feeds the vref cost key; the server re-probes and corrects an understated value upward.'),
2743
- audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO, cited as "audio 1", "audio 2"… in the order given. ${multimodalRefModels().join(' / ')} only. No billing surcharge. ${multimodalRefModels().map(multimodalRefSummary).join(' ')}`),
3278
+ audioReferenceAssetIds: z.array(z.string()).optional().describe(`Reference AUDIO, cited as "audio 1", "audio 2"… in the order given. ${multimodalRefModels().join(' / ')} only. No surcharge on Seedance; on H3 Max, audio counts toward the reference-token pool. ${multimodalRefModels().map(multimodalRefSummary).join(' ')}`),
2744
3279
  audioReferenceSpokenText: z.array(z.string()).optional().describe('The exact words in each reference clip — same order and length as audioReferenceAssetIds, "" for a clip with no speech. The model RE-TRANSCRIBES a take rather than using it verbatim, so audio decides voice/accent/timing and only this decides the WORDS. Omit it and the words are a guess.'),
2745
- sound: z.boolean().optional().describe('Kling Omni / Veo / Seedance: Sound on (true) or Silent (false). Default true.'),
2746
- audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Kling Omni only — language for dialogue.'),
2747
- generateMusic: z.boolean().optional().describe('Kling Omni only — auto-generate background music.'),
3280
+ sound: z.boolean().optional().describe('Kling (every tier) and LTX: sound on (default) or silent (false); Kling bills audio as its own key. Seedance, Omni Flash and H3 always generate audio.'),
3281
+ audioLanguage: z.enum(['EN', 'ZH', 'JA', 'KO', 'ES']).optional().describe('Any Kling model with sound on — language for dialogue.'),
3282
+ generateMusic: z.boolean().optional().describe('Any Kling model with sound on — auto-generate background music.'),
2748
3283
  seedanceFace: z.boolean().optional().describe('Seedance ONLY: a reference shows an AI CHARACTER\'s face. Faces are blocked on the default route, so this reroutes to a face-capable provider at ~45% more. A REAL person fails here with [REAL_FACE_DETECTED] — see seedanceRealFace.'),
2749
- seedanceRealFace: z.boolean().optional().describe('Seedance ONLY: a reference shows a REAL person. Real-person route, roughly 2x the AI-face price — quote it first. REQUIRES realFaceConsent=true.'),
3284
+ seedanceRealFace: z.boolean().optional().describe('Seedance ONLY: a reference shows a REAL person. Real-person route, about 1.4x the AI-face price (about 2x faceless) — quote it first. REQUIRES realFaceConsent=true.'),
2750
3285
  realFaceConsent: z.boolean().optional().describe('MANDATORY with seedanceRealFace: true ONLY after the user has explicitly confirmed they hold rights/consent to this likeness and it does not impersonate or misrepresent them. Refused without it; public figures fail on every route.'),
2751
3286
  negativePrompt: z.string().optional(),
2752
3287
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
2753
3288
  confirm: z.boolean().optional().describe('Set true after explicit user OK to bypass the confirm gate (which fires for almost every video gen since they\'re expensive).'),
3289
+ folderId: folderIdField,
2754
3290
  }),
2755
3291
  async run(input, ctx) {
3292
+ // A desktop before the folder rule ignores folderId and files at the root.
3293
+ if (input.folderId !== undefined)
3294
+ await ctx.desktop().requireCapability('media-folder', 'choosing the folder a result lands in');
2756
3295
  // Resolve the model FIRST — forgiving normalization (cost keys, alias
2757
3296
  // spellings) with a teaching error, so a wrong id never costs the agent
2758
3297
  // a retry spiral. Anything the key encoded (duration/res/audio) fills
@@ -2832,7 +3371,7 @@ export const generateVideo = {
2832
3371
  return ok({
2833
3372
  requires_clarification: true,
2834
3373
  missing: [],
2835
- message: `Omni Flash takes a prompt, an optional start frame, and up to ${omniFlashRefCap} reference IMAGES — last frames and video/audio references are not supported. Drop those, or switch to seedance-2 (video/audio refs, last frame) or veo (last frame).`,
3374
+ message: `Omni Flash takes a prompt, an optional start frame, and up to ${omniFlashRefCap} reference IMAGES — last frames and video/audio references are not supported. Drop those, or switch to seedance-2.5 (video/audio refs, last frame).`,
2836
3375
  });
2837
3376
  }
2838
3377
  const refCount = (input.ingredientAssetIds?.length ?? 0) +
@@ -3206,6 +3745,7 @@ export const generateVideo = {
3206
3745
  seedanceRealFace: input.seedanceRealFace,
3207
3746
  realFaceConsent: input.realFaceConsent,
3208
3747
  negativePrompt: input.negativePrompt,
3748
+ ...folderBody(input.folderId),
3209
3749
  background: input.background,
3210
3750
  });
3211
3751
  if (!result.success) {
@@ -3309,10 +3849,9 @@ export const generateAudio = {
3309
3849
  id: 'slates_generate_audio',
3310
3850
  billable: true,
3311
3851
  description: `Generate project audio using credits. Choose the surface via the model routing below. ` +
3312
- 'Read slates-cost-discipline and the matching prompting skill first (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts). ' +
3852
+ 'The estimate returns the chosen surface\'s prompting card; its full guide (slates-prompting-seed-audio | slates-prompting-elevenlabs | slates-prompting-inworld-tts) covers what the card leaves out. ' +
3313
3853
  'Seed Audio bills the requested duration, which is appended to the prompt regardless of output length. Kling "SFX:" / "Ambient noise:" syntax does not transfer. ' +
3314
- CONFIRM_GATE_SENTENCE +
3315
- ' No skill files installed? Call slates_get_prompting_guide first.',
3854
+ CONFIRM_GATE_SENTENCE,
3316
3855
  input: z.object({
3317
3856
  projectId: z.string().uuid().describe('Slates project the audio asset lands in. Required — the renderer refreshes live.'),
3318
3857
  model: z
@@ -3365,8 +3904,12 @@ export const generateAudio = {
3365
3904
  .describe('seed-audio only — ONE image asset to score what is in frame. MUTUALLY EXCLUSIVE with audioReferenceAssetIds.'),
3366
3905
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
3367
3906
  confirm: z.boolean().optional().describe('Set true to bypass the confirm gate after explicit user OK.'),
3907
+ folderId: folderIdField,
3368
3908
  }),
3369
3909
  run: async (input, ctx) => {
3910
+ // A desktop before the folder rule ignores folderId and files at the root.
3911
+ if (input.folderId !== undefined)
3912
+ await ctx.desktop().requireCapability('media-folder', 'choosing the folder a result lands in');
3370
3913
  // ── Per-surface clarification + constraint gates ──
3371
3914
  //
3372
3915
  // `null` for the TTS seat is load-bearing rather than a placeholder: that
@@ -3500,6 +4043,7 @@ export const generateAudio = {
3500
4043
  promptInfluence: input.promptInfluence,
3501
4044
  audioReferenceAssetIds: (input.audioReferenceAssetIds ?? []).map((r) => rid(r)),
3502
4045
  imageReferenceAssetId: rid(input.imageReferenceAssetId),
4046
+ ...folderBody(input.folderId),
3503
4047
  background: input.background,
3504
4048
  });
3505
4049
  if (!result.success)
@@ -3559,7 +4103,8 @@ async function quoteToolBlocks(ctx, stem, seconds, voiceStep) {
3559
4103
  export const generateLipSync = {
3560
4104
  id: 'slates_generate_lip_sync',
3561
4105
  billable: true,
3562
- description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip, sourceType=image animates a still avatar (avatar-standard, or avatar-pro for the premium seat). Quote each with slates_estimate_generation_cost rather than from memory. Audio from TTS (ttsText + ttsVoice) or an uploaded file. Billed per 5s of the output (the clip, or on a still the voice track). For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-lip-sync skills. projectId is REQUIRED.',
4106
+ description: 'Lip-sync a still image (avatar) or a video clip to audio. KLING-ONLY — this tool wraps Kling\'s dedicated lip-sync endpoints and nothing else: sourceType=video re-syncs a clip, sourceType=image animates a still avatar (avatar-standard, or avatar-pro for the premium seat). Quote each with slates_estimate_generation_cost rather than from memory. Audio from TTS (ttsText + ttsVoice) or an uploaded file. Billed per 5s of the output (the clip, or on a still the voice track). For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the clip attached as a video reference and the dialogue written into the prompt; that is the same call, with the prompt visible and editable. The craft is slates-prompting-lip-sync; the confirm response shows the exact cost before anything is charged. projectId is REQUIRED. ' +
4107
+ CONFIRM_GATE_SENTENCE,
3563
4108
  input: z.object({
3564
4109
  projectId: z.string().uuid().describe('Slates project the source asset lives in. The new lip-synced video lands here.'),
3565
4110
  sourceAssetId: z.string().uuid().describe('Asset id of the still image (avatar flow) or video clip (lip-sync flow). Must already exist in the project — use slates_upload_reference_image or slates_generate_image / slates_generate_video first if needed.'),
@@ -3682,11 +4227,11 @@ export const generateLipSync = {
3682
4227
  export const generateMotionTransfer = {
3683
4228
  id: 'slates_generate_motion_transfer',
3684
4229
  billable: true,
3685
- description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std or kling-mc-pro, structured skeleton/depth retargeting, billed per 5s of the driving clip. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. REQUIRED before calling: slates-cost-discipline + slates-prompting-motion-transfer skills. projectId is REQUIRED — both assets must exist in the project. ' +
4230
+ description: 'Transfer the motion from a reference video onto a target image character. KLING-ONLY — this tool wraps Kling Motion Control and nothing else: kling-mc-std or kling-mc-pro, structured skeleton/depth retargeting, billed per 5s of the driving clip. For a Seedance version, do NOT look for an engine switch here — run a normal slates_generate_video on seedance-2 with the driving clip attached as a video reference and the motion described in the prompt ("the character from image 1 performs the exact motion from video 1"); that is the same call, with the prompt visible and editable. The craft is slates-prompting-motion-transfer; the confirm response shows the exact cost before anything is charged. projectId is REQUIRED — both assets must exist in the project. ' +
3686
4231
  CONFIRM_GATE_SENTENCE,
3687
4232
  input: z.object({
3688
4233
  projectId: z.string().uuid().describe('Slates project. Both source and target assets must live here.'),
3689
- sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s.'),
4234
+ sourceVideoAssetId: z.string().uuid().describe('Asset id of the reference video — its motion will be retargeted onto the target image. Must already exist in the project. Up to 30s with characterOrientation video, 10s with image; billed per 5s block.'),
3690
4235
  targetImageAssetId: z.string().uuid().describe('Asset id of the target image (the character that will perform the motion). Must already exist in the project.'),
3691
4236
  motionModel: z.enum(['kling-mc-std', 'kling-mc-pro']).optional().describe('kling-mc-std general motion; kling-mc-pro cleaner anatomy — default. Quote both with slates_estimate_generation_cost.'),
3692
4237
  characterOrientation: z.enum(['video', 'image']).optional().describe('"video" = use the source video\'s framing. "image" = use the target image\'s framing. Default video.'),
@@ -3793,7 +4338,7 @@ export const editVideo = {
3793
4338
  // point to in a tool result) and go stale on the next rate change; the
3794
4339
  // windows are owned by MODEL_CAPABILITIES and are generated below into the
3795
4340
  // params that enforce them.
3796
- 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default — subject/style refs via elements), omni-flash-edit (PROMPT-ONLY, no refs, the cheapest seat), or seedance-2.5-edit (the only engine that takes a clip over 15s; seedanceFace:true for AI-character faces). Clip-length and resolution windows per engine are on the `model` param. Cost is per second of OUTPUT (≈ clip length, rounded up), and an edit bills input + output seconds on every provider — read the quote from the confirm gate or slates_estimate_generation_cost, never from memory. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; Seedance edit/relocate for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
4341
+ 'Edit an EXISTING video clip with one instruction — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default — subject/style refs via elements), omni-flash-edit (PROMPT-ONLY, no refs, priced level with Kling O3 Edit Standard; the first pick in the routing guide for footage-synced edits), or seedance-2.5-edit (the only engine that takes a clip over 15s; seedanceFace:true for AI-character faces). Clip-length and resolution windows per engine are on the `model` param. Cost is per second of OUTPUT (≈ clip length, rounded up), and an edit bills input + output seconds on Seedance and output seconds on Kling and Omni Flash — read the quote from the confirm gate or slates_estimate_generation_cost, never from memory. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements — Kling models only); style refs as styleAssetIds; max 4 combined. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; Seedance edit for style-transfer-heavy jobs — see slates-model-selection. Prompting: slates-prompting-kling-v3 §Edit / slates-prompting-omni-flash.',
3797
4342
  input: z.object({
3798
4343
  projectId: z.string().uuid().describe('Project the source clip lives in.'),
3799
4344
  sourceVideoAssetId: z.string().describe('The VIDEO asset to edit — UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3–15s clips; omni-flash-edit: 3–10s.'),
@@ -3810,12 +4355,16 @@ export const editVideo = {
3810
4355
  styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY — rejected on omni-flash-edit.'),
3811
4356
  keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only — omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
3812
4357
  videoResolution: zEnum(EDIT_VIDEO_RESOLUTIONS).optional().describe(`seedance-2.5-edit ONLY (default ${defaultVideoResolutionFor('seedance-2.5-edit')}). Ignored by the Kling and Omni Flash engines, whose output follows the source clip.`),
3813
- seedanceFace: z.boolean().optional().describe('seedance-2.5-edit ONLY: set true when a CHARACTER FACE is visible in the clip. Faceless edits run on BytePlus; faces are blocked there and must route to the relaxed provider, which costs ~35% more. A face edit submitted without this flag is rejected by the provider, not silently downgraded.'),
4358
+ seedanceFace: z.boolean().optional().describe('seedance-2.5-edit ONLY: set true when a CHARACTER FACE is visible in the clip. Faceless edits run on BytePlus; faces are blocked there and must route to the relaxed provider, which costs about 40-50% more. A face edit submitted without this flag is rejected by the provider, not silently downgraded.'),
3814
4359
  background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
3815
4360
  confirm: z.boolean().optional().describe('Set true to bypass the cost confirm gate after the user OKs the spend.'),
4361
+ folderId: folderIdField,
3816
4362
  }),
3817
4363
  async run(input, ctx) {
3818
4364
  const desktop = ctx.desktop();
4365
+ // A desktop before the folder rule ignores folderId and files at the root.
4366
+ if (input.folderId !== undefined)
4367
+ await desktop.requireCapability('media-folder', 'choosing the folder a result lands in');
3819
4368
  await desktop.requireCapability('edit-video', 'video editing (Kling O3 edit)');
3820
4369
  if (input.background) {
3821
4370
  await desktop.requireCapability('background-generation', 'background generation');
@@ -3918,6 +4467,7 @@ export const editVideo = {
3918
4467
  keepAudio: input.keepAudio !== false,
3919
4468
  videoResolution: input.videoResolution,
3920
4469
  seedanceFace: input.seedanceFace,
4470
+ ...folderBody(input.folderId),
3921
4471
  background: input.background,
3922
4472
  });
3923
4473
  if (!result.success)
@@ -3955,7 +4505,7 @@ export const editVideo = {
3955
4505
  // ── Trim a video to an exact window (fit-to-model primitive) ────
3956
4506
  export const trimVideo = {
3957
4507
  id: 'slates_trim_video',
3958
- description: 'Trim a video asset to an exact [inSec, outSec] window and save the result as a NEW clip linked to the original (the original is untouched). This is the fit-to-model primitive: an 11s clip will not run on omni-flash-edit (3–10s) or Kling edit (3–15s), and a Seedance video reference must be 2–15s — trim it first, then edit/relocate the trimmed clip. EXACT re-encode (not a keyframe-snapped cut) so the result honors hard duration caps to the frame; any phone rotation flag is baked into the pixels in the same pass. The new clip lands with correct duration/width/height immediately. inSec defaults to 0.',
4508
+ description: 'Trim a video asset to an exact [inSec, outSec] window and save the result as a NEW clip linked to the original (the original is untouched). This is the fit-to-model primitive: an 11s clip will not run on omni-flash-edit (3–10s), a 16s one not on Kling edit (3–15s), and a Seedance 2.0 video reference must be 2–15s (2.5 takes up to 30s combined) — trim it first, then edit the trimmed clip. EXACT re-encode (not a keyframe-snapped cut) so the result honors hard duration caps to the frame; any phone rotation flag is baked into the pixels in the same pass. The new clip lands with correct duration/width/height immediately. inSec defaults to 0. Pass pieces to cut the window into SEVERAL clips in one call, as the Trim & split dialog does (split points, or Auto-split by longest piece, with seconds shared between neighbours); every new clip comes back.',
3959
4509
  input: z.object({
3960
4510
  projectId: z.string().uuid().describe('Project the clip lives in.'),
3961
4511
  assetId: z
@@ -3963,6 +4513,15 @@ export const trimVideo = {
3963
4513
  .describe('The VIDEO asset to trim — UUID or badge code ("VID-V3", bare "V3"); resolves against the project at call time.'),
3964
4514
  inSec: z.number().min(0).optional().describe('Trim start in seconds (default 0).'),
3965
4515
  outSec: z.number().positive().describe('Trim end in seconds. Must be greater than inSec.'),
4516
+ pieces: z
4517
+ .object({
4518
+ splitsAtSec: z.array(z.number().positive()).max(60).optional().describe('Cut the window at these seconds of the clip (Split here): each at least 0.15 s inside inSec and outSec and 0.2 s from the next.'),
4519
+ longestSec: z.number().min(1).max(60).optional().describe('Auto-split: even pieces, none longer than this (Longest piece). Use instead of splitsAtSec.'),
4520
+ shareSec: z.number().min(0).max(5).optional().describe('Pieces share: each piece also holds this many seconds of the next (default 0), so an edit can carry on into the next clip.'),
4521
+ })
4522
+ .strict()
4523
+ .optional()
4524
+ .describe('Cut the window into several clips instead of one. Needs splitsAtSec or longestSec.'),
3966
4525
  }),
3967
4526
  async run(input, ctx) {
3968
4527
  const desktop = ctx.desktop();
@@ -3972,6 +4531,24 @@ export const trimVideo = {
3972
4531
  if (input.outSec - inSec < 0.05) {
3973
4532
  throw new Error('outSec must be at least ~0.1s after inSec.');
3974
4533
  }
4534
+ if (input.pieces) {
4535
+ // 1.6.0 would cut the one window and ignore pieces without a word.
4536
+ await desktop.requireCapability('trim-pieces', 'cutting a clip into pieces');
4537
+ if (input.pieces.splitsAtSec === undefined && input.pieces.longestSec === undefined) {
4538
+ throw new Error('pieces needs splitsAtSec (where to cut) or longestSec (Auto-split by longest piece).');
4539
+ }
4540
+ const cut = await desktop.post('/agent/assets/trim-video', {
4541
+ assetId,
4542
+ inSec,
4543
+ outSec: input.outSec,
4544
+ pieces: input.pieces,
4545
+ });
4546
+ const named = cut.assets.map((a, i) => `${a.code ?? a.id} (${cut.pieces[i].start.toFixed(1)}–${cut.pieces[i].end.toFixed(1)}s)`);
4547
+ return {
4548
+ text: `Cut ${cut.assets.length} clip${cut.assets.length === 1 ? '' : 's'} from the ${inSec.toFixed(1)}–${input.outSec.toFixed(1)}s window: ${named.join(', ')}. Edit or generate from them by id/code.`,
4549
+ data: { assets: cut.assets, pieces: cut.pieces },
4550
+ };
4551
+ }
3975
4552
  const r = await desktop.post('/agent/assets/trim-video', {
3976
4553
  assetId,
3977
4554
  inSec,
@@ -4284,7 +4861,7 @@ export const exportVideo = {
4284
4861
  };
4285
4862
  export const exportTimelineXml = {
4286
4863
  id: 'slates_export_timeline_xml',
4287
- description: "Export the project's timeline as FCP7/XMEML XML — the file DaVinci Resolve imports directly (File → Import → Timeline) to recreate the edit with references to the original clip media on disk. This is the 'Export for DaVinci, Premiere or Final Cut' path. Pass an absolute outputPath ending in .xml; fails if the file exists unless overwrite=true.",
4864
+ description: "Export the project's timeline as FCP7/XMEML XML — the file DaVinci Resolve imports directly (File → Import → Timeline) to recreate the edit with references to the original clip media on disk. Use this for DaVinci Resolve or Premiere; current Final Cut Pro requires FCPXML, which this tool does not export. Pass an absolute outputPath ending in .xml; fails if the file exists unless overwrite=true.",
4288
4865
  input: z
4289
4866
  .object({
4290
4867
  projectId: z.string().uuid().optional(),
@@ -4792,7 +5369,7 @@ export const updateFrame = {
4792
5369
  // of what the Shot in this slot already encodes — the image's role and the
4793
5370
  // beat's words — and they were backfilled into Shots on 2026-08-31. Use
4794
5371
  // slates_update_shot for either.
4795
- 'Update a slot: its shot label, notes, bound asset (assetId=null unbinds), scene or position. A scene or position change carries the words in this slot with it on the Script page. The BEAT — its line, references, model, prompt and framing — lives on the Shot in this slot; use slates_update_shot for that.',
5372
+ 'Update a slot: its shot label, notes, bound asset (assetId=null unbinds), scene or position. A scene or position change carries the words in this slot with it on the Script page. preferredClipId picks the clip the timeline takes for this slot (a video asset id or VID code; null goes back to the slot\'s most recent linked clip), as the take menu\'s "Use in the timeline" does. The BEAT — its line, references, model, prompt and framing — lives on the Shot in this slot; use slates_update_shot for that.',
4796
5373
  input: z.object({
4797
5374
  frameId: z.string().uuid(),
4798
5375
  shotLabel: z.string().optional(),
@@ -4800,8 +5377,12 @@ export const updateFrame = {
4800
5377
  assetId: z.string().uuid().nullable().optional(),
4801
5378
  sceneId: z.string().uuid().optional(),
4802
5379
  position: z.number().int().min(0).optional(),
5380
+ preferredClipId: z.string().min(1).nullable().optional().describe('The clip the timeline takes for this slot: a video asset id or badge code in the project; null clears it.'),
4803
5381
  }),
4804
5382
  async run(input, ctx) {
5383
+ // 1.6.0 takes the call and drops preferredClipId without a word.
5384
+ if (input.preferredClipId !== undefined)
5385
+ await ctx.desktop().requireCapability('frame-preferred-clip', 'choosing the clip a slot uses');
4805
5386
  return ok(await ctx.desktop().post('/agent/frames/update', {
4806
5387
  id: input.frameId,
4807
5388
  data: {
@@ -4810,6 +5391,7 @@ export const updateFrame = {
4810
5391
  assetId: input.assetId,
4811
5392
  sceneId: input.sceneId,
4812
5393
  position: input.position,
5394
+ preferredClipId: input.preferredClipId,
4813
5395
  },
4814
5396
  }));
4815
5397
  },
@@ -4829,7 +5411,7 @@ export const updateFrame = {
4829
5411
  */
4830
5412
  export const batchUpdateFrames = {
4831
5413
  id: 'slates_batch_update_frames',
4832
- description: 'Update MANY slots in one call — shot labels, notes, asset binding, scene/position. Prefer this over repeated slates_update_frame when re-arranging a scene: it is one round-trip and one UI refresh, and every id is validated before anything is written, so the batch never lands half-applied. A scene or position change carries the words in each slot with it on the Script page. To write the BEATS themselves, use slates_create_shot / slates_update_shot.',
5414
+ description: 'Update MANY slots in one call — shot labels, notes, asset binding, scene/position, and preferredClipId (the clip the timeline takes for the slot: a video asset id or VID code, null to clear). Prefer this over repeated slates_update_frame when re-arranging a scene: it is one round-trip and one UI refresh, and every id is validated before anything is written, so the batch never lands half-applied. A scene or position change carries the words in each slot with it on the Script page. To write the BEATS themselves, use slates_create_shot / slates_update_shot.',
4833
5415
  input: z.object({
4834
5416
  updates: z
4835
5417
  .array(z.object({
@@ -4839,10 +5421,14 @@ export const batchUpdateFrames = {
4839
5421
  assetId: z.string().uuid().nullable().optional(),
4840
5422
  sceneId: z.string().uuid().optional(),
4841
5423
  position: z.number().int().min(0).optional(),
5424
+ preferredClipId: z.string().min(1).nullable().optional().describe('A video asset id or badge code in the project; null clears it.'),
4842
5425
  }))
4843
5426
  .min(1),
4844
5427
  }),
4845
5428
  async run(input, ctx) {
5429
+ // 1.6.0 takes the call and drops preferredClipId without a word.
5430
+ if (input.updates.some((u) => u.preferredClipId !== undefined))
5431
+ await ctx.desktop().requireCapability('frame-preferred-clip', 'choosing the clip a slot uses');
4846
5432
  return ok(await ctx.desktop().post('/agent/frames/batch-update', {
4847
5433
  updates: input.updates.map((u) => ({
4848
5434
  id: u.frameId,
@@ -4852,6 +5438,7 @@ export const batchUpdateFrames = {
4852
5438
  assetId: u.assetId,
4853
5439
  sceneId: u.sceneId,
4854
5440
  position: u.position,
5441
+ preferredClipId: u.preferredClipId,
4855
5442
  },
4856
5443
  })),
4857
5444
  }));
@@ -5143,6 +5730,60 @@ async function buildShotSpecInput(ctx, projectId, input) {
5143
5730
  refEcho: describeResolvedRefs(refInputs, resolvedRefs),
5144
5731
  };
5145
5732
  }
5733
+ /**
5734
+ * The spec fields the caller NAMED, read out of the COMPLETE spec `buildShotSpecInput` built from the whole
5735
+ * input. Only those go to a route that merges a partial spec over a stored one (`slates_update_shot`, and
5736
+ * `slates_create_shot` from a result), so anything the caller omitted is left exactly as it was.
5737
+ */
5738
+ function namedSpecPatch(input, spec) {
5739
+ const patch = {};
5740
+ if (input.prompt !== undefined)
5741
+ patch.prompt = input.prompt;
5742
+ if (input.model !== undefined) {
5743
+ patch.model = input.model;
5744
+ // 🚨 `authoredFor` is NOT re-stamped on a model swap. The whole point of
5745
+ // recording it is that the prompt stays written for the model it was
5746
+ // written for — the grammars genuinely differ — so the card can say so.
5747
+ }
5748
+ if (input.params !== undefined)
5749
+ patch.params = spec.params;
5750
+ // 🚨 ONLY THE ROLES THE CALLER NAMED. `buildShotSpecInput` always returns a
5751
+ // COMPLETE refs record, and the route merges one level deep — so sending all
5752
+ // five would clear every role the caller never mentioned. Clearing a role is
5753
+ // explicit: send `[]`.
5754
+ if (input.refs !== undefined) {
5755
+ const built = spec.refs;
5756
+ const named = {};
5757
+ for (const role of ORDERED_ATTACHMENT_ROLES) {
5758
+ if (input.refs[role] !== undefined)
5759
+ named[role] = built[role];
5760
+ }
5761
+ if (Object.keys(named).length > 0)
5762
+ patch.refs = named;
5763
+ }
5764
+ if (input.firstFrameAssetId !== undefined)
5765
+ patch.firstFrameAssetId = spec.firstFrameAssetId;
5766
+ if (input.lastFrameAssetId !== undefined)
5767
+ patch.lastFrameAssetId = spec.lastFrameAssetId;
5768
+ if (input.audioRefSpokenText !== undefined)
5769
+ patch.audioRefSpokenText = spec.audioRefSpokenText;
5770
+ Object.assign(patch, shotScriptPatch(input));
5771
+ // Same rule for the three mention lists — naming one must not clear the
5772
+ // other two.
5773
+ {
5774
+ const built = spec.mentions;
5775
+ const named = {};
5776
+ if (input.characterIds !== undefined)
5777
+ named.characterIds = built.characterIds;
5778
+ if (input.environmentIds !== undefined)
5779
+ named.environmentIds = built.environmentIds;
5780
+ if (input.styleIds !== undefined)
5781
+ named.styleIds = built.styleIds;
5782
+ if (Object.keys(named).length > 0)
5783
+ patch.mentions = named;
5784
+ }
5785
+ return patch;
5786
+ }
5146
5787
  /**
5147
5788
  * The capability gate, applied to a SAVED recipe.
5148
5789
  *
@@ -5179,7 +5820,8 @@ export const createShot = {
5179
5820
  input: z.object({
5180
5821
  projectId: z.string().uuid(),
5181
5822
  name: z.string().max(120).optional().describe('What to call it. Shown on the card; the prompt supplies one if you omit it.'),
5182
- prompt: z.string().min(1).max(4000).describe('The RAW prompt, @mentions intact. Never write "image 1" yourself — the composer numbers references, and a hand-written number is wrong the moment one moves.'),
5823
+ prompt: z.string().min(1).max(4000).optional().describe('The RAW prompt, @mentions intact. Never write "image 1" yourself — the composer numbers references, and a hand-written number is wrong the moment one moves. Required unless fromAssetId supplies the recipe.'),
5824
+ fromAssetId: z.string().optional().describe("Save as shot: start from this asset's RECORDED recipe (UUID or badge code) — its prompt, model, settings and references — as Media's Save as shot does. That result becomes the Shot's first take unless another Shot already holds it. Anything else you pass overrides the recipe."),
5183
5825
  model: z.string().optional().describe('Model id — the same ids slates_generate_image / slates_generate_video / slates_generate_audio take, and their descriptions carry the routing. Optional: a Shot can be planned before the model is decided.'),
5184
5826
  params: shotParamsSchema,
5185
5827
  refs: shotRefsSchema,
@@ -5196,6 +5838,8 @@ export const createShot = {
5196
5838
  ...shotScriptSchema,
5197
5839
  }),
5198
5840
  async run(input, ctx) {
5841
+ if (!input.prompt && !input.fromAssetId)
5842
+ throw new Error('Pass a prompt, or fromAssetId to save a result\'s recorded recipe as the Shot.');
5199
5843
  const capErr = assertShotCapabilities(input.model, input.params);
5200
5844
  if (capErr)
5201
5845
  return capErr;
@@ -5204,6 +5848,9 @@ export const createShot = {
5204
5848
  return alignErr;
5205
5849
  const desktop = ctx.desktop();
5206
5850
  await desktop.requireCapability('shots', 'saved Shots');
5851
+ // 1.6.0 ignores fromAssetId, and would then refuse the missing prompt.
5852
+ if (input.fromAssetId)
5853
+ await desktop.requireCapability('shot-from-asset', 'saving a result as a Shot');
5207
5854
  // 1.5.8 ignores `position` and files the Shot last.
5208
5855
  if (input.position !== undefined)
5209
5856
  await desktop.requireCapability('shot-position', 'placing a new Shot at a slot');
@@ -5211,15 +5858,21 @@ export const createShot = {
5211
5858
  const r = await desktop.post('/agent/shots', {
5212
5859
  projectId: input.projectId,
5213
5860
  name: input.name,
5214
- spec: { ...spec, ...shotScriptPatch(input) },
5861
+ // From a result, only what the caller NAMED goes over its recorded recipe: the route merges it.
5862
+ spec: input.fromAssetId ? namedSpecPatch(input, spec) : { ...spec, ...shotScriptPatch(input) },
5863
+ fromAssetId: input.fromAssetId,
5215
5864
  frameId: input.frameId ?? null,
5216
5865
  sceneId: input.sceneId ?? null,
5217
5866
  storyboardId: input.storyboardId ?? null,
5218
5867
  position: input.position,
5219
5868
  });
5869
+ const fromNote = input.fromAssetId
5870
+ ? ` Saved from ${input.fromAssetId}: its recorded recipe is the Shot's, and that result is its first take (unless another Shot already holds it).` +
5871
+ (r.unsavedPaths?.length ? ` ${r.unsavedPaths.length} recorded attachment(s) are not project assets and are not on the Shot.` : '')
5872
+ : '';
5220
5873
  // The CODE is the address the user sees on the row — say it back so the
5221
5874
  // next call, and the next sentence to the user, can point at it.
5222
- return ok(r.shot, `${r.shot?.code || 'Shot'} — "${r.shot?.name || 'Untitled'}". ${refEcho}`.trim());
5875
+ return ok(r.shot, `${r.shot?.code || 'Shot'} — "${r.shot?.name || 'Untitled'}". ${refEcho}${fromNote}`.trim());
5223
5876
  },
5224
5877
  };
5225
5878
  export const updateShot = {
@@ -5257,52 +5910,7 @@ export const updateShot = {
5257
5910
  const { spec } = await buildShotSpecInput(ctx, input.projectId, input);
5258
5911
  // Only send the halves the caller actually named; the route merges a PARTIAL
5259
5912
  // spec over the stored one, so an omitted field is never silently cleared.
5260
- const patch = {};
5261
- if (input.prompt !== undefined)
5262
- patch.prompt = input.prompt;
5263
- if (input.model !== undefined) {
5264
- patch.model = input.model;
5265
- // 🚨 `authoredFor` is NOT re-stamped on a model swap. The whole point of
5266
- // recording it is that the prompt stays written for the model it was
5267
- // written for — the grammars genuinely differ — so the card can say so.
5268
- }
5269
- if (input.params !== undefined)
5270
- patch.params = spec.params;
5271
- // 🚨 ONLY THE ROLES THE CALLER NAMED. `buildShotSpecInput` always returns a
5272
- // COMPLETE refs record, and the route merges one level deep — so sending all
5273
- // five would clear every role the caller never mentioned. Clearing a role is
5274
- // explicit: send `[]`.
5275
- if (input.refs !== undefined) {
5276
- const built = spec.refs;
5277
- const named = {};
5278
- for (const role of ORDERED_ATTACHMENT_ROLES) {
5279
- if (input.refs[role] !== undefined)
5280
- named[role] = built[role];
5281
- }
5282
- if (Object.keys(named).length > 0)
5283
- patch.refs = named;
5284
- }
5285
- if (input.firstFrameAssetId !== undefined)
5286
- patch.firstFrameAssetId = spec.firstFrameAssetId;
5287
- if (input.lastFrameAssetId !== undefined)
5288
- patch.lastFrameAssetId = spec.lastFrameAssetId;
5289
- if (input.audioRefSpokenText !== undefined)
5290
- patch.audioRefSpokenText = spec.audioRefSpokenText;
5291
- Object.assign(patch, shotScriptPatch(input));
5292
- // Same rule for the three mention lists — naming one must not clear the
5293
- // other two.
5294
- {
5295
- const built = spec.mentions;
5296
- const named = {};
5297
- if (input.characterIds !== undefined)
5298
- named.characterIds = built.characterIds;
5299
- if (input.environmentIds !== undefined)
5300
- named.environmentIds = built.environmentIds;
5301
- if (input.styleIds !== undefined)
5302
- named.styleIds = built.styleIds;
5303
- if (Object.keys(named).length > 0)
5304
- patch.mentions = named;
5305
- }
5913
+ const patch = namedSpecPatch(input, spec);
5306
5914
  const r = await desktop.post('/agent/shots/update', {
5307
5915
  id: input.shotId,
5308
5916
  data: {
@@ -5407,7 +6015,7 @@ function describeVarietyReport(v) {
5407
6015
  }
5408
6016
  export const listShots = {
5409
6017
  id: 'slates_list_shots',
5410
- description: "Read the shot list — every Shot as a compact row IN BOARD ORDER (scene, then position), with the piece's cut count, runtime, saved-recipe price and its variety distribution. Read this before firing a set: if one shot size is the plurality or three cuts in a row share a camera move, the batch is wrong before a credit is spent.",
6018
+ description: "Read the shot list — every Shot as a compact row IN BOARD ORDER (scene, then position), with its place on the board (2C: scene number and slot letter, as the window shows it and the user says it), the piece's cut count, runtime, saved-recipe price and its variety distribution. Read this before firing a set: if one shot size is the plurality or three cuts in a row share a camera move, the batch is wrong before a credit is spent.",
5411
6019
  input: z.object({
5412
6020
  projectId: z.string().uuid(),
5413
6021
  storyboardId: z.string().uuid().optional().describe('Only Shots attached to a frame in this board.'),
@@ -5448,6 +6056,7 @@ export const listShots = {
5448
6056
  name: s.name,
5449
6057
  scene: s.sceneName,
5450
6058
  position: s.position,
6059
+ ...(s.place !== undefined ? { place: s.place } : {}),
5451
6060
  model: s.model,
5452
6061
  references: s.referenceCount,
5453
6062
  cuts: s.cuts,
@@ -5472,7 +6081,7 @@ ${describeVarietyReport(r.variety)}` : ''));
5472
6081
  };
5473
6082
  export const getShot = {
5474
6083
  id: 'slates_get_shot',
5475
- description: 'Read one Shot in full — the COMPOSED prompt the request will actually carry, its numbered references, anything it points at that no longer exists, and its exact credit quote. Audit your own work here before firing.',
6084
+ description: 'Read one Shot in full — the COMPOSED prompt the request will actually carry, its numbered references, anything it points at that no longer exists, its exact credit quote, and its place on the board (2C, as the window shows it). Audit your own work here before firing.',
5476
6085
  input: z.object({
5477
6086
  shotId: z.string().describe('The Shot id, or its SHOT-A code as shown on the row.'),
5478
6087
  }),
@@ -5484,7 +6093,7 @@ export const getShot = {
5484
6093
  const r = await desktop.get('/agent/shots/get', { id: input.shotId });
5485
6094
  const quote = await desktop.get('/agent/shots/quote', { input: JSON.stringify({ projectId: r.shot.projectId, shotIds: [r.shot.id] }) });
5486
6095
  const q = quote.items[0];
5487
- return ok({ ...r.shot, credits: q.credits, quoteFingerprint: quote.fingerprint }, `"${r.shot.name || 'Untitled'}" — ${r.shot.model ?? 'no model set'}, ` +
6096
+ return ok({ ...r.shot, credits: q.credits, quoteFingerprint: quote.fingerprint }, `${r.shot.place ? `${r.shot.place} ` : ''}"${r.shot.name || 'Untitled'}" — ${r.shot.model ?? 'no model set'}, ` +
5488
6097
  (q.credits != null && !r.shot.blocked
5489
6098
  ? `${fmtCredits(q.credits ?? 0)}.`
5490
6099
  : `CANNOT FIRE YET: ${r.shot.blocked ?? 'not priceable — set a model and a duration.'}`) +
@@ -5492,6 +6101,31 @@ export const getShot = {
5492
6101
  `\nCOMPOSED PROMPT (what the model is told): ${r.shot.composedPrompt}`);
5493
6102
  },
5494
6103
  };
6104
+ /**
6105
+ * The Board's "Use pictures as first frames": each video Shot with a picture and no first frame starts its
6106
+ * video from that picture. The desktop runs the command's own pick and write (`firstFrameCandidate`,
6107
+ * `withFirstFrame`), so the recipe changes exactly as a click changes it. Free: nothing is generated.
6108
+ */
6109
+ export const usePicturesAsFirstFrames = {
6110
+ id: 'slates_use_pictures_as_first_frames',
6111
+ description: "The Board's 'Use pictures as first frames': each video Shot that has a picture and no first frame starts its video from that picture (its tile picture, else its newest picture take; a picture that was its only plain Reference moves into the frame slot). Nothing is generated and nothing is charged. Pass shotIds (ids or SHOT-A codes) to act on those Shots; omit them to act on every Shot of the board open in the window, as the command does (open one with slates_set_view first). Returns, per Shot, the picture it set or why it had none.",
6112
+ input: z
6113
+ .object({
6114
+ projectId: z.string().uuid(),
6115
+ shotIds: z.array(z.string().min(1)).min(1).max(200).optional().describe('Only these Shots. Omit for every Shot of the board open in the window.'),
6116
+ })
6117
+ .strict(),
6118
+ async run(input, ctx) {
6119
+ await ctx.desktop().requireCapability('first-frames', 'using pictures as first frames');
6120
+ const r = await ctx.desktop().post('/agent/shots/first-frames', input);
6121
+ const name = (row) => row.place ?? row.code ?? row.shotId;
6122
+ const lines = r.results.map((row) => `${name(row)}: ${row.set ? 'first frame set' : (row.reason ?? 'nothing to set')}`);
6123
+ return {
6124
+ text: `${r.set} of ${r.results.length} Shot${r.results.length === 1 ? '' : 's'} now start from their picture.${lines.length ? `\n${lines.join('\n')}` : ''}`,
6125
+ data: r,
6126
+ };
6127
+ },
6128
+ };
5495
6129
  /**
5496
6130
  * SPLIT and MERGE — the chop decision, and the only two cross-row operations on
5497
6131
  * this surface.
@@ -5944,7 +6578,7 @@ export const editCut = {
5944
6578
  };
5945
6579
  export function resolveGuideTopic(topic) {
5946
6580
  const t = topic.trim().toLowerCase();
5947
- if (SKILLS[t])
6581
+ if (Object.hasOwn(SKILLS, t))
5948
6582
  return t;
5949
6583
  if (t === 'slates-character-turnaround' || t === 'character-turnaround') {
5950
6584
  return 'slates-character-identity';
@@ -6018,8 +6652,6 @@ export function resolveGuideTopic(topic) {
6018
6652
  return 'slates-prompting-flux-2-max';
6019
6653
  if (t.startsWith('seedream'))
6020
6654
  return 'slates-prompting-seedream-5-lite';
6021
- if (t.startsWith('veo'))
6022
- return 'slates-prompting-veo-3';
6023
6655
  if (t.startsWith('omni-flash') || t.startsWith('gemini-omni') || t === 'omni flash')
6024
6656
  return 'slates-prompting-omni-flash';
6025
6657
  // MiniMax H3 — both seats share one skill. Placed BEFORE the seed/seedance
@@ -6040,7 +6672,7 @@ export function resolveGuideTopic(topic) {
6040
6672
  if (t.startsWith('kling-mc'))
6041
6673
  return 'slates-prompting-motion-transfer';
6042
6674
  if (t === 'edit-video' || t === 'video-edit' || t === 'edit video' || t === 'video edit')
6043
- return 'slates-prompting-kling-v3';
6675
+ return resolveGuideTopic(defaultModelFor('video', 'edit'));
6044
6676
  if (t.startsWith('kling-v3'))
6045
6677
  return 'slates-prompting-kling-v3';
6046
6678
  // Audio — the TTS seat FIRST, then seed-audio, then eleven-sfx.
@@ -6101,70 +6733,96 @@ export function resolveGuideTopic(topic) {
6101
6733
  }
6102
6734
  return null;
6103
6735
  }
6104
- /**
6105
- * The guide index, GENERATED from SKILLS.
6106
- *
6107
- * The list here was hand-typed and had drifted to 25 of 32 names — the prompting
6108
- * guides for GPT Image, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
6109
- * all missing, so an agent reading this description could not learn they exist.
6110
- * A hand-typed index of a generated corpus is a stale index; it is only a matter
6111
- * of when.
6112
- *
6113
- * Per-model guides are listed as BARE NAMES: the name is the description, and
6114
- * `resolveGuideTopic()` resolves a model id to the right one anyway.
6115
- */
6116
- function describeGuideTopics() {
6117
- const names = Object.keys(SKILLS).sort();
6118
- const perModel = names.filter((n) => n.startsWith('slates-prompting-'));
6119
- const rest = names.filter((n) => !n.startsWith('slates-prompting-'));
6120
- return (`Workflow and cross-cutting guides: ${rest.join(', ')}. ` +
6121
- `Per-model prompting guides (or just pass the model id, which resolves to one of these): ` +
6122
- `${perModel.join(', ')}. ` +
6123
- `Style names (photoreal, anime, painterly, 3d-render) resolve to slates-style-prompting.`);
6736
+ // Discovery is the first call of most briefs: a slow or unentitled member check
6737
+ // must not cost every call a round trip, or hang free craft behind a 30s read.
6738
+ const MEMBER_CATALOG_TTL_MS = 60_000;
6739
+ const MEMBER_CATALOG_TIMEOUT_MS = 5_000;
6740
+ const memberCatalogCache = new WeakMap();
6741
+ /** Private bodies stay on the entitled member feed, never in public packages. */
6742
+ function memberGuideCatalog(ctx) {
6743
+ const cached = memberCatalogCache.get(ctx.cloud);
6744
+ if (cached && Date.now() - cached.at < MEMBER_CATALOG_TTL_MS)
6745
+ return cached.value;
6746
+ const value = fetchMemberGuideCatalog(ctx);
6747
+ memberCatalogCache.set(ctx.cloud, { at: Date.now(), value });
6748
+ return value;
6749
+ }
6750
+ async function fetchMemberGuideCatalog(ctx) {
6751
+ try {
6752
+ // Request first: a missing token throws here, before any timer exists.
6753
+ const request = ctx.cloud().get('/members/manifest.json');
6754
+ let timer;
6755
+ const timeout = new Promise((_, reject) => { timer = setTimeout(() => reject(new Error('Member guide check timed out')), MEMBER_CATALOG_TIMEOUT_MS); });
6756
+ const manifest = await Promise.race([request, timeout]).finally(() => clearTimeout(timer));
6757
+ if (!Array.isArray(manifest?.skills))
6758
+ throw new Error('Invalid member guide manifest');
6759
+ // A malformed entry is skipped; it never hides the account's other playbooks.
6760
+ const entries = manifest.skills
6761
+ .filter(entry => entry.tier === 'paid' && typeof entry.name === 'string' && /^slates-[a-z0-9]+(?:-[a-z0-9]+)*$/.test(entry.name) && entry.name.length <= 64 && typeof entry.description === 'string' && entry.description.length <= 1024)
6762
+ .map(entry => ({ name: entry.name, description: entry.description, tier: 'paid' }));
6763
+ return { entries, access: 'available' };
6764
+ }
6765
+ catch (error) {
6766
+ if (error.code === 'CLOUD_TOKEN_MISSING')
6767
+ return { entries: [], access: 'not connected; bundled guides available' };
6768
+ if (error instanceof SlatesCloudHttpError && error.status === 402)
6769
+ return { entries: [], access: 'no active skills entitlement; bundled guides available' };
6770
+ if (error instanceof SlatesCloudHttpError && error.status === 404)
6771
+ return { entries: [], access: 'member feed unavailable on this API version; bundled guides available' };
6772
+ // Keep free craft usable during a cloud outage, but expose the failed access check.
6773
+ return { entries: [], access: error instanceof SlatesCloudHttpError && error.status === 401 ? 'reconnect Slates to access member guides; bundled guides available' : 'member access check failed; retry for private guides; bundled guides available' };
6774
+ }
6124
6775
  }
6125
6776
  export const getPromptingGuide = {
6126
6777
  id: 'slates_get_prompting_guide',
6127
- description: 'For app help and exact UI instructions use topic "app-manual" with a query such as "voice recording". This returns the canonical product manual, shared by every agent surface. ' +
6128
- // 🚨 NO "ALWAYS READ THIS FIRST" SENTENCE. It stood here for months and was
6129
- // MEASURED at 13% compliance before and after the enforcement work — pointer
6130
- // prose is the shape that does not move the agent. What replaced it is
6131
- // structural: the never-use list rides the generate ops' descriptions and
6132
- // the craft card rides the estimate result, so the facts arrive whether or
6133
- // not this op is ever called.
6134
- "Return a bundled Slates prompting/workflow guide. MCP-only clients (Claude Desktop, Smithery) don't get the CLI-installed skill files — call this instead. Accepts a guide name or a model id ('veo-3.1-fast', 'kling-v3.0-pro', 'seedance-2', 'nano-banana-2'), which maps to the right guide. Reach for it when a card is not enough: the failure modes, the worked examples and the sources are only in the full text.",
6778
+ description: 'Find production craft from the user vision: pass query alone with the brief (for example "two friends talking in a rainy diner") for keyword-ranked guides with matching sections, followed by every other guide\'s description so you choose by judgment. Omit topic and query, or use topic "catalog", for the whole catalog. Then request a guide name, model id or style name with card/index/section/full depth. Cards are short; query with a topic selects one section or cinematic technique; full returns worked examples, failure modes and sources. Reuse current guidance already in context. Entitled member playbooks are fetched privately through the connected Slates account. For exact buttons and UI paths use topic "app-manual" with relevant question keywords; no query returns its surface map. Users supply the vision; you find the guides.',
6135
6779
  input: z.object({
6136
- query: z.string().max(200).optional().describe('Keywords, section heading, or exact cinematic technique ID. Returns only the matching section or technique.'),
6137
- topic: z
6138
- .string()
6139
- .min(1)
6140
- .describe(`Guide name, model id, or style name. ${describeGuideTopics()}`),
6141
- depth: z.enum(['card', 'index', 'section', 'full']).optional().describe('Default "card" is a short overview. "index" lists sections; query selects one section or technique; "full" explicitly returns the complete guide.'),
6780
+ query: z.string().min(1).max(1000).optional().describe('Without topic: creative brief or craft need to discover guides. With topic: section keywords, heading or technique ID.'),
6781
+ topic: z.string().min(1).optional().describe('Optional guide name, model id, style name, catalog, or app-manual. Omit to search by intent or browse.'),
6782
+ depth: z.enum(['card', 'index', 'section', 'full']).optional().describe('Default card is a short overview; index lists sections; query selects a section; full returns the complete guide.'),
6783
+ limit: z.number().int().min(1).max(20).optional().describe('Page size: a search returns 8 ranked matches by default, browsing the whole catalog. Does not change guide bodies.'),
6784
+ offset: z.number().int().min(0).optional().describe('Catalog/search offset from nextOffset in the preceding result.'),
6142
6785
  }),
6143
- async run(input) {
6144
- if (input.topic.trim().toLowerCase() === 'app-manual') {
6145
- const content = input.query || input.depth === 'full'
6146
- ? appManualSections(input.query)
6147
- : 'App manual sections. Pass query to read a section, or depth "full" for the complete manual.\n\n' +
6148
- guideSections(appManualSections()).map((s) => `- ${s.title}`).join('\n');
6786
+ async run(input, ctx) {
6787
+ const topic = input.topic?.trim().toLowerCase();
6788
+ if (topic === 'app-manual') {
6789
+ const content = input.query || input.depth === 'full' ? appManualSections(input.query) : appManualIndex();
6149
6790
  return { text: content, data: { topic: 'app-manual', bytes: Buffer.byteLength(content, 'utf8'), guide: content } };
6150
6791
  }
6151
- const resolved = resolveGuideTopic(input.topic);
6152
- const content = resolved ? SKILLS[resolved] : undefined;
6153
- if (!resolved || content === undefined) {
6154
- throw new Error(`Unknown guide topic: ${input.topic}. Valid topics: ${Object.keys(SKILLS).sort().join(', ')}`);
6792
+ const resolved = topic ? resolveGuideTopic(topic) : null;
6793
+ if (resolved) {
6794
+ const depth = input.depth ?? 'card';
6795
+ const guide = retrieveGuide(resolved, SKILLS[resolved], depth, input.query);
6796
+ return { text: guide, data: { topic: resolved, tier: 'free', depth, bytes: Buffer.byteLength(guide, 'utf8'), guide } };
6155
6797
  }
6156
- const depth = input.depth ?? 'card';
6157
- const guide = retrieveGuide(resolved, content, depth, input.query);
6158
- // Both fields carry the body: some native clients expose structured data only.
6159
- return { text: guide, data: { topic: resolved, depth, bytes: Buffer.byteLength(guide, 'utf8'), guide } };
6798
+ const member = await memberGuideCatalog(ctx);
6799
+ const privateEntry = member.entries.find(entry => entry.name === topic);
6800
+ if (privateEntry) {
6801
+ // The feed's 402/404 bodies are purchase pages; the agent needs the fact, not the page.
6802
+ const result = await ctx.cloud().get(`/members/skills/${encodeURIComponent(privateEntry.name)}.md?format=json`).catch((error) => {
6803
+ if (error instanceof SlatesCloudHttpError && error.status === 402)
6804
+ throw new SlatesCloudHttpError(`${privateEntry.name} needs an active skills entitlement; bundled guides remain available.`, 402);
6805
+ if (error instanceof SlatesCloudHttpError && error.status === 404)
6806
+ throw new SlatesCloudHttpError(`${privateEntry.name} is no longer in the member feed.`, 404);
6807
+ throw error;
6808
+ });
6809
+ if (typeof result?.markdown !== 'string')
6810
+ throw new Error('Invalid member guide body');
6811
+ const depth = input.depth ?? 'card';
6812
+ const guide = retrieveGuide(privateEntry.name, result.markdown, depth, input.query);
6813
+ return { text: guide, data: { topic: privateEntry.name, tier: 'paid', depth, bytes: Buffer.byteLength(guide, 'utf8'), guide } };
6814
+ }
6815
+ const catalog = [...guideCatalog(SKILLS), ...member.entries].sort((a, b) => a.name.localeCompare(b.name));
6816
+ const result = discoverGuides(catalog, SKILLS, input.query ?? (topic && topic !== 'catalog' && topic !== 'index' ? input.topic : undefined), input.limit, input.offset);
6817
+ const guide = result.guides.map(entry => `${entry.name} (${entry.tier}): ${entry.description}${entry.sections.length ? `\n Matching sections: ${entry.sections.join('; ')}` : ''}`).join('\n\n');
6818
+ const rest = result.rest.length ? `\n\nEvery other guide (keyword hints above only see shared words; choose by the brief):\n${result.rest.map(entry => `- ${entry.name} (${entry.tier}): ${entry.description}`).join('\n')}` : '';
6819
+ const text = `${result.fallback ? 'No keyword match; choose relevant craft from the catalog.\n\n' : ''}${guide}${rest}\n\n${result.total} ${result.query && !result.fallback ? 'keyword matches' : 'guides'}; nextOffset: ${result.nextOffset ?? 'none'}. Member guides: ${member.access}. Retrieve a name with depth card/index or query for a section. The user does not need to choose guides.`;
6820
+ return { text, data: { ...result, memberAccess: member.access, bytes: Buffer.byteLength(text, 'utf8') } };
6160
6821
  },
6161
6822
  };
6162
6823
  /**
6163
- * The one op that changes what OTHER ops are visible.
6164
- *
6165
- * 🚨 IT EXISTS BECAUSE THE SURFACE IS 112 KB AND EVERY TURN PAYS FOR ALL OF IT.
6166
- * Both surfaces start with the shared core set. Search returns compact metadata;
6167
- * names/group returns exact schemas and replaces the optional selection.
6824
+ * Task discovery and exact schemas share one operation. The desktop uses
6825
+ * names/group to replace its optional selection; MCP keeps its full list fixed.
6168
6826
  */
6169
6827
  export const loadTools = {
6170
6828
  id: 'slates_load_tools',
@@ -6179,9 +6837,7 @@ export const loadTools = {
6179
6837
  }).refine((v) => [v.group, v.query, v.names].filter(Boolean).length === 1, 'Pass exactly one of query, names, or group.'),
6180
6838
  async run(input) {
6181
6839
  if (input.query) {
6182
- const words = input.query.toLowerCase().split(/[^a-z0-9]+/).filter(Boolean);
6183
- const ranked = ALL_OPERATIONS.map((op) => ({ op, score: words.reduce((n, w) => n + (op.id.includes(w) ? 4 : op.description.toLowerCase().includes(w) ? 1 : 0), 0) }))
6184
- .filter((x) => x.score > 0).sort((a, b) => b.score - a.score).slice(0, 10);
6840
+ const ranked = searchTools(ALL_OPERATIONS, input.query);
6185
6841
  const matches = ranked.map(({ op }) => ({ name: op.id, description: op.description.split(/(?<=\.)\s/)[0], billable: !!op.billable, annotations: op.annotations }));
6186
6842
  return ok({ matches }, matches.map((o) => `${o.name}: ${o.description}`).join('\n') || 'No matching tools. Try another task description.');
6187
6843
  }
@@ -6350,8 +7006,21 @@ export const blenderRenderBlocking = {
6350
7006
  export const ALL_OPERATIONS = [
6351
7007
  getWorkspaceState,
6352
7008
  getSelection,
7009
+ setSelection,
6353
7010
  getView,
6354
7011
  setView,
7012
+ getManualPicture,
7013
+ highlightControl,
7014
+ getComposer,
7015
+ setComposer,
7016
+ reorderFolders,
7017
+ reorderPins,
7018
+ getUsage,
7019
+ getAppSettings,
7020
+ setAppSettings,
7021
+ getAsset,
7022
+ linkAssetSource,
7023
+ extractVideoFrame,
6355
7024
  getMe,
6356
7025
  getCreditBalance,
6357
7026
  listAvailableModels,
@@ -6400,6 +7069,7 @@ export const ALL_OPERATIONS = [
6400
7069
  editVideo,
6401
7070
  trimVideo,
6402
7071
  editImage,
7072
+ extractGridCells,
6403
7073
  getGenerationStatus,
6404
7074
  listGenerations,
6405
7075
  listTimelines,
@@ -6485,6 +7155,7 @@ export const ALL_OPERATIONS = [
6485
7155
  pasteScript,
6486
7156
  listShots,
6487
7157
  getShot,
7158
+ usePicturesAsFirstFrames,
6488
7159
  generateFromShots,
6489
7160
  quoteBoard,
6490
7161
  getBoardProgress,