@slatesvideo/shared 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/api-url.d.ts +5 -3
  2. package/dist/api-url.js +5 -3
  3. package/dist/index.d.ts +1 -0
  4. package/dist/index.js +1 -0
  5. package/dist/manual/content.d.ts +2 -0
  6. package/dist/manual/content.js +3 -0
  7. package/dist/manual/index.d.ts +5 -0
  8. package/dist/manual/index.js +20 -0
  9. package/dist/operations/index.d.ts +67 -17
  10. package/dist/operations/index.js +310 -87
  11. package/dist/prompts/agent-doctrine.js +1 -0
  12. package/dist/prompts/character-sheet.js +10 -0
  13. package/dist/prompts/model-capabilities.js +52 -3
  14. package/dist/prompts/model-facts.js +12 -4
  15. package/dist/prompts/prompting-tips.js +9 -3
  16. package/dist/prompts/reference-composer.js +14 -0
  17. package/dist/prompts/shot-spec.d.ts +30 -10
  18. package/dist/prompts/shot-spec.js +41 -9
  19. package/dist/skills/content.js +12 -12
  20. package/exports/slates-prompt-builder/generated/reference-character.md +1 -1
  21. package/exports/slates-prompt-builder/generated/reference-seedance.md +3 -1
  22. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +10 -10
  23. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  24. package/package.json +1 -1
  25. package/skills/slates-character-identity.md +1 -1
  26. package/skills/slates-model-selection.md +10 -7
  27. package/skills/slates-prompting-elevenlabs.md +1 -1
  28. package/skills/slates-prompting-gpt-image-2-5.md +183 -0
  29. package/skills/slates-prompting-inworld-tts.md +174 -166
  30. package/skills/slates-prompting-lip-sync.md +1 -1
  31. package/skills/slates-prompting-nano-banana-2.md +1 -1
  32. package/skills/slates-prompting-seed-audio.md +1 -1
  33. package/skills/slates-prompting-seedance-2-5.md +3 -1
  34. package/skills/slates-prompting-seedance.md +3 -1
  35. package/skills/slates-ugc-influencer-ad.md +5 -3
  36. package/skills/slates-vision-feedback-loop.md +2 -2
  37. package/skills/slates-prompting-gpt-image-2.md +0 -109
@@ -79,6 +79,7 @@ export function buildSkillIndex() {
79
79
  const PREAMBLE = fork(`You are the Slates Studio Agent — a production assistant living inside Slates, the AI video creation studio. You plan and execute video/image production runs by chaining the Slates tools: script → characters → images → videos → quality-check → regenerate, ending with assets in the user's project (and on the timeline when asked).`, `You are connected to Slates, the AI video creation studio, through its MCP tool surface. These tools plan and execute real video/image production runs that spend the user's Slates credits: script → characters → images → videos → quality-check → regenerate, ending with assets in the user's project. Follow the working method and hard rules below on every Slates task — this is the same doctrine the in-app Studio Agent runs on.`);
80
80
  // ── The working method ─────────────────────────────────────────────
81
81
  export const WORKING_METHOD = [
82
+ both(`For HOW/WHERE questions, load slates_get_prompting_guide with topic "app-manual" and relevant query keywords. Teach the documented buttons and tabs, preserving model-specific conditions; do not invent UI paths or mutate the project when the user only asks for instructions. Slates is a sandbox of optional tools, not a required pipeline.`),
82
83
  both(`1. UNDERSTAND the outcome the user wants. If intent is clear, act with sane defaults — don't interrogate. If genuinely ambiguous, batch every question into ONE message.`),
83
84
  both(`2. ORIENT: call slates_get_workspace_state once at the start of a workflow. Work in the user's CURRENT project — this chat lives inside it. NEVER create a new project unless explicitly asked; if there's no current project, ask which to use.`),
84
85
  both(`3. LOAD KNOWLEDGE ON DEMAND: before prompting any model or running a multi-step workflow, load the matching guide with slates_get_prompting_guide (index below). Only the guides the task needs, when it needs them.`),
@@ -78,6 +78,16 @@
78
78
  // without the absence clause. If that happens, the fix is a
79
79
  // model-conditional phrasing, not restoring the 422.
80
80
  //
81
+ // 2026-09-09 — THE MODEL IT WAS MEASURED ON IS RETIRED; THE RULE IS NOT.
82
+ // GPT Image 2.5 replaced gpt-image-2 in the picker. The 422 above was
83
+ // measured on gpt-image-2 and that wording is left exactly as recorded,
84
+ // because a receipt names what was actually tested. What carries over is
85
+ // the MECHANISM, not the measurement: the classifier belongs to OpenAI,
86
+ // not to a model version, so phrase exclusions as framing on 2.5 too.
87
+ // ⚠️ INHERITED, NOT RE-MEASURED — nobody has re-run the 422 on Flare or
88
+ // Sunburst. If one of them accepts the absence clause, that is a new
89
+ // receipt to write down, not a reason to delete this one.
90
+ //
81
91
  // 2026-07-30 (b) — THE GENRE ANCHOR HAS TO BE SCOPED TO THE FACE, and
82
92
  // this one cost real generations. "an invisible-mannequin presentation
83
93
  // WHERE THE CLOTHING HOLDS ITS OWN SHAPE" is the e-commerce genre stated
@@ -133,10 +133,48 @@ export const MODEL_CAPABILITIES = {
133
133
  aspectRatios: FULL_ASPECT_RATIOS,
134
134
  maxRefImages: 14,
135
135
  },
136
- 'gpt-image-2': {
137
- // FIVE, not ten. The op's flat enum offered eleven for every image model.
136
+ // `gpt-image-2` WAS HERE AND IS DELIBERATELY GONE (retired 2026-09-09).
137
+ //
138
+ // An earlier pass kept the row so that a Shot saved before the swap could
139
+ // still resolve its caps by stored id. That was the wrong fix and the
140
+ // desktop refuses it: `pricing.ts` throws at MODULE LOAD for any capability
141
+ // row with no MODEL_REGISTRY entry, because an orphan row makes THIS op
142
+ // advertise, validate and quote a model the desktop can no longer render —
143
+ // the agent passes every gate and then hits 'Unsupported model' at the
144
+ // handler.
145
+ //
146
+ // Old Shots are handled where they are READ instead: `migrateGptImageModel`
147
+ // in slate/src/shared/pricing.ts rewrites the stored id to Flare inside
148
+ // `transform` (slate/src/main/storage/shots.ts), the one place a DB row
149
+ // becomes a Shot. That is strictly better than keeping the row — the Shot
150
+ // comes back FIREABLE on a live model, rather than merely openable on a dead
151
+ // one. Do not re-add this row to make a stale id resolve; migrate it.
152
+ // GPT Image 2.5 — 16 references, which IS fal's documented ceiling rather
153
+ // than a number of ours: `image_urls` carries `maxItems: 16` on both 2.5
154
+ // endpoints AND on both gpt-image-2 endpoints (schema, read 2026-09-09,
155
+ // ripped verbatim to second-brain/business/projects/slates/research/
156
+ // gpt-image-2-5-fal-api-docs.md).
157
+ //
158
+ // 🚨 IT WAS 10 UNTIL 2026-09-09, AND 10 WAS NEVER ANYBODY'S LIMIT. The
159
+ // comment here used to call it "the 10-reference ceiling ... unchanged by the
160
+ // version bump", which reads as a verified fal constraint and was not one —
161
+ // nobody had checked. Six reference slots were being given away, worst on
162
+ // Sunburst, whose entire reason for shipping is multi-reference edit work.
163
+ // Raised by Eric 2026-09-09 ("did we go completely full-on with all of the
164
+ // options on fal?").
165
+ //
166
+ // The five aspect ratios ARE a product choice; fal takes any custom size
167
+ // inside its own bounds (see GPT_IMAGE_25_SIZES in slate/src/shared/
168
+ // pricing.ts). Two SLATES caps sit outside this file and are not model
169
+ // limits either: the MCP's 4,000-character prompt against fal's 32,000, and
170
+ // image quantity, which is a fan-out and has no provider ceiling at all.
171
+ 'gpt-image-2-5-flare': {
138
172
  aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
139
- maxRefImages: 10,
173
+ maxRefImages: 16,
174
+ },
175
+ 'gpt-image-2-5-sunburst': {
176
+ aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
177
+ maxRefImages: 16,
140
178
  },
141
179
  'flux-2-max': {
142
180
  aspectRatios: FULL_ASPECT_RATIOS,
@@ -368,6 +406,17 @@ export const MODEL_CAPABILITIES = {
368
406
  },
369
407
  'minimax-h3-max': {
370
408
  aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
409
+ // 🚨 ZERO, DECLARED — not omitted. `minimax/h3-max/reference-to-video`
410
+ // returns 404, so there is no transport for a reference of any modality.
411
+ // Leaving this undeclared does NOT mean "none": `getMaxRefImages` falls
412
+ // back to `?? 3` for the ingredients mode, which handed this row three
413
+ // reference slots it cannot send. Measured 2026-09-09 — a pinned image on
414
+ // an h3-max generation was accepted by the composer, dropped in transit,
415
+ // and the model rendered the prompt text alone, returning a different
416
+ // person than the reference. Frames (start/end) remain the ONLY image
417
+ // transport on this seat.
418
+ maxIngredientImages: 0,
419
+ maxRefImages: 0,
371
420
  // 480p/768p ONLY — fal's post-train of the open weights, and the 2K
372
421
  // upscaler was never open-sourced. Declaring the shorter ladder here IS the
373
422
  // whole Max-seat mechanism: `assertVideoCapabilities` refuses 2K/4K on this
@@ -146,12 +146,20 @@ export const MODEL_FACTS = [
146
146
  notes: 'HERO-FRAME / typography PREMIUM image tier. NB2 is about 95% of Pro — escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene or deep multi-element reasoning must be perfect, and say why.',
147
147
  },
148
148
  {
149
- id: 'gpt-image-2',
149
+ id: 'gpt-image-2-5-flare',
150
150
  route: 'generate',
151
- label: 'GPT Image 2',
151
+ label: 'GPT Image 2.5 Flare',
152
152
  kind: 'image',
153
- ...caps('gpt-image-2'),
154
- notes: 'TEXT / DIAGRAM / PANEL king — near-perfect character-level text, ordered panels, exact placement. Route here for character sheets, shot grids and text-bearing panels. ALSO THE PHOTOREAL FRONT-RUNNER (Eric, 2026-08-24): at quality high it beat both Nano Banana rails head-to-head on skin realism, so route photoreal people HERE rather than away. Banana still owns edit-heavy work and the largest reference ceiling. Its own content filter, distinct from Gemini\'s. Killed if a head-to-head at the intended crop goes the other way — re-run the evidence test, never carry this forward on reputation.',
153
+ ...caps('gpt-image-2-5-flare'),
154
+ notes: 'THE FAST GPT IMAGE SEAT — OpenAI\'s small model, optimized for SPEED, quality COMPARABLE to GPT Image 2 (not better) at roughly half the latency. Route here when speed matters: drafts, exploration, volume. TEXT / DIAGRAM / PANEL work — character sheets, shot grids, text-bearing panels. When quality outranks speed, escalate to Sunburst. Own content filter, distinct from Gemini\'s. Killed by a head-to-head at the intended crop going the other way.',
155
+ },
156
+ {
157
+ id: 'gpt-image-2-5-sunburst',
158
+ route: 'generate',
159
+ label: 'GPT Image 2.5 Sunburst',
160
+ kind: 'image',
161
+ ...caps('gpt-image-2-5-sunburst'),
162
+ notes: 'THE QUALITY GPT IMAGE SEAT — OpenAI\'s most capable image model, higher quality than GPT Image 2, same price as Flare, deliberately SLOWER. Route here whenever quality outranks speed: finals, hero frames, photoreal people, and multi-reference edits where every reference must survive into one frame — its widest lead. Not for drafts; you pay latency on every frame. Explore on Flare, finish on Sunburst.',
155
163
  },
156
164
  {
157
165
  id: 'flux-2-max',
@@ -83,8 +83,14 @@ const SEEDANCE = {
83
83
  },
84
84
  {
85
85
  heading: 'Images, clips and audio in ONE generation',
86
- example: 'Marcus (image 1) performs the motion from video 1, speaking the line in audio 1.',
87
- note: 'Attaching a clip does NOT mean "edit this clip". A video or audio attachment is a REFERENCE, numbered in the rail exactly like an image, and it sits alongside your images in the same generation — the composer cites them as "image N", "video N", "audio N", in rail order, and shows you the exact sentence before you press Generate. Reorder the tiles to change what those numbers mean. To actually rewrite a clip, use Edit with AI instead — that is a different, deliberate choice.',
86
+ example: 'Marcus (image 1) performs the motion from video 1 and uses the voice timbre from audio 1.',
87
+ note: 'Attaching a clip does NOT mean "edit this clip". A video or audio attachment is a REFERENCE, numbered in the rail exactly like an image, and it sits alongside your images in the same generation — the composer cites them as "image N", "video N", "audio N", in rail order, and shows you the exact sentence before you press Generate. Reorder the tiles to change what those numbers mean. To actually rewrite a clip, use Edit with AI instead — that is a different, deliberate choice. EACH MODALITY IS NUMBERED SEPARATELY, from 1 — two images and one clip are "image 1", "image 2" and "video 1", never a single running count, so an audio reference is "audio 1" no matter how many images sit in front of it.',
88
+ critical: true,
89
+ },
90
+ {
91
+ heading: 'Give a character a voice',
92
+ example: 'Sarah (image 1) uses the voice timbre from audio 1. She says, "We open in ten minutes."',
93
+ note: 'An audio reference can mean five different things to Seedance — music, dialogue, voice, tone or timbre — so SAY WHICH. Name it as the voice timbre and the clip supplies the voice while your prompt supplies the words; leave it unroled and the model falls back to dialogue, re-transcribes the clip and speaks ITS words instead (a real take came back as "a map called Slates" for "an app called Slates"). Bind each speaker in a sentence rather than by attachment order — position carries nothing: "Images 1-2 are Character 1 and correspond to Audio 1; Images 3-4 are Character 2 and correspond to Audio 2." Verbatim from ByteDance. Audio-alone works on 2.5; on 2.0 pair it with at least one image or video.',
88
94
  critical: true,
89
95
  },
90
96
  {
@@ -613,7 +619,7 @@ const MINIMAX_H3 = {
613
619
  {
614
620
  heading: 'Cite references by number',
615
621
  example: 'Marcus (image 1) walks into the workshop (image 2)...',
616
- note: `H3 takes references as typed slots and expects plain numbered prose — image 1, video 1, audio 1. Do not hand-write angle-bracket tags. ${PARTIALS['reference-tips-short']}`,
622
+ note: `H3 takes references as typed slots and expects plain numbered prose — image 1, video 1, audio 1, each modality counted separately from 1. fal's own prompt-field description: "Refer to reference assets by their modality and order in the reference lists: Image 1, Image 2, Video 1, Audio 1, and so on." Do not hand-write angle-bracket tags — MiniMax's guides use \`<Subject N>\` / \`<Audio N>\` as DOCUMENTATION notation and typing them puts literal brackets in the prompt. An audio reference binds as a voice timbre to a named speaker, the same primitive Seedance uses. ${PARTIALS['reference-tips-short']}`,
617
623
  },
618
624
  {
619
625
  heading: 'Say how much of a reference survives',
@@ -107,6 +107,20 @@ export function isHexColorToken(sigil, token) {
107
107
  }
108
108
  export function composeReferences(rawPrompt, groups, opts = {}) {
109
109
  // ── 1. Assign global numbers by walking the list in order ──
110
+ //
111
+ // 🚨 ONE COUNTER PER MODALITY, EACH STARTING AT 1 — never a single running
112
+ // count across the attachments. `audio 1` is the first AUDIO, however many
113
+ // images precede it. This is the vendors' scheme, not a convenience:
114
+ // BytePlus states it ("The numbering should correspond to the upload order of
115
+ // the assets, such as Image 1 / Video 1 / Audio 1") and PROVES it in a mixed
116
+ // example that numbers two audios 1 and 2 behind four images — a global
117
+ // counter would make them 5 and 6. fal says the same for MiniMax H3 in its
118
+ // own prompt-field description. Receipts and line refs:
119
+ // second-brain/business/projects/slates/research/model-prompting-research.md
120
+ // § 2026-09-09 Multimodal reference GRAMMAR.
121
+ //
122
+ // Collapsing these into one counter would renumber every citation the prompt
123
+ // makes, silently — the model would be told "audio 1" about an image.
110
124
  let imageNum = opts.startImageNumber ?? 0;
111
125
  let videoNum = opts.startVideoNumber ?? 0;
112
126
  let audioNum = opts.startAudioNumber ?? 0;
@@ -60,8 +60,15 @@ export interface ShotParams {
60
60
  imageResolution?: string;
61
61
  videoResolution?: string;
62
62
  quality?: string;
63
- /** gpt-image-2's tier. Always sent explicitly: fal's own default is `high`. */
64
- gptQuality?: 'medium' | 'high';
63
+ /** GPT Image 2.5's tier. Always sent explicitly: fal's own default is
64
+ * `high`, which is the third of five rungs, not the top of two. */
65
+ gptQuality?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
66
+ /** GPT Image's alpha switch. `auto` is fal's default and ours; `transparent`
67
+ * asks for a real alpha channel rather than a painted backdrop. Costs
68
+ * nothing — fal prices this family on size × quality only, so it is NOT a
69
+ * cost-key segment. Named `gptBackground` because `background` already
70
+ * means "generate asynchronously" on every op that carries a Shot. */
71
+ gptBackground?: 'auto' | 'transparent' | 'opaque';
65
72
  duration?: number;
66
73
  imageQuantity?: number;
67
74
  gridMode?: 'off' | '2x2' | '3x3';
@@ -91,6 +98,15 @@ export interface ShotParams {
91
98
  audioLoop?: boolean;
92
99
  audioPromptInfluence?: number;
93
100
  audioMultilingual?: boolean;
101
+ /**
102
+ * The VOICE a text-to-speech Shot speaks in — exactly one of the three, the
103
+ * same three `slates_generate_audio` takes. A Shot that carried the words but
104
+ * not the voice would fire in whatever voice happened to be on the bar, which
105
+ * is not the recipe that was saved.
106
+ */
107
+ voiceId?: string;
108
+ voiceReferenceAssetId?: string;
109
+ voiceDescription?: string;
94
110
  }
95
111
  /** Prompt-owned identity — the entities the prompt text NAMES. */
96
112
  export interface ShotMentions {
@@ -174,14 +190,14 @@ export interface ShotSpec {
174
190
  * same failure as a column nothing renders.
175
191
  */
176
192
  export declare const SCRIPT_FIELD_DESCRIPTION: {
177
- readonly speaker: "Who says the line — a character id, a bare name (a character that does not exist yet is fine), or \"VO\". Null for a shot with no words.";
178
- readonly line: "What is SAID, verbatim. Never camera, scene or prompt language — this is the half a person reads aloud.";
179
- readonly delivery: "The parenthetical: how it is said. \"(flat, exhausted)\"";
180
- readonly action: "What happens in the shot, screenplay-style. One line covering everyone in frame.";
181
- readonly prop: "The one readable object carrying the beat.";
182
- readonly shotSize: "Framing, in your own words — \"wide\", \"long-lens CU, other head blurred\". FREE TEXT: it is bucketed for the variety count and never rejected or rewritten.";
183
- readonly camera: "Camera move, in your own words — \"slow push in\", \"through the rearview, eyes only\". FREE TEXT, same rule as shotSize.";
184
- readonly continues: "True when this row's line runs on from the previous row's — one sentence split across two cuts. The signature VO move; set it deliberately.";
193
+ readonly speaker: "Speaker: character id, bare name (including a new character), or \"VO\". Null when silent.";
194
+ readonly line: "Words spoken verbatim; no camera or scene instructions.";
195
+ readonly delivery: "Optional performance note. Leave null unless a specific direction is needed; do not fill every line with stock adjectives. Not sent to TTS: put supported inline cues in the spoken text using the selected model's prompting guide.";
196
+ readonly action: "Screenplay action covering everyone in frame.";
197
+ readonly prop: "The readable object carrying the beat.";
198
+ readonly shotSize: "Free-text framing, e.g. \"wide\" or \"long-lens CU, other head blurred\"; bucketed only for variety counts.";
199
+ readonly camera: "Free-text camera move, e.g. \"slow push in\"; never rejected or rewritten.";
200
+ readonly continues: "True when this line continues the previous row: one sentence across two cuts.";
185
201
  };
186
202
  /** The script fields, as a list. Sorted from the description map so the two
187
203
  * cannot disagree — never a second hand-written array. */
@@ -223,6 +239,10 @@ export declare function effectivePrompt(spec: ShotSpec): string;
223
239
  * overlays what it actually found, so a missing field is never `undefined`
224
240
  * leaking into a request. */
225
241
  export declare function emptyShotSpec(): ShotSpec;
242
+ /** The mutually exclusive TTS source fields, shared by readers and patch merging. */
243
+ export declare const VOICE_SOURCE_FIELDS: readonly ["voiceId", "voiceReferenceAssetId", "voiceDescription"];
244
+ /** Setting a voice replaces the previous source; unrelated parameter edits preserve it. */
245
+ export declare function mergeShotParams(existing: ShotParams, patch: Record<string, unknown>): ShotParams;
226
246
  /**
227
247
  * Read a `ShotSpec` out of whatever is on disk — a row written by an older
228
248
  * build, a partial object from an op, `null`.
@@ -46,14 +46,14 @@ export const ORDERED_ATTACHMENT_ROLES = Object.keys(ORDERED_ROLE_EMISSION).sort(
46
46
  * same failure as a column nothing renders.
47
47
  */
48
48
  export const SCRIPT_FIELD_DESCRIPTION = {
49
- speaker: 'Who says the line — a character id, a bare name (a character that does not exist yet is fine), or "VO". Null for a shot with no words.',
50
- line: 'What is SAID, verbatim. Never camera, scene or prompt language — this is the half a person reads aloud.',
51
- delivery: 'The parenthetical: how it is said. "(flat, exhausted)"',
52
- action: 'What happens in the shot, screenplay-style. One line covering everyone in frame.',
53
- prop: 'The one readable object carrying the beat.',
54
- shotSize: 'Framing, in your own words — "wide", "long-lens CU, other head blurred". FREE TEXT: it is bucketed for the variety count and never rejected or rewritten.',
55
- camera: 'Camera move, in your own words — "slow push in", "through the rearview, eyes only". FREE TEXT, same rule as shotSize.',
56
- continues: 'True when this row\'s line runs on from the previous row\'s — one sentence split across two cuts. The signature VO move; set it deliberately.',
49
+ speaker: 'Speaker: character id, bare name (including a new character), or "VO". Null when silent.',
50
+ line: 'Words spoken verbatim; no camera or scene instructions.',
51
+ delivery: 'Optional performance note. Leave null unless a specific direction is needed; do not fill every line with stock adjectives. Not sent to TTS: put supported inline cues in the spoken text using the selected model\'s prompting guide.',
52
+ action: 'Screenplay action covering everyone in frame.',
53
+ prop: 'The readable object carrying the beat.',
54
+ shotSize: 'Free-text framing, e.g. "wide" or "long-lens CU, other head blurred"; bucketed only for variety counts.',
55
+ camera: 'Free-text camera move, e.g. "slow push in"; never rejected or rewritten.',
56
+ continues: 'True when this line continues the previous row: one sentence across two cuts.',
57
57
  };
58
58
  /** The seven free-text script fields (everything but the `continues` flag) —
59
59
  * the set an op accepts as `string | null` and a layer renders as text. */
@@ -151,6 +151,17 @@ export function emptyShotSpec() {
151
151
  }
152
152
  const str = (v) => (typeof v === 'string' && v.length > 0 ? v : null);
153
153
  const strArray = (v) => Array.isArray(v) ? v.filter((x) => typeof x === 'string' && x.length > 0) : [];
154
+ /** The mutually exclusive TTS source fields, shared by readers and patch merging. */
155
+ export const VOICE_SOURCE_FIELDS = ['voiceId', 'voiceReferenceAssetId', 'voiceDescription'];
156
+ /** Setting a voice replaces the previous source; unrelated parameter edits preserve it. */
157
+ export function mergeShotParams(existing, patch) {
158
+ const next = { ...existing };
159
+ if (VOICE_SOURCE_FIELDS.some((k) => typeof patch[k] === 'string' && patch[k].trim())) {
160
+ for (const k of VOICE_SOURCE_FIELDS)
161
+ delete next[k];
162
+ }
163
+ return readParams({ ...next, ...patch });
164
+ }
154
165
  function readParams(v) {
155
166
  if (!v || typeof v !== 'object')
156
167
  return {};
@@ -175,6 +186,9 @@ function readParams(v) {
175
186
  s('audioLanguage');
176
187
  s('audioAccent');
177
188
  s('negativePrompt');
189
+ s('voiceId');
190
+ s('voiceReferenceAssetId');
191
+ s('voiceDescription');
178
192
  n('duration');
179
193
  n('imageQuantity');
180
194
  n('audioDurationSeconds');
@@ -185,8 +199,24 @@ function readParams(v) {
185
199
  b('multiShot');
186
200
  b('audioLoop');
187
201
  b('audioMultilingual');
188
- if (raw.gptQuality === 'medium' || raw.gptQuality === 'high')
202
+ // Shape validation only. The pre-2026-09-09 tier MIGRATION deliberately does
203
+ // NOT live here: this file is a dependency-free leaf, and the migration's
204
+ // SSOT is `migrateGptQuality` in slate/src/shared/pricing.ts, applied in
205
+ // `transform` (slate/src/main/storage/shots.ts) — the one place a DB row
206
+ // becomes a `Shot`. NOT in `resolveShotSpec`: by the time a spec reaches that
207
+ // function its gate value (`gpt-image-2`) has already been rewritten, so a
208
+ // copy there could never fire. A stored value outside the union is dropped
209
+ // exactly as before — it just has five legal names now instead of two.
210
+ if (raw.gptQuality === 'low' ||
211
+ raw.gptQuality === 'medium' ||
212
+ raw.gptQuality === 'high' ||
213
+ raw.gptQuality === 'xhigh' ||
214
+ raw.gptQuality === 'max') {
189
215
  out.gptQuality = raw.gptQuality;
216
+ }
217
+ if (raw.gptBackground === 'auto' || raw.gptBackground === 'transparent' || raw.gptBackground === 'opaque') {
218
+ out.gptBackground = raw.gptBackground;
219
+ }
190
220
  if (raw.gridMode === 'off' || raw.gridMode === '2x2' || raw.gridMode === '3x3')
191
221
  out.gridMode = raw.gridMode;
192
222
  if (Array.isArray(raw.multiShotSegments)) {
@@ -276,6 +306,8 @@ export function shotAssetIds(spec) {
276
306
  ids.push(spec.firstFrameAssetId);
277
307
  if (spec.lastFrameAssetId)
278
308
  ids.push(spec.lastFrameAssetId);
309
+ if (spec.params.voiceReferenceAssetId)
310
+ ids.push(spec.params.voiceReferenceAssetId);
279
311
  return [...new Set(ids.filter(Boolean))];
280
312
  }
281
313
  /** Total attachment count — what a list row shows without composing anything. */