@slatesvideo/shared 0.6.9 → 0.6.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -100,6 +100,105 @@ const MINIMAX_H3_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
100
100
  * to lose track of what was actually generated.
101
101
  */
102
102
  const LTX_2_5_ASPECT_RATIOS = ['16:9', '9:16'];
103
+ export const GPT_QUALITY_TIERS = ['low', 'medium', 'high', 'xhigh', 'max'];
104
+ export const GPT_BACKGROUNDS = ['auto', 'transparent', 'opaque'];
105
+ // Product output sizes; schema bounds and metering receipt live in the GPT harvest.
106
+ export const GPT_IMAGE_25_SIZES = {
107
+ '1k': {
108
+ '1:1': { width: 1024, height: 1024 },
109
+ '16:9': { width: 1360, height: 768 },
110
+ '9:16': { width: 768, height: 1360 },
111
+ '4:3': { width: 1168, height: 880 },
112
+ '3:4': { width: 880, height: 1168 },
113
+ },
114
+ '2k': {
115
+ '1:1': { width: 1440, height: 1440 },
116
+ '16:9': { width: 1920, height: 1080 },
117
+ '9:16': { width: 1080, height: 1920 },
118
+ '4:3': { width: 1664, height: 1248 },
119
+ '3:4': { width: 1248, height: 1664 },
120
+ },
121
+ '3k': {
122
+ '1:1': { width: 1920, height: 1920 },
123
+ '16:9': { width: 2560, height: 1440 },
124
+ '9:16': { width: 1440, height: 2560 },
125
+ '4:3': { width: 2224, height: 1664 },
126
+ '3:4': { width: 1664, height: 2224 },
127
+ },
128
+ '4k': {
129
+ // 1:1 and 16:9 sit EXACTLY on the 8,294,400 ceiling — 3840×2160 is one of
130
+ // fal's own priced sizes, so the bound is inclusive. 4:3 / 3:4 are the two
131
+ // that had to move; see the constraint note above.
132
+ '1:1': { width: 2880, height: 2880 },
133
+ '16:9': { width: 3840, height: 2160 },
134
+ '9:16': { width: 2160, height: 3840 },
135
+ '4:3': { width: 3264, height: 2448 },
136
+ '3:4': { width: 2448, height: 3264 },
137
+ },
138
+ };
139
+ /**
140
+ * fal's named ~1MP presets per aspect ratio, with custom dims where fal has no
141
+ * preset. The `1k` rung of every non-GPT image model resolves through this.
142
+ */
143
+ export const FAL_1MP_SIZES = {
144
+ '1:1': 'square_hd',
145
+ '4:3': 'landscape_4_3',
146
+ '3:4': 'portrait_4_3',
147
+ '16:9': 'landscape_16_9',
148
+ '9:16': 'portrait_16_9',
149
+ '2:3': { width: 832, height: 1248 },
150
+ '3:2': { width: 1248, height: 832 },
151
+ '4:5': { width: 896, height: 1120 },
152
+ '5:4': { width: 1120, height: 896 },
153
+ '21:9': { width: 1344, height: 576 },
154
+ };
155
+ /** Pixel dims for a megapixel target at an aspect ratio, rounded to multiples of 8. */
156
+ export function computeFalDimensions(aspectRatio, targetMP) {
157
+ const parts = aspectRatio.split(':').map(Number);
158
+ const w = parts[0] || 16;
159
+ const h = parts[1] || 9;
160
+ const ratio = w / h;
161
+ const targetPixels = targetMP * 1_000_000;
162
+ return {
163
+ width: Math.round(Math.sqrt(targetPixels * ratio) / 8) * 8,
164
+ height: Math.round(Math.sqrt(targetPixels / ratio) / 8) * 8,
165
+ };
166
+ }
167
+ /**
168
+ * The `image_size` a non-GPT fal image request carries, for one model × aspect ×
169
+ * resolution rung.
170
+ *
171
+ * 🚨 THIS IS A BILLING INPUT, WHICH IS WHY IT LIVES HERE (moved out of
172
+ * slate/src/main/api/fal.ts, 2026-09-10). The resolution rung is a segment of
173
+ * every image cost key, and nothing in the REQUEST names it — fal is told pixel
174
+ * dimensions, not "2k". The proxy therefore recovers the rung by running this
175
+ * function over the model's declared `imageResolutions` × `aspectRatios` and
176
+ * matching the body's `image_size`, exactly as it recovers a GPT Image rung from
177
+ * `GPT_IMAGE_25_SIZES`. A second copy of this arithmetic would mean the desktop
178
+ * and the server could disagree about what a request is worth, silently.
179
+ *
180
+ * GPT Image does NOT come through here — that family carries explicit pixel
181
+ * classes in `GPT_IMAGE_25_SIZES` and an explicit `quality` rung.
182
+ */
183
+ export function falImageSize(model, aspectRatio, resolution) {
184
+ const ar = aspectRatio || '16:9';
185
+ const isSeedream5 = model === 'seedream-5-lite';
186
+ const res = resolution || (isSeedream5 ? '2k' : '1k');
187
+ // 1K (~1MP): fal's named presets, or small custom dims where there is none.
188
+ if (res === '1k') {
189
+ return FAL_1MP_SIZES[ar] || 'landscape_16_9';
190
+ }
191
+ // Seedream 5 Lite: custom dims must be ≥3.69MP (2560×1440) and ≤9.44MP
192
+ // (3072×3072), so its three rungs target 4 / 7 / 9 MP — all inside that band.
193
+ if (isSeedream5) {
194
+ if (res === '4k')
195
+ return computeFalDimensions(ar, 9);
196
+ return res === '3k' ? computeFalDimensions(ar, 7) : computeFalDimensions(ar, 4);
197
+ }
198
+ if (res === '2k')
199
+ return computeFalDimensions(ar, 2);
200
+ return computeFalDimensions(ar, 4);
201
+ }
103
202
  /**
104
203
  * The provider every AGENT generation actually lands on for Kling and Veo.
105
204
  *
@@ -122,14 +221,22 @@ export const AGENT_ROUTE_PROVIDER = 'fal';
122
221
  export const MODEL_CAPABILITIES = {
123
222
  // ── Image models ───────────────────────────────────────────────────────────
124
223
  'nano-banana-2': {
224
+ imageResolutions: ['1k', '2k', '4k'],
225
+ // fal's nano-banana-2 schema caps `num_images` at 4 (read 2026-09-09). This
226
+ // is the ONLY model that batches: the MCP's headless path (no projectId) asks
227
+ // fal for one batch, and every other route — desktop and agent alike — fires
228
+ // N separate single-image generations. The proxy bills the batch size.
229
+ maxBatchImages: 4,
125
230
  aspectRatios: FULL_ASPECT_RATIOS,
126
231
  maxRefImages: 14,
127
232
  },
128
233
  'nano-banana-2-lite': {
234
+ imageResolutions: ['1k'],
129
235
  aspectRatios: FULL_ASPECT_RATIOS,
130
236
  maxRefImages: 4, // fal edit endpoint caps input images at 4
131
237
  },
132
238
  'nano-banana-pro': {
239
+ imageResolutions: ['1k', '2k', '4k'],
133
240
  aspectRatios: FULL_ASPECT_RATIOS,
134
241
  maxRefImages: 14,
135
242
  },
@@ -169,18 +276,22 @@ export const MODEL_CAPABILITIES = {
169
276
  // limits either: the MCP's 4,000-character prompt against fal's 32,000, and
170
277
  // image quantity, which is a fan-out and has no provider ceiling at all.
171
278
  'gpt-image-2-5-flare': {
279
+ imageResolutions: ['2k', '3k', '4k'],
172
280
  aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
173
281
  maxRefImages: 16,
174
282
  },
175
283
  'gpt-image-2-5-sunburst': {
284
+ imageResolutions: ['2k', '3k', '4k'],
176
285
  aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
177
286
  maxRefImages: 16,
178
287
  },
179
288
  'flux-2-max': {
289
+ imageResolutions: ['1k', '2k', '4k'],
180
290
  aspectRatios: FULL_ASPECT_RATIOS,
181
291
  maxRefImages: 4,
182
292
  },
183
293
  'seedream-5-lite': {
294
+ imageResolutions: ['2k', '3k', '4k'],
184
295
  aspectRatios: FULL_ASPECT_RATIOS,
185
296
  maxRefImages: 10,
186
297
  },
@@ -377,6 +488,9 @@ export const MODEL_CAPABILITIES = {
377
488
  // the base row's branch — a different ladder AND a different price at the one
378
489
  // tier they share. Every lookup downstream is an exact-id map, not a prefix.
379
490
  'minimax-h3': {
491
+ // fal reference-to-video schema, 2026-09-09: each audio clip is 2-15s.
492
+ referenceAudioDuration: { min: 2, max: 15 },
493
+ referenceVideoDuration: { min: 2, max: 15 },
380
494
  aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
381
495
  // The full ladder. 480p/768p are NATIVE generation modes; 2K and 4K upscale
382
496
  // a 768p base result through H3-Regenerate-2K, which is API-only and not in
@@ -408,6 +522,9 @@ export const MODEL_CAPABILITIES = {
408
522
  maxReferenceAudioSeconds: 15,
409
523
  },
410
524
  'minimax-h3-max': {
525
+ // fal reference-to-video schema, 2026-09-09: each audio clip is 2-15s.
526
+ referenceAudioDuration: { min: 2, max: 15 },
527
+ referenceVideoDuration: { min: 2, max: 15 },
411
528
  aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
412
529
  // 🚨 REFERENCES LANDED 2026-09-09, AFTER A FALSE CLAIM WAS RETIRED. This row
413
530
  // shipped from v1.5.5 declaring zero reference capacity because a comment
@@ -431,15 +548,7 @@ export const MODEL_CAPABILITIES = {
431
548
  // arms — quoted off this endpoint, not inherited.
432
549
  maxReferenceVideoSeconds: 15,
433
550
  maxReferenceAudioSeconds: 15,
434
- // 1080P IS REAL ON THIS ROW and was missing until 2026-09-09. The schema's
435
- // resolution enum is ["480P","768P","1080P"] on all three h3-max endpoints.
436
- // 2K/4K genuinely are absent: the H3-Regenerate-2K upscaler is API-only and
437
- // is not in the open weights fal self-hosts, which is the actual mechanism
438
- // behind the shorter ladder — 1080p was never part of that story.
439
- //
440
- // DEFAULT stays 768p: it is the tier the model natively generates, and
441
- // 1080p is a 2x price step ($0.160/s against $0.080/s).
442
- videoResolution: { options: ['480p', '768p', '1080p'], default: '768p' },
551
+ videoResolution: { options: ['480p', '768p'], default: '768p' },
443
552
  duration: { min: 5, max: 15, mode: 'continuous' },
444
553
  },
445
554
  // ── LTX-2.5 (both seats on fal — added 2026-08-29) ─────────────────────────
@@ -817,4 +926,28 @@ export function describeReferenceImageCaps(models) {
817
926
  return n === 0 ? '0 (prompt + source clip only)' : String(n);
818
927
  });
819
928
  }
929
+ /** H3 Max reference accounting, fal's worked tables read 2026-09-09.
930
+ * https://fal.ai/models/minimax/h3-max/reference-to-video
931
+ * 1080p video-reference pricing is unpublished; never infer it from output rates.
932
+ */
933
+ export const MINIMAX_MAX_REFERENCE = {
934
+ freeTokens: 4096,
935
+ imagePixelsPerToken: 1024,
936
+ normalizedImageEdge: 1024,
937
+ audioTokensPerSecond: 80,
938
+ videoTokensPerSecond: { '480p': 2886, '768p': 7459.2 },
939
+ };
940
+ export function minimaxMaxReferenceTokens(input) {
941
+ const rate = MINIMAX_MAX_REFERENCE.videoTokensPerSecond[input.resolution];
942
+ if (input.videoSeconds > 0 && rate === undefined) {
943
+ throw new Error(`H3 Max video-reference pricing is unavailable at ${input.resolution}; choose a priced resolution.`);
944
+ }
945
+ for (const n of [input.imagePixels, input.videoSeconds, input.audioSeconds]) {
946
+ if (!Number.isFinite(n) || n < 0)
947
+ throw new Error('Reference metadata must be finite and nonnegative');
948
+ }
949
+ return Math.max(0, Math.ceil(input.imagePixels / MINIMAX_MAX_REFERENCE.imagePixelsPerToken +
950
+ input.videoSeconds * (rate ?? 0) + input.audioSeconds * MINIMAX_MAX_REFERENCE.audioTokensPerSecond -
951
+ MINIMAX_MAX_REFERENCE.freeTokens));
952
+ }
820
953
  //# sourceMappingURL=model-capabilities.js.map
@@ -14,7 +14,21 @@ export interface ReferenceGroup {
14
14
  /** Display + citation name: 'Marcus' | 'the cafe' | 'noir'. Used verbatim. */
15
15
  name: string;
16
16
  kind: ReferenceKind;
17
- /** A group can carry several images for workflows that genuinely need them. */
17
+ /**
18
+ * A group can carry several images for workflows that genuinely need them.
19
+ *
20
+ * 🚨 A `character` GROUP MAY ALSO CARRY ONE `audio` MEDIUM — that character's
21
+ * assigned VOICE (2026-09-09). It is the same idea as the identity image, on
22
+ * the other axis of identity: the mention attaches what the character IS, and
23
+ * a voice is part of that. The audio takes its number from the audio counter,
24
+ * so adding one renumbers no image, and it is cited INLINE beside her name
25
+ * ("Sarah (image 1, voice timbre from audio 1)") rather than as the neutral
26
+ * `audio-ref` sentence — step 3e holds the receipt for why that is legal.
27
+ *
28
+ * A character group with a voice and NO identity image is a real state, not a
29
+ * defect: `voice-without-photo` is legal on any model that reads audio alone
30
+ * (Seedance 2.5). It cites no image and the timbre line carries the binding.
31
+ */
18
32
  media: ReferenceMedia[];
19
33
  /**
20
34
  * What is SAID in this group's reference audio, typed by the user.
@@ -37,6 +37,23 @@ function citeImages(nums) {
37
37
  const noun = nums.length === 1 ? 'image' : 'images';
38
38
  return `${noun} ${joinNums(nums)}`;
39
39
  }
40
+ /**
41
+ * "voice timbre from audio 1" — a character's VOICE, cited inline beside her
42
+ * name (lowercase, for inline use, exactly like `citeImages`).
43
+ *
44
+ * 🚨 THE ROLE WORDS ARE THE LOAD-BEARING HALF, not decoration. A bare
45
+ * "(image 1, audio 1)" would be the UNROLED state: BytePlus's capability table
46
+ * gives an audio reference five possible jobs — "music, dialogue, voice, tone,
47
+ * or timbre" — and an unroled clip falls back to DIALOGUE, so the model
48
+ * transcribes it and speaks ITS words instead of the prompt's. That is the
49
+ * shipped defect where a supplied take came back as "a map called Slates" for
50
+ * "an app called Slates" (2026-08-28). "voice timbre" is the vendor's own
51
+ * phrase for the half we want: the sound of her, not her words.
52
+ */
53
+ function citeVoice(nums) {
54
+ const noun = nums.length === 1 ? 'audio' : 'audios';
55
+ return `voice timbre from ${noun} ${joinNums(nums)}`;
56
+ }
40
57
  function joinNums(nums) {
41
58
  if (nums.length === 1)
42
59
  return String(nums[0]);
@@ -140,7 +157,12 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
140
157
  videoNums.push(videoNum);
141
158
  orderedVideoPaths.push(m.path);
142
159
  }
143
- else if (m.mediaKind === 'audio' && g.kind === 'audio-ref') {
160
+ else if (m.mediaKind === 'audio' && (g.kind === 'audio-ref' || g.kind === 'character')) {
161
+ // ONE audio counter across hand-attached clips and a CHARACTER'S VOICE,
162
+ // for the same reason the video counter is shared: the two are the same
163
+ // numbered space on the wire, and a second counter would emit two
164
+ // "Audio 1"s the moment a request carried both. Which SENTENCE names
165
+ // the clip is what differs (step 3 vs step 3e), never the number.
144
166
  audioNum += 1;
145
167
  audioNums.push(audioNum);
146
168
  orderedAudioPaths.push(m.path);
@@ -208,7 +230,30 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
208
230
  return ''; // styles never inline — trailing clause only
209
231
  if (!seenFirst.has(key)) {
210
232
  seenFirst.add(key);
211
- return `${g.name} (${citeImages(g.imageNums)})`;
233
+ // 🚨 ONE BINDING SITE PER ENTITY, CARRYING EVERY MEDIUM SHE OWNS
234
+ // (2026-09-09). The mention attaches her face AND her voice, so both are
235
+ // cited where her name appears rather than one inline and the other in a
236
+ // preamble sentence above the user's own words. That split was the first
237
+ // shape this shipped in, and it read backwards: a two-speaker prompt made
238
+ // you hold two name→audio mappings in your head before you reached the
239
+ // sentence, and a character with a voice and no photo said her name twice
240
+ // while citing nothing.
241
+ //
242
+ // It is also closer to the vendor, not further. BytePlus's binding
243
+ // example is ONE sentence covering both media — "Image 1 depicts the
244
+ // protagonist John and uses the voice timbre from Audio 1." — and the
245
+ // preamble form had already split it in half.
246
+ //
247
+ // 🚨 EMPTY MEANS OMITTED, NEVER AN EMPTY PARENTHESIS. A voice-only
248
+ // character has no `imageNums` and `citeImages([])` would compose the
249
+ // literal "images " — a citation pointing at nothing, inside the one
250
+ // function whose whole job is that citations point at what is sent.
251
+ const cites = [];
252
+ if (g.imageNums.length > 0)
253
+ cites.push(citeImages(g.imageNums));
254
+ if (g.audioNums.length > 0)
255
+ cites.push(citeVoice(g.audioNums));
256
+ return cites.length > 0 ? `${g.name} (${cites.join(', ')})` : g.name;
212
257
  }
213
258
  return g.name;
214
259
  });
@@ -239,25 +284,6 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
239
284
  topKeys.push(`${noun} ${joinNums(g.videoNums)} ${tail}`);
240
285
  }
241
286
  }
242
- // Reference audio ("Audio 1 is a provided reference."), plus THE WORDS when
243
- // the user has typed them — see ReferenceGroup.spokenText for why the words
244
- // have to travel as text as well as audio.
245
- for (const g of numbered) {
246
- if (g.kind === 'audio-ref' && g.audioNums.length > 0) {
247
- const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
248
- const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
249
- topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
250
- // Trimmed, never rewritten: the words between the quotes are the user's
251
- // exactly as typed. The delimiters are CURLY on purpose — a straight
252
- // quote inside the user's own line then sits beside them without
253
- // colliding, so nothing has to be escaped and nothing is edited.
254
- const spoken = (g.spokenText ?? '').trim();
255
- if (spoken) {
256
- const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
257
- topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: “${spoken}”`);
258
- }
259
- }
260
- }
261
287
  // 🚨 A PINNED REFERENCE IMAGE GETS NO KEY LINE, DELIBERATELY (2026-08-10).
262
288
  // It used to emit "Image 1 is a provided reference." — the only branch here
263
289
  // that assigns NO role, and therefore says nothing: every image in the request
@@ -295,6 +321,104 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
295
321
  }
296
322
  }
297
323
  }
324
+ // ── Every AUDIO line, in one pass, in audio-NUMBER order ────────────────
325
+ //
326
+ // 🚨 AFTER the subject lines, deliberately. A key line that names a subject
327
+ // ("Image 1 is Marcus.") has to come before a line that gives that subject's
328
+ // media a job, or the prompt describes a voice before it says whose face it
329
+ // belongs to. Binding is carried by the SENTENCE rather than by adjacency —
330
+ // the vendor states that outright — so nothing on the wire depends on this;
331
+ // what depends on it is whether the composed preview can be read top to
332
+ // bottom, and that preview is the surface the transparency invariant rests
333
+ // on.
334
+ //
335
+ // 🚨 ONE LOOP OVER THE GROUPS, NOT ONE LOOP PER KIND, AND THAT IS THE WHOLE
336
+ // POINT OF ITS SHAPE. There are two audio sentences — a hand-attached clip's
337
+ // neutral "Audio N is a provided reference." and a character's roled
338
+ // "Sarah uses the voice timbre from Audio N." — and they draw their numbers
339
+ // from the SAME counter walking THIS list. Emitting them in two passes
340
+ // printed them in kind order instead of number order, so a request with two
341
+ // voices and one room-tone clip opened with "Audio 3 is a provided
342
+ // reference." and named Audio 1 and Audio 2 after it. Nothing was wrong on
343
+ // the wire — binding is carried by the sentence, not by adjacency, which the
344
+ // vendor states outright — but a prompt that counts backwards is a prompt
345
+ // nobody can proofread, and the composed preview is the surface the whole
346
+ // transparency invariant rests on. One pass over `numbered` IS number order,
347
+ // because the numbers were assigned by the same walk.
348
+ for (const g of numbered) {
349
+ if (g.audioNums.length === 0)
350
+ continue;
351
+ const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
352
+ if (g.kind === 'character') {
353
+ // A CHARACTER'S VOICE. When her token appears in the prompt the binding
354
+ // rides INLINE on her name (step 2) and there is nothing to say up here.
355
+ // This is the same duality her IMAGE already has — "Sarah (image 1)"
356
+ // inline versus "Image 1 is Sarah." when the prompt never names her — so
357
+ // it is the existing pattern rather than a second grammar.
358
+ //
359
+ // Emitted from THIS pass, not from a block of its own, so it keeps its
360
+ // place in audio-NUMBER order among the neutral lines. See the header.
361
+ const namedInPrompt = g.token && matchedInPrompt.has(normToken(g.token));
362
+ if (!namedInPrompt) {
363
+ topKeys.push(`${g.name} uses the ${citeVoice(g.audioNums)}.`);
364
+ }
365
+ continue;
366
+ }
367
+ if (g.kind !== 'audio-ref')
368
+ continue;
369
+ // A clip the user dragged on declared no role, so it keeps the neutral
370
+ // line, plus THE WORDS when the user has typed them — see
371
+ // ReferenceGroup.spokenText for why the words have to travel as text as
372
+ // well as audio.
373
+ const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
374
+ topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
375
+ // Trimmed, never rewritten: the words between the quotes are the user's
376
+ // exactly as typed. The delimiters are CURLY on purpose — a straight
377
+ // quote inside the user's own line then sits beside them without
378
+ // colliding, so nothing has to be escaped and nothing is edited.
379
+ const spoken = (g.spokenText ?? '').trim();
380
+ if (spoken) {
381
+ const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
382
+ topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: “${spoken}”`);
383
+ }
384
+ }
385
+ // ── 3e. WHY A CHARACTER'S VOICE GETS A ROLE AT ALL (2026-09-09) ──────────
386
+ //
387
+ // The binding itself is composed INLINE beside her name (step 2), or as a
388
+ // fallback line in the audio pass above when the prompt never names her.
389
+ // This is the receipt for why composing a role is legal at all.
390
+ //
391
+ // 🚨 AUDIO IS THE ONE MODALITY WHERE THE NEUTRAL LINE UNDER-SPECIFIES, and
392
+ // this is the sentence that closes it. An image is definitionally a
393
+ // reference and a video has two possible roles, so both are settled by a
394
+ // neutral line. An audio attachment has FIVE — BytePlus's own capability
395
+ // table lists "music, dialogue, voice, tone, or timbre" — so
396
+ // "Audio 1 is a provided reference." distinguishes a clip from nothing while
397
+ // leaving four roles open, and an unroled clip falls back to DIALOGUE: the
398
+ // model re-transcribes it and speaks ITS words. That is the shipped defect
399
+ // where a supplied take came back as "a map called Slates" for "an app
400
+ // called Slates" (2026-08-28).
401
+ //
402
+ // The wording is the vendor's, not ours. BytePlus's own binding sentence is
403
+ // "Image 1 depicts the protagonist John and uses the voice timbre from
404
+ // Audio 1."; MiniMax builds the same primitive into H3's notation
405
+ // ("<Audio 1> is the voice-timbre reference for <Subject 1>"). Two vendors,
406
+ // independently. The DIALOGUE therefore comes from the prompt and the clip
407
+ // carries only the voice — receipts and line refs:
408
+ // second-brain/business/projects/slates/research/model-prompting-research.md
409
+ // § 2026-09-09 Multimodal reference GRAMMAR, facts 2 and 3.
410
+ //
411
+ // 🚨 IT IS LEGAL COMPOSITION ONLY BECAUSE THE ROLE WAS DECLARED. Assigning a
412
+ // voice to a character IS the declaration; a clip dragged onto the rail is
413
+ // not, and keeps the neutral line above. Inferring a role nobody declared
414
+ // stays forbidden (`slate/.claude/rules/prompt-surface.md`).
415
+ //
416
+ // The citation is lowercase (`voice timbre from audio 1`) like every other
417
+ // inline citation this composer emits. An earlier draft capitalised it to
418
+ // match the vendor's example verbatim, which left a single capitalised
419
+ // `Audio 1` sitting mid-sentence among lowercase `image 1`s; moving the
420
+ // binding inline removed the reason for the exception along with the
421
+ // exception.
298
422
  // ── 4. Style trailing clause (one, at the end — style reads best last) ──
299
423
  const styleNums = [];
300
424
  for (const g of numbered) {
@@ -1,3 +1,4 @@
1
+ import { type GptQuality, type GptBackground } from './model-capabilities.js';
1
2
  /**
2
3
  * Role an attachment carries in the composer tray. User-set, never inferred.
3
4
  *
@@ -62,13 +63,15 @@ export interface ShotParams {
62
63
  quality?: string;
63
64
  /** GPT Image 2.5's tier. Always sent explicitly: fal's own default is
64
65
  * `high`, which is the third of five rungs, not the top of two. */
65
- gptQuality?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
66
+ gptQuality?: GptQuality;
66
67
  /** GPT Image's alpha switch. `auto` is fal's default and ours; `transparent`
67
68
  * asks for a real alpha channel rather than a painted backdrop. Costs
68
69
  * nothing — fal prices this family on size × quality only, so it is NOT a
69
70
  * cost-key segment. Named `gptBackground` because `background` already
70
71
  * means "generate asynchronously" on every op that carries a Shot. */
71
- gptBackground?: 'auto' | 'transparent' | 'opaque';
72
+ gptBackground?: GptBackground;
73
+ /** Character voices explicitly removed from this recipe. */
74
+ detachedVoiceCharacterIds?: string[];
72
75
  duration?: number;
73
76
  imageQuantity?: number;
74
77
  gridMode?: 'off' | '2x2' | '3x3';
@@ -1,22 +1,4 @@
1
- // The Shot — the prompt bar, serialized.
2
- //
3
- // THE PRINCIPLE: a generation's full recipe already exists (the desktop writes
4
- // `referenceGroups` into every `settings_json`), but only as a byproduct of
5
- // spending money on it. This module gives that structure a NAME, so it can be
6
- // listed, forked, agent-authored and restored without loss — before anything
7
- // has been generated.
8
- //
9
- // This is the canonical implementation. It is mirrored byte-for-byte into the
10
- // desktop app's `slate/src/shared/shotSpec.ts` (the desktop installs the
11
- // published @slatesvideo/shared from npm and cannot file-import this source, so
12
- // the mirror carries a header pointing here — the same rule
13
- // `reference-composer.ts` follows). `slate/scripts/composer-mirror-check.mjs`
14
- // asserts the two agree; do not invent a second sync mechanism.
15
- //
16
- // 🚨 KEEP THIS A DEPENDENCY-FREE LEAF. It imports nothing, in either repo. The
17
- // desktop's renderer bundles its mirror, the desktop's MAIN process reads it,
18
- // and the op surface here builds Zod schemas from it — a single `node:` import
19
- // would break the first of those.
1
+ import { GPT_BACKGROUNDS } from './model-capabilities.js';
20
2
  /**
21
3
  * Emission ORDER of the ordered roles — the order `buildReferenceGroups` pushes
22
4
  * them in, which is the order the composer numbers them in, which is the order
@@ -214,7 +196,10 @@ function readParams(v) {
214
196
  raw.gptQuality === 'max') {
215
197
  out.gptQuality = raw.gptQuality;
216
198
  }
217
- if (raw.gptBackground === 'auto' || raw.gptBackground === 'transparent' || raw.gptBackground === 'opaque') {
199
+ if (Array.isArray(raw.detachedVoiceCharacterIds)) {
200
+ out.detachedVoiceCharacterIds = strArray(raw.detachedVoiceCharacterIds);
201
+ }
202
+ if (GPT_BACKGROUNDS.includes(raw.gptBackground)) {
218
203
  out.gptBackground = raw.gptBackground;
219
204
  }
220
205
  if (raw.gridMode === 'off' || raw.gridMode === '2x2' || raw.gridMode === '3x3')