@slatesvideo/shared 0.6.9 β†’ 0.6.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -37,6 +37,23 @@ function citeImages(nums) {
37
37
  const noun = nums.length === 1 ? 'image' : 'images';
38
38
  return `${noun} ${joinNums(nums)}`;
39
39
  }
40
+ /**
41
+ * "voice timbre from audio 1" β€” a character's VOICE, cited inline beside her
42
+ * name (lowercase, for inline use, exactly like `citeImages`).
43
+ *
44
+ * 🚨 THE ROLE WORDS ARE THE LOAD-BEARING HALF, not decoration. A bare
45
+ * "(image 1, audio 1)" would be the UNROLED state: BytePlus's capability table
46
+ * gives an audio reference five possible jobs β€” "music, dialogue, voice, tone,
47
+ * or timbre" β€” and an unroled clip falls back to DIALOGUE, so the model
48
+ * transcribes it and speaks ITS words instead of the prompt's. That is the
49
+ * shipped defect where a supplied take came back as "a map called Slates" for
50
+ * "an app called Slates" (2026-08-28). "voice timbre" is the vendor's own
51
+ * phrase for the half we want: the sound of her, not her words.
52
+ */
53
+ function citeVoice(nums) {
54
+ const noun = nums.length === 1 ? 'audio' : 'audios';
55
+ return `voice timbre from ${noun} ${joinNums(nums)}`;
56
+ }
40
57
  function joinNums(nums) {
41
58
  if (nums.length === 1)
42
59
  return String(nums[0]);
@@ -140,7 +157,12 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
140
157
  videoNums.push(videoNum);
141
158
  orderedVideoPaths.push(m.path);
142
159
  }
143
- else if (m.mediaKind === 'audio' && g.kind === 'audio-ref') {
160
+ else if (m.mediaKind === 'audio' && (g.kind === 'audio-ref' || g.kind === 'character')) {
161
+ // ONE audio counter across hand-attached clips and a CHARACTER'S VOICE,
162
+ // for the same reason the video counter is shared: the two are the same
163
+ // numbered space on the wire, and a second counter would emit two
164
+ // "Audio 1"s the moment a request carried both. Which SENTENCE names
165
+ // the clip is what differs (step 3 vs step 3e), never the number.
144
166
  audioNum += 1;
145
167
  audioNums.push(audioNum);
146
168
  orderedAudioPaths.push(m.path);
@@ -208,7 +230,30 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
208
230
  return ''; // styles never inline β€” trailing clause only
209
231
  if (!seenFirst.has(key)) {
210
232
  seenFirst.add(key);
211
- return `${g.name} (${citeImages(g.imageNums)})`;
233
+ // 🚨 ONE BINDING SITE PER ENTITY, CARRYING EVERY MEDIUM SHE OWNS
234
+ // (2026-09-09). The mention attaches her face AND her voice, so both are
235
+ // cited where her name appears rather than one inline and the other in a
236
+ // preamble sentence above the user's own words. That split was the first
237
+ // shape this shipped in, and it read backwards: a two-speaker prompt made
238
+ // you hold two name→audio mappings in your head before you reached the
239
+ // sentence, and a character with a voice and no photo said her name twice
240
+ // while citing nothing.
241
+ //
242
+ // It is also closer to the vendor, not further. BytePlus's binding
243
+ // example is ONE sentence covering both media β€” "Image 1 depicts the
244
+ // protagonist John and uses the voice timbre from Audio 1." β€” and the
245
+ // preamble form had already split it in half.
246
+ //
247
+ // 🚨 EMPTY MEANS OMITTED, NEVER AN EMPTY PARENTHESIS. A voice-only
248
+ // character has no `imageNums` and `citeImages([])` would compose the
249
+ // literal "images " β€” a citation pointing at nothing, inside the one
250
+ // function whose whole job is that citations point at what is sent.
251
+ const cites = [];
252
+ if (g.imageNums.length > 0)
253
+ cites.push(citeImages(g.imageNums));
254
+ if (g.audioNums.length > 0)
255
+ cites.push(citeVoice(g.audioNums));
256
+ return cites.length > 0 ? `${g.name} (${cites.join(', ')})` : g.name;
212
257
  }
213
258
  return g.name;
214
259
  });
@@ -239,25 +284,6 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
239
284
  topKeys.push(`${noun} ${joinNums(g.videoNums)} ${tail}`);
240
285
  }
241
286
  }
242
- // Reference audio ("Audio 1 is a provided reference."), plus THE WORDS when
243
- // the user has typed them β€” see ReferenceGroup.spokenText for why the words
244
- // have to travel as text as well as audio.
245
- for (const g of numbered) {
246
- if (g.kind === 'audio-ref' && g.audioNums.length > 0) {
247
- const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
248
- const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
249
- topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
250
- // Trimmed, never rewritten: the words between the quotes are the user's
251
- // exactly as typed. The delimiters are CURLY on purpose β€” a straight
252
- // quote inside the user's own line then sits beside them without
253
- // colliding, so nothing has to be escaped and nothing is edited.
254
- const spoken = (g.spokenText ?? '').trim();
255
- if (spoken) {
256
- const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
257
- topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: β€œ${spoken}”`);
258
- }
259
- }
260
- }
261
287
  // 🚨 A PINNED REFERENCE IMAGE GETS NO KEY LINE, DELIBERATELY (2026-08-10).
262
288
  // It used to emit "Image 1 is a provided reference." β€” the only branch here
263
289
  // that assigns NO role, and therefore says nothing: every image in the request
@@ -295,6 +321,104 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
295
321
  }
296
322
  }
297
323
  }
324
+ // ── Every AUDIO line, in one pass, in audio-NUMBER order ────────────────
325
+ //
326
+ // 🚨 AFTER the subject lines, deliberately. A key line that names a subject
327
+ // ("Image 1 is Marcus.") has to come before a line that gives that subject's
328
+ // media a job, or the prompt describes a voice before it says whose face it
329
+ // belongs to. Binding is carried by the SENTENCE rather than by adjacency β€”
330
+ // the vendor states that outright β€” so nothing on the wire depends on this;
331
+ // what depends on it is whether the composed preview can be read top to
332
+ // bottom, and that preview is the surface the transparency invariant rests
333
+ // on.
334
+ //
335
+ // 🚨 ONE LOOP OVER THE GROUPS, NOT ONE LOOP PER KIND, AND THAT IS THE WHOLE
336
+ // POINT OF ITS SHAPE. There are two audio sentences β€” a hand-attached clip's
337
+ // neutral "Audio N is a provided reference." and a character's roled
338
+ // "Sarah uses the voice timbre from Audio N." β€” and they draw their numbers
339
+ // from the SAME counter walking THIS list. Emitting them in two passes
340
+ // printed them in kind order instead of number order, so a request with two
341
+ // voices and one room-tone clip opened with "Audio 3 is a provided
342
+ // reference." and named Audio 1 and Audio 2 after it. Nothing was wrong on
343
+ // the wire β€” binding is carried by the sentence, not by adjacency, which the
344
+ // vendor states outright β€” but a prompt that counts backwards is a prompt
345
+ // nobody can proofread, and the composed preview is the surface the whole
346
+ // transparency invariant rests on. One pass over `numbered` IS number order,
347
+ // because the numbers were assigned by the same walk.
348
+ for (const g of numbered) {
349
+ if (g.audioNums.length === 0)
350
+ continue;
351
+ const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
352
+ if (g.kind === 'character') {
353
+ // A CHARACTER'S VOICE. When her token appears in the prompt the binding
354
+ // rides INLINE on her name (step 2) and there is nothing to say up here.
355
+ // This is the same duality her IMAGE already has β€” "Sarah (image 1)"
356
+ // inline versus "Image 1 is Sarah." when the prompt never names her β€” so
357
+ // it is the existing pattern rather than a second grammar.
358
+ //
359
+ // Emitted from THIS pass, not from a block of its own, so it keeps its
360
+ // place in audio-NUMBER order among the neutral lines. See the header.
361
+ const namedInPrompt = g.token && matchedInPrompt.has(normToken(g.token));
362
+ if (!namedInPrompt) {
363
+ topKeys.push(`${g.name} uses the ${citeVoice(g.audioNums)}.`);
364
+ }
365
+ continue;
366
+ }
367
+ if (g.kind !== 'audio-ref')
368
+ continue;
369
+ // A clip the user dragged on declared no role, so it keeps the neutral
370
+ // line, plus THE WORDS when the user has typed them β€” see
371
+ // ReferenceGroup.spokenText for why the words have to travel as text as
372
+ // well as audio.
373
+ const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
374
+ topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
375
+ // Trimmed, never rewritten: the words between the quotes are the user's
376
+ // exactly as typed. The delimiters are CURLY on purpose β€” a straight
377
+ // quote inside the user's own line then sits beside them without
378
+ // colliding, so nothing has to be escaped and nothing is edited.
379
+ const spoken = (g.spokenText ?? '').trim();
380
+ if (spoken) {
381
+ const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
382
+ topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: β€œ${spoken}”`);
383
+ }
384
+ }
385
+ // ── 3e. WHY A CHARACTER'S VOICE GETS A ROLE AT ALL (2026-09-09) ──────────
386
+ //
387
+ // The binding itself is composed INLINE beside her name (step 2), or as a
388
+ // fallback line in the audio pass above when the prompt never names her.
389
+ // This is the receipt for why composing a role is legal at all.
390
+ //
391
+ // 🚨 AUDIO IS THE ONE MODALITY WHERE THE NEUTRAL LINE UNDER-SPECIFIES, and
392
+ // this is the sentence that closes it. An image is definitionally a
393
+ // reference and a video has two possible roles, so both are settled by a
394
+ // neutral line. An audio attachment has FIVE β€” BytePlus's own capability
395
+ // table lists "music, dialogue, voice, tone, or timbre" β€” so
396
+ // "Audio 1 is a provided reference." distinguishes a clip from nothing while
397
+ // leaving four roles open, and an unroled clip falls back to DIALOGUE: the
398
+ // model re-transcribes it and speaks ITS words. That is the shipped defect
399
+ // where a supplied take came back as "a map called Slates" for "an app
400
+ // called Slates" (2026-08-28).
401
+ //
402
+ // The wording is the vendor's, not ours. BytePlus's own binding sentence is
403
+ // "Image 1 depicts the protagonist John and uses the voice timbre from
404
+ // Audio 1."; MiniMax builds the same primitive into H3's notation
405
+ // ("<Audio 1> is the voice-timbre reference for <Subject 1>"). Two vendors,
406
+ // independently. The DIALOGUE therefore comes from the prompt and the clip
407
+ // carries only the voice β€” receipts and line refs:
408
+ // second-brain/business/projects/slates/research/model-prompting-research.md
409
+ // Β§ 2026-09-09 Multimodal reference GRAMMAR, facts 2 and 3.
410
+ //
411
+ // 🚨 IT IS LEGAL COMPOSITION ONLY BECAUSE THE ROLE WAS DECLARED. Assigning a
412
+ // voice to a character IS the declaration; a clip dragged onto the rail is
413
+ // not, and keeps the neutral line above. Inferring a role nobody declared
414
+ // stays forbidden (`slate/.claude/rules/prompt-surface.md`).
415
+ //
416
+ // The citation is lowercase (`voice timbre from audio 1`) like every other
417
+ // inline citation this composer emits. An earlier draft capitalised it to
418
+ // match the vendor's example verbatim, which left a single capitalised
419
+ // `Audio 1` sitting mid-sentence among lowercase `image 1`s; moving the
420
+ // binding inline removed the reason for the exception along with the
421
+ // exception.
298
422
  // ── 4. Style trailing clause (one, at the end β€” style reads best last) ──
299
423
  const styleNums = [];
300
424
  for (const g of numbered) {
@@ -1,3 +1,4 @@
1
+ import { type GptQuality, type GptBackground } from './model-capabilities.js';
1
2
  /**
2
3
  * Role an attachment carries in the composer tray. User-set, never inferred.
3
4
  *
@@ -62,13 +63,15 @@ export interface ShotParams {
62
63
  quality?: string;
63
64
  /** GPT Image 2.5's tier. Always sent explicitly: fal's own default is
64
65
  * `high`, which is the third of five rungs, not the top of two. */
65
- gptQuality?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
66
+ gptQuality?: GptQuality;
66
67
  /** GPT Image's alpha switch. `auto` is fal's default and ours; `transparent`
67
68
  * asks for a real alpha channel rather than a painted backdrop. Costs
68
69
  * nothing β€” fal prices this family on size Γ— quality only, so it is NOT a
69
70
  * cost-key segment. Named `gptBackground` because `background` already
70
71
  * means "generate asynchronously" on every op that carries a Shot. */
71
- gptBackground?: 'auto' | 'transparent' | 'opaque';
72
+ gptBackground?: GptBackground;
73
+ /** Character voices explicitly removed from this recipe. */
74
+ detachedVoiceCharacterIds?: string[];
72
75
  duration?: number;
73
76
  imageQuantity?: number;
74
77
  gridMode?: 'off' | '2x2' | '3x3';
@@ -1,22 +1,4 @@
1
- // The Shot β€” the prompt bar, serialized.
2
- //
3
- // THE PRINCIPLE: a generation's full recipe already exists (the desktop writes
4
- // `referenceGroups` into every `settings_json`), but only as a byproduct of
5
- // spending money on it. This module gives that structure a NAME, so it can be
6
- // listed, forked, agent-authored and restored without loss β€” before anything
7
- // has been generated.
8
- //
9
- // This is the canonical implementation. It is mirrored byte-for-byte into the
10
- // desktop app's `slate/src/shared/shotSpec.ts` (the desktop installs the
11
- // published @slatesvideo/shared from npm and cannot file-import this source, so
12
- // the mirror carries a header pointing here β€” the same rule
13
- // `reference-composer.ts` follows). `slate/scripts/composer-mirror-check.mjs`
14
- // asserts the two agree; do not invent a second sync mechanism.
15
- //
16
- // 🚨 KEEP THIS A DEPENDENCY-FREE LEAF. It imports nothing, in either repo. The
17
- // desktop's renderer bundles its mirror, the desktop's MAIN process reads it,
18
- // and the op surface here builds Zod schemas from it β€” a single `node:` import
19
- // would break the first of those.
1
+ import { GPT_BACKGROUNDS } from './model-capabilities.js';
20
2
  /**
21
3
  * Emission ORDER of the ordered roles β€” the order `buildReferenceGroups` pushes
22
4
  * them in, which is the order the composer numbers them in, which is the order
@@ -214,7 +196,10 @@ function readParams(v) {
214
196
  raw.gptQuality === 'max') {
215
197
  out.gptQuality = raw.gptQuality;
216
198
  }
217
- if (raw.gptBackground === 'auto' || raw.gptBackground === 'transparent' || raw.gptBackground === 'opaque') {
199
+ if (Array.isArray(raw.detachedVoiceCharacterIds)) {
200
+ out.detachedVoiceCharacterIds = strArray(raw.detachedVoiceCharacterIds);
201
+ }
202
+ if (GPT_BACKGROUNDS.includes(raw.gptBackground)) {
218
203
  out.gptBackground = raw.gptBackground;
219
204
  }
220
205
  if (raw.gridMode === 'off' || raw.gridMode === '2x2' || raw.gridMode === '3x3')