@slatesvideo/shared 0.6.0 → 0.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -16,6 +16,24 @@ export interface ReferenceGroup {
16
16
  kind: ReferenceKind;
17
17
  /** A group can carry several images for workflows that genuinely need them. */
18
18
  media: ReferenceMedia[];
19
+ /**
20
+ * What is SAID in this group's reference audio, typed by the user.
21
+ * `audio-ref` groups only; every other kind ignores it.
22
+ *
23
+ * 🚨 THE MODEL RE-TRANSCRIBES REFERENCE AUDIO AND GUESSES THE WORDS.
24
+ * Seedance does not consume a supplied take verbatim — it re-synthesises
25
+ * something very close to it, and a field test on 2026-08-28 heard "an app
26
+ * called Slates" come back as "a map called Slates". The audio carries the
27
+ * voice, the accent and the timing; only TEXT carries the words. Composing
28
+ * this line is the whole fix, and every user who attached a voice take since
29
+ * v1.5.2 was exposed to silent mistranscription without it.
30
+ *
31
+ * It is USER-AUTHORED and optional. Nothing transcribes the clip and fills
32
+ * this in: a second model in the request path silently rewriting the prompt
33
+ * is exactly what the prompt-transparency invariant forbids. An empty value
34
+ * composes byte-identically to before this field existed.
35
+ */
36
+ spokenText?: string;
19
37
  }
20
38
  export interface ComposedReferences {
21
39
  /** The composed prompt: user's words lead, references cited inline as "image N". */
@@ -61,6 +79,45 @@ export interface ComposeOptions {
61
79
  startAudioNumber?: number;
62
80
  }
63
81
  export declare function composeReferences(rawPrompt: string, groups: ReferenceGroup[], opts?: ComposeOptions): ComposedReferences;
82
+ /** One character's voice sample, in the order it is attached. */
83
+ export interface VoiceCitation {
84
+ /**
85
+ * The mention this voice answers to — '@sarah', or null when the character
86
+ * is attached without being mentioned.
87
+ *
88
+ * Matched through `normToken`, so '@Big Red', '@big_red' and '@big-red' are
89
+ * the same token. It is therefore NOT guaranteed byte-equal to what the
90
+ * prompt authored: the desktop builds it from the character's NAME while the
91
+ * agent passes the token as typed, and both resolve identically. Do not echo
92
+ * it back to a user as "what they wrote".
93
+ */
94
+ token: string | null;
95
+ /** 'Sarah' — the character's name, used verbatim. */
96
+ name: string;
97
+ }
98
+ /**
99
+ * Translate `@character` mentions into Seed Audio's OWN `@AudioN` notation.
100
+ *
101
+ * 🚨 AN ADAPTER, NOT A SECOND COMPOSER — the same relationship
102
+ * `composeKlingEdit` has to `composeReferences`. Seed Audio does not read
103
+ * "audio 1"; it reads `@Audio1`-`@Audio3`, and a clip the prompt never cites is
104
+ * a clip the model ignores. Attaching a voice without citing it would be the
105
+ * silent-drop failure this repo catalogues by name: attach the reference, get
106
+ * no error, get no warning, and get a request that does nothing with it.
107
+ *
108
+ * It emits the SAME grammar the image composer uses — the first mention
109
+ * becomes `Sarah (@Audio1)`, later ones stay `Sarah` — so the NAME still
110
+ * carries the identity and the citation carries the slot. A voice that is
111
+ * attached but never mentioned gets one short key line instead, exactly as an
112
+ * unmentioned character's image does.
113
+ *
114
+ * Everything else is left as authored: an `@token` that is not one of these
115
+ * voices is not ours to touch.
116
+ *
117
+ * A `null` slot means "this position is a clip the caller cites itself" — it
118
+ * holds its @AudioN number and is otherwise ignored.
119
+ */
120
+ export declare function composeVoiceCitations(rawPrompt: string, voices: Array<VoiceCitation | null>): string;
64
121
  export interface KlingEditElement {
65
122
  /** Frontal image path/URL (element primary view) */
66
123
  frontal: string;
@@ -170,12 +170,23 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
170
170
  topKeys.push(`${noun} ${joinNums(g.videoNums)} ${tail}`);
171
171
  }
172
172
  }
173
- // Reference audio ("Audio 1 is a provided reference.").
173
+ // Reference audio ("Audio 1 is a provided reference."), plus THE WORDS when
174
+ // the user has typed them — see ReferenceGroup.spokenText for why the words
175
+ // have to travel as text as well as audio.
174
176
  for (const g of numbered) {
175
177
  if (g.kind === 'audio-ref' && g.audioNums.length > 0) {
176
178
  const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
177
179
  const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
178
180
  topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
181
+ // Trimmed, never rewritten: the words between the quotes are the user's
182
+ // exactly as typed. The delimiters are CURLY on purpose — a straight
183
+ // quote inside the user's own line then sits beside them without
184
+ // colliding, so nothing has to be escaped and nothing is edited.
185
+ const spoken = (g.spokenText ?? '').trim();
186
+ if (spoken) {
187
+ const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
188
+ topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: “${spoken}”`);
189
+ }
179
190
  }
180
191
  }
181
192
  // 🚨 A PINNED REFERENCE IMAGE GETS NO KEY LINE, DELIBERATELY (2026-08-10).
@@ -241,6 +252,64 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
241
252
  unresolvedTokens,
242
253
  };
243
254
  }
255
+ /**
256
+ * Translate `@character` mentions into Seed Audio's OWN `@AudioN` notation.
257
+ *
258
+ * 🚨 AN ADAPTER, NOT A SECOND COMPOSER — the same relationship
259
+ * `composeKlingEdit` has to `composeReferences`. Seed Audio does not read
260
+ * "audio 1"; it reads `@Audio1`-`@Audio3`, and a clip the prompt never cites is
261
+ * a clip the model ignores. Attaching a voice without citing it would be the
262
+ * silent-drop failure this repo catalogues by name: attach the reference, get
263
+ * no error, get no warning, and get a request that does nothing with it.
264
+ *
265
+ * It emits the SAME grammar the image composer uses — the first mention
266
+ * becomes `Sarah (@Audio1)`, later ones stay `Sarah` — so the NAME still
267
+ * carries the identity and the citation carries the slot. A voice that is
268
+ * attached but never mentioned gets one short key line instead, exactly as an
269
+ * unmentioned character's image does.
270
+ *
271
+ * Everything else is left as authored: an `@token` that is not one of these
272
+ * voices is not ours to touch.
273
+ *
274
+ * A `null` slot means "this position is a clip the caller cites itself" — it
275
+ * holds its @AudioN number and is otherwise ignored.
276
+ */
277
+ export function composeVoiceCitations(rawPrompt, voices) {
278
+ if (voices.length === 0)
279
+ return rawPrompt;
280
+ const byNorm = new Map();
281
+ voices.forEach((v, i) => {
282
+ if (v?.token)
283
+ byNorm.set(normToken(v.token), { n: i + 1, name: v.name });
284
+ });
285
+ const seen = new Set();
286
+ const body = rawPrompt.replace(/@([\w-]+)/g, (full, tok) => {
287
+ const key = normToken(`@${tok}`);
288
+ const v = byNorm.get(key);
289
+ if (!v)
290
+ return full;
291
+ if (seen.has(key))
292
+ return v.name;
293
+ seen.add(key);
294
+ return `${v.name} (@Audio${v.n})`;
295
+ });
296
+ const keys = [];
297
+ voices.forEach((v, i) => {
298
+ // A null slot is a clip the CALLER already cited itself (the agent route
299
+ // passes its own `audioReferenceAssetIds` this way). It still occupies its
300
+ // @AudioN position — renumbering would repoint every citation a published
301
+ // CLI build wrote — but it is not ours to name.
302
+ if (!v)
303
+ return;
304
+ if (v.token && seen.has(normToken(v.token)))
305
+ return;
306
+ keys.push(`${v.name} is @Audio${i + 1}.`);
307
+ });
308
+ if (keys.length === 0)
309
+ return body;
310
+ const trimmed = body.trim();
311
+ return trimmed ? `${keys.join(' ')}\n\n${trimmed}` : keys.join(' ');
312
+ }
244
313
  /** fal cap: max 4 combined element + style-image references per edit request. */
245
314
  export const KLING_EDIT_MAX_REFS = 4;
246
315
  /**