@slatesvideo/shared 0.6.9 β 0.6.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.js +3 -0
- package/dist/manual/content.d.ts +1 -1
- package/dist/manual/content.js +1 -1
- package/dist/operations/index.d.ts +8 -10
- package/dist/operations/index.js +99 -76
- package/dist/prompts/model-capabilities.d.ts +80 -0
- package/dist/prompts/model-capabilities.js +142 -9
- package/dist/prompts/model-facts.d.ts +27 -0
- package/dist/prompts/model-facts.js +58 -13
- package/dist/prompts/prompting-tips.js +2 -2
- package/dist/prompts/reference-composer.d.ts +15 -1
- package/dist/prompts/reference-composer.js +145 -21
- package/dist/prompts/shot-spec.d.ts +5 -2
- package/dist/prompts/shot-spec.js +5 -20
- package/dist/skills/content.js +4 -4
- package/dist/update-check.d.ts +22 -0
- package/dist/update-check.js +109 -0
- package/exports/slates-prompt-builder/generated/SKILL.md +2 -2
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +5 -5
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +2 -2
- package/skills/slates-model-selection.md +7 -7
- package/skills/slates-one-prompt-film.md +4 -3
- package/skills/slates-prompting-minimax-h3.md +9 -10
- package/skills/slates-prompting-seedance-2-5.md +5 -6
|
@@ -37,6 +37,23 @@ function citeImages(nums) {
|
|
|
37
37
|
const noun = nums.length === 1 ? 'image' : 'images';
|
|
38
38
|
return `${noun} ${joinNums(nums)}`;
|
|
39
39
|
}
|
|
40
|
+
/**
|
|
41
|
+
* "voice timbre from audio 1" β a character's VOICE, cited inline beside her
|
|
42
|
+
* name (lowercase, for inline use, exactly like `citeImages`).
|
|
43
|
+
*
|
|
44
|
+
* π¨ THE ROLE WORDS ARE THE LOAD-BEARING HALF, not decoration. A bare
|
|
45
|
+
* "(image 1, audio 1)" would be the UNROLED state: BytePlus's capability table
|
|
46
|
+
* gives an audio reference five possible jobs β "music, dialogue, voice, tone,
|
|
47
|
+
* or timbre" β and an unroled clip falls back to DIALOGUE, so the model
|
|
48
|
+
* transcribes it and speaks ITS words instead of the prompt's. That is the
|
|
49
|
+
* shipped defect where a supplied take came back as "a map called Slates" for
|
|
50
|
+
* "an app called Slates" (2026-08-28). "voice timbre" is the vendor's own
|
|
51
|
+
* phrase for the half we want: the sound of her, not her words.
|
|
52
|
+
*/
|
|
53
|
+
function citeVoice(nums) {
|
|
54
|
+
const noun = nums.length === 1 ? 'audio' : 'audios';
|
|
55
|
+
return `voice timbre from ${noun} ${joinNums(nums)}`;
|
|
56
|
+
}
|
|
40
57
|
function joinNums(nums) {
|
|
41
58
|
if (nums.length === 1)
|
|
42
59
|
return String(nums[0]);
|
|
@@ -140,7 +157,12 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
140
157
|
videoNums.push(videoNum);
|
|
141
158
|
orderedVideoPaths.push(m.path);
|
|
142
159
|
}
|
|
143
|
-
else if (m.mediaKind === 'audio' && g.kind === 'audio-ref') {
|
|
160
|
+
else if (m.mediaKind === 'audio' && (g.kind === 'audio-ref' || g.kind === 'character')) {
|
|
161
|
+
// ONE audio counter across hand-attached clips and a CHARACTER'S VOICE,
|
|
162
|
+
// for the same reason the video counter is shared: the two are the same
|
|
163
|
+
// numbered space on the wire, and a second counter would emit two
|
|
164
|
+
// "Audio 1"s the moment a request carried both. Which SENTENCE names
|
|
165
|
+
// the clip is what differs (step 3 vs step 3e), never the number.
|
|
144
166
|
audioNum += 1;
|
|
145
167
|
audioNums.push(audioNum);
|
|
146
168
|
orderedAudioPaths.push(m.path);
|
|
@@ -208,7 +230,30 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
208
230
|
return ''; // styles never inline β trailing clause only
|
|
209
231
|
if (!seenFirst.has(key)) {
|
|
210
232
|
seenFirst.add(key);
|
|
211
|
-
|
|
233
|
+
// π¨ ONE BINDING SITE PER ENTITY, CARRYING EVERY MEDIUM SHE OWNS
|
|
234
|
+
// (2026-09-09). The mention attaches her face AND her voice, so both are
|
|
235
|
+
// cited where her name appears rather than one inline and the other in a
|
|
236
|
+
// preamble sentence above the user's own words. That split was the first
|
|
237
|
+
// shape this shipped in, and it read backwards: a two-speaker prompt made
|
|
238
|
+
// you hold two nameβaudio mappings in your head before you reached the
|
|
239
|
+
// sentence, and a character with a voice and no photo said her name twice
|
|
240
|
+
// while citing nothing.
|
|
241
|
+
//
|
|
242
|
+
// It is also closer to the vendor, not further. BytePlus's binding
|
|
243
|
+
// example is ONE sentence covering both media β "Image 1 depicts the
|
|
244
|
+
// protagonist John and uses the voice timbre from Audio 1." β and the
|
|
245
|
+
// preamble form had already split it in half.
|
|
246
|
+
//
|
|
247
|
+
// π¨ EMPTY MEANS OMITTED, NEVER AN EMPTY PARENTHESIS. A voice-only
|
|
248
|
+
// character has no `imageNums` and `citeImages([])` would compose the
|
|
249
|
+
// literal "images " β a citation pointing at nothing, inside the one
|
|
250
|
+
// function whose whole job is that citations point at what is sent.
|
|
251
|
+
const cites = [];
|
|
252
|
+
if (g.imageNums.length > 0)
|
|
253
|
+
cites.push(citeImages(g.imageNums));
|
|
254
|
+
if (g.audioNums.length > 0)
|
|
255
|
+
cites.push(citeVoice(g.audioNums));
|
|
256
|
+
return cites.length > 0 ? `${g.name} (${cites.join(', ')})` : g.name;
|
|
212
257
|
}
|
|
213
258
|
return g.name;
|
|
214
259
|
});
|
|
@@ -239,25 +284,6 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
239
284
|
topKeys.push(`${noun} ${joinNums(g.videoNums)} ${tail}`);
|
|
240
285
|
}
|
|
241
286
|
}
|
|
242
|
-
// Reference audio ("Audio 1 is a provided reference."), plus THE WORDS when
|
|
243
|
-
// the user has typed them β see ReferenceGroup.spokenText for why the words
|
|
244
|
-
// have to travel as text as well as audio.
|
|
245
|
-
for (const g of numbered) {
|
|
246
|
-
if (g.kind === 'audio-ref' && g.audioNums.length > 0) {
|
|
247
|
-
const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
|
|
248
|
-
const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
|
|
249
|
-
topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
|
|
250
|
-
// Trimmed, never rewritten: the words between the quotes are the user's
|
|
251
|
-
// exactly as typed. The delimiters are CURLY on purpose β a straight
|
|
252
|
-
// quote inside the user's own line then sits beside them without
|
|
253
|
-
// colliding, so nothing has to be escaped and nothing is edited.
|
|
254
|
-
const spoken = (g.spokenText ?? '').trim();
|
|
255
|
-
if (spoken) {
|
|
256
|
-
const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
|
|
257
|
-
topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: β${spoken}β`);
|
|
258
|
-
}
|
|
259
|
-
}
|
|
260
|
-
}
|
|
261
287
|
// π¨ A PINNED REFERENCE IMAGE GETS NO KEY LINE, DELIBERATELY (2026-08-10).
|
|
262
288
|
// It used to emit "Image 1 is a provided reference." β the only branch here
|
|
263
289
|
// that assigns NO role, and therefore says nothing: every image in the request
|
|
@@ -295,6 +321,104 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
295
321
|
}
|
|
296
322
|
}
|
|
297
323
|
}
|
|
324
|
+
// ββ Every AUDIO line, in one pass, in audio-NUMBER order ββββββββββββββββ
|
|
325
|
+
//
|
|
326
|
+
// π¨ AFTER the subject lines, deliberately. A key line that names a subject
|
|
327
|
+
// ("Image 1 is Marcus.") has to come before a line that gives that subject's
|
|
328
|
+
// media a job, or the prompt describes a voice before it says whose face it
|
|
329
|
+
// belongs to. Binding is carried by the SENTENCE rather than by adjacency β
|
|
330
|
+
// the vendor states that outright β so nothing on the wire depends on this;
|
|
331
|
+
// what depends on it is whether the composed preview can be read top to
|
|
332
|
+
// bottom, and that preview is the surface the transparency invariant rests
|
|
333
|
+
// on.
|
|
334
|
+
//
|
|
335
|
+
// π¨ ONE LOOP OVER THE GROUPS, NOT ONE LOOP PER KIND, AND THAT IS THE WHOLE
|
|
336
|
+
// POINT OF ITS SHAPE. There are two audio sentences β a hand-attached clip's
|
|
337
|
+
// neutral "Audio N is a provided reference." and a character's roled
|
|
338
|
+
// "Sarah uses the voice timbre from Audio N." β and they draw their numbers
|
|
339
|
+
// from the SAME counter walking THIS list. Emitting them in two passes
|
|
340
|
+
// printed them in kind order instead of number order, so a request with two
|
|
341
|
+
// voices and one room-tone clip opened with "Audio 3 is a provided
|
|
342
|
+
// reference." and named Audio 1 and Audio 2 after it. Nothing was wrong on
|
|
343
|
+
// the wire β binding is carried by the sentence, not by adjacency, which the
|
|
344
|
+
// vendor states outright β but a prompt that counts backwards is a prompt
|
|
345
|
+
// nobody can proofread, and the composed preview is the surface the whole
|
|
346
|
+
// transparency invariant rests on. One pass over `numbered` IS number order,
|
|
347
|
+
// because the numbers were assigned by the same walk.
|
|
348
|
+
for (const g of numbered) {
|
|
349
|
+
if (g.audioNums.length === 0)
|
|
350
|
+
continue;
|
|
351
|
+
const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
|
|
352
|
+
if (g.kind === 'character') {
|
|
353
|
+
// A CHARACTER'S VOICE. When her token appears in the prompt the binding
|
|
354
|
+
// rides INLINE on her name (step 2) and there is nothing to say up here.
|
|
355
|
+
// This is the same duality her IMAGE already has β "Sarah (image 1)"
|
|
356
|
+
// inline versus "Image 1 is Sarah." when the prompt never names her β so
|
|
357
|
+
// it is the existing pattern rather than a second grammar.
|
|
358
|
+
//
|
|
359
|
+
// Emitted from THIS pass, not from a block of its own, so it keeps its
|
|
360
|
+
// place in audio-NUMBER order among the neutral lines. See the header.
|
|
361
|
+
const namedInPrompt = g.token && matchedInPrompt.has(normToken(g.token));
|
|
362
|
+
if (!namedInPrompt) {
|
|
363
|
+
topKeys.push(`${g.name} uses the ${citeVoice(g.audioNums)}.`);
|
|
364
|
+
}
|
|
365
|
+
continue;
|
|
366
|
+
}
|
|
367
|
+
if (g.kind !== 'audio-ref')
|
|
368
|
+
continue;
|
|
369
|
+
// A clip the user dragged on declared no role, so it keeps the neutral
|
|
370
|
+
// line, plus THE WORDS when the user has typed them β see
|
|
371
|
+
// ReferenceGroup.spokenText for why the words have to travel as text as
|
|
372
|
+
// well as audio.
|
|
373
|
+
const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
|
|
374
|
+
topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
|
|
375
|
+
// Trimmed, never rewritten: the words between the quotes are the user's
|
|
376
|
+
// exactly as typed. The delimiters are CURLY on purpose β a straight
|
|
377
|
+
// quote inside the user's own line then sits beside them without
|
|
378
|
+
// colliding, so nothing has to be escaped and nothing is edited.
|
|
379
|
+
const spoken = (g.spokenText ?? '').trim();
|
|
380
|
+
if (spoken) {
|
|
381
|
+
const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
|
|
382
|
+
topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: β${spoken}β`);
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
// ββ 3e. WHY A CHARACTER'S VOICE GETS A ROLE AT ALL (2026-09-09) ββββββββββ
|
|
386
|
+
//
|
|
387
|
+
// The binding itself is composed INLINE beside her name (step 2), or as a
|
|
388
|
+
// fallback line in the audio pass above when the prompt never names her.
|
|
389
|
+
// This is the receipt for why composing a role is legal at all.
|
|
390
|
+
//
|
|
391
|
+
// π¨ AUDIO IS THE ONE MODALITY WHERE THE NEUTRAL LINE UNDER-SPECIFIES, and
|
|
392
|
+
// this is the sentence that closes it. An image is definitionally a
|
|
393
|
+
// reference and a video has two possible roles, so both are settled by a
|
|
394
|
+
// neutral line. An audio attachment has FIVE β BytePlus's own capability
|
|
395
|
+
// table lists "music, dialogue, voice, tone, or timbre" β so
|
|
396
|
+
// "Audio 1 is a provided reference." distinguishes a clip from nothing while
|
|
397
|
+
// leaving four roles open, and an unroled clip falls back to DIALOGUE: the
|
|
398
|
+
// model re-transcribes it and speaks ITS words. That is the shipped defect
|
|
399
|
+
// where a supplied take came back as "a map called Slates" for "an app
|
|
400
|
+
// called Slates" (2026-08-28).
|
|
401
|
+
//
|
|
402
|
+
// The wording is the vendor's, not ours. BytePlus's own binding sentence is
|
|
403
|
+
// "Image 1 depicts the protagonist John and uses the voice timbre from
|
|
404
|
+
// Audio 1."; MiniMax builds the same primitive into H3's notation
|
|
405
|
+
// ("<Audio 1> is the voice-timbre reference for <Subject 1>"). Two vendors,
|
|
406
|
+
// independently. The DIALOGUE therefore comes from the prompt and the clip
|
|
407
|
+
// carries only the voice β receipts and line refs:
|
|
408
|
+
// second-brain/business/projects/slates/research/model-prompting-research.md
|
|
409
|
+
// Β§ 2026-09-09 Multimodal reference GRAMMAR, facts 2 and 3.
|
|
410
|
+
//
|
|
411
|
+
// π¨ IT IS LEGAL COMPOSITION ONLY BECAUSE THE ROLE WAS DECLARED. Assigning a
|
|
412
|
+
// voice to a character IS the declaration; a clip dragged onto the rail is
|
|
413
|
+
// not, and keeps the neutral line above. Inferring a role nobody declared
|
|
414
|
+
// stays forbidden (`slate/.claude/rules/prompt-surface.md`).
|
|
415
|
+
//
|
|
416
|
+
// The citation is lowercase (`voice timbre from audio 1`) like every other
|
|
417
|
+
// inline citation this composer emits. An earlier draft capitalised it to
|
|
418
|
+
// match the vendor's example verbatim, which left a single capitalised
|
|
419
|
+
// `Audio 1` sitting mid-sentence among lowercase `image 1`s; moving the
|
|
420
|
+
// binding inline removed the reason for the exception along with the
|
|
421
|
+
// exception.
|
|
298
422
|
// ββ 4. Style trailing clause (one, at the end β style reads best last) ββ
|
|
299
423
|
const styleNums = [];
|
|
300
424
|
for (const g of numbered) {
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type GptQuality, type GptBackground } from './model-capabilities.js';
|
|
1
2
|
/**
|
|
2
3
|
* Role an attachment carries in the composer tray. User-set, never inferred.
|
|
3
4
|
*
|
|
@@ -62,13 +63,15 @@ export interface ShotParams {
|
|
|
62
63
|
quality?: string;
|
|
63
64
|
/** GPT Image 2.5's tier. Always sent explicitly: fal's own default is
|
|
64
65
|
* `high`, which is the third of five rungs, not the top of two. */
|
|
65
|
-
gptQuality?:
|
|
66
|
+
gptQuality?: GptQuality;
|
|
66
67
|
/** GPT Image's alpha switch. `auto` is fal's default and ours; `transparent`
|
|
67
68
|
* asks for a real alpha channel rather than a painted backdrop. Costs
|
|
68
69
|
* nothing β fal prices this family on size Γ quality only, so it is NOT a
|
|
69
70
|
* cost-key segment. Named `gptBackground` because `background` already
|
|
70
71
|
* means "generate asynchronously" on every op that carries a Shot. */
|
|
71
|
-
gptBackground?:
|
|
72
|
+
gptBackground?: GptBackground;
|
|
73
|
+
/** Character voices explicitly removed from this recipe. */
|
|
74
|
+
detachedVoiceCharacterIds?: string[];
|
|
72
75
|
duration?: number;
|
|
73
76
|
imageQuantity?: number;
|
|
74
77
|
gridMode?: 'off' | '2x2' | '3x3';
|
|
@@ -1,22 +1,4 @@
|
|
|
1
|
-
|
|
2
|
-
//
|
|
3
|
-
// THE PRINCIPLE: a generation's full recipe already exists (the desktop writes
|
|
4
|
-
// `referenceGroups` into every `settings_json`), but only as a byproduct of
|
|
5
|
-
// spending money on it. This module gives that structure a NAME, so it can be
|
|
6
|
-
// listed, forked, agent-authored and restored without loss β before anything
|
|
7
|
-
// has been generated.
|
|
8
|
-
//
|
|
9
|
-
// This is the canonical implementation. It is mirrored byte-for-byte into the
|
|
10
|
-
// desktop app's `slate/src/shared/shotSpec.ts` (the desktop installs the
|
|
11
|
-
// published @slatesvideo/shared from npm and cannot file-import this source, so
|
|
12
|
-
// the mirror carries a header pointing here β the same rule
|
|
13
|
-
// `reference-composer.ts` follows). `slate/scripts/composer-mirror-check.mjs`
|
|
14
|
-
// asserts the two agree; do not invent a second sync mechanism.
|
|
15
|
-
//
|
|
16
|
-
// π¨ KEEP THIS A DEPENDENCY-FREE LEAF. It imports nothing, in either repo. The
|
|
17
|
-
// desktop's renderer bundles its mirror, the desktop's MAIN process reads it,
|
|
18
|
-
// and the op surface here builds Zod schemas from it β a single `node:` import
|
|
19
|
-
// would break the first of those.
|
|
1
|
+
import { GPT_BACKGROUNDS } from './model-capabilities.js';
|
|
20
2
|
/**
|
|
21
3
|
* Emission ORDER of the ordered roles β the order `buildReferenceGroups` pushes
|
|
22
4
|
* them in, which is the order the composer numbers them in, which is the order
|
|
@@ -214,7 +196,10 @@ function readParams(v) {
|
|
|
214
196
|
raw.gptQuality === 'max') {
|
|
215
197
|
out.gptQuality = raw.gptQuality;
|
|
216
198
|
}
|
|
217
|
-
if (
|
|
199
|
+
if (Array.isArray(raw.detachedVoiceCharacterIds)) {
|
|
200
|
+
out.detachedVoiceCharacterIds = strArray(raw.detachedVoiceCharacterIds);
|
|
201
|
+
}
|
|
202
|
+
if (GPT_BACKGROUNDS.includes(raw.gptBackground)) {
|
|
218
203
|
out.gptBackground = raw.gptBackground;
|
|
219
204
|
}
|
|
220
205
|
if (raw.gridMode === 'off' || raw.gridMode === '2x2' || raw.gridMode === '3x3')
|