@slatesvideo/shared 0.6.9 → 0.6.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/manual/content.d.ts +1 -1
- package/dist/manual/content.js +1 -1
- package/dist/operations/index.d.ts +8 -10
- package/dist/operations/index.js +90 -68
- package/dist/prompts/model-capabilities.d.ts +80 -0
- package/dist/prompts/model-capabilities.js +142 -9
- package/dist/prompts/reference-composer.d.ts +15 -1
- package/dist/prompts/reference-composer.js +145 -21
- package/dist/prompts/shot-spec.d.ts +5 -2
- package/dist/prompts/shot-spec.js +5 -20
- package/dist/skills/content.js +3 -3
- package/package.json +1 -1
- package/skills/slates-model-selection.md +133 -133
- package/skills/slates-one-prompt-film.md +95 -95
- package/skills/slates-prompting-minimax-h3.md +9 -10
|
@@ -100,6 +100,105 @@ const MINIMAX_H3_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
|
|
|
100
100
|
* to lose track of what was actually generated.
|
|
101
101
|
*/
|
|
102
102
|
const LTX_2_5_ASPECT_RATIOS = ['16:9', '9:16'];
|
|
103
|
+
export const GPT_QUALITY_TIERS = ['low', 'medium', 'high', 'xhigh', 'max'];
|
|
104
|
+
export const GPT_BACKGROUNDS = ['auto', 'transparent', 'opaque'];
|
|
105
|
+
// Product output sizes; schema bounds and metering receipt live in the GPT harvest.
|
|
106
|
+
export const GPT_IMAGE_25_SIZES = {
|
|
107
|
+
'1k': {
|
|
108
|
+
'1:1': { width: 1024, height: 1024 },
|
|
109
|
+
'16:9': { width: 1360, height: 768 },
|
|
110
|
+
'9:16': { width: 768, height: 1360 },
|
|
111
|
+
'4:3': { width: 1168, height: 880 },
|
|
112
|
+
'3:4': { width: 880, height: 1168 },
|
|
113
|
+
},
|
|
114
|
+
'2k': {
|
|
115
|
+
'1:1': { width: 1440, height: 1440 },
|
|
116
|
+
'16:9': { width: 1920, height: 1080 },
|
|
117
|
+
'9:16': { width: 1080, height: 1920 },
|
|
118
|
+
'4:3': { width: 1664, height: 1248 },
|
|
119
|
+
'3:4': { width: 1248, height: 1664 },
|
|
120
|
+
},
|
|
121
|
+
'3k': {
|
|
122
|
+
'1:1': { width: 1920, height: 1920 },
|
|
123
|
+
'16:9': { width: 2560, height: 1440 },
|
|
124
|
+
'9:16': { width: 1440, height: 2560 },
|
|
125
|
+
'4:3': { width: 2224, height: 1664 },
|
|
126
|
+
'3:4': { width: 1664, height: 2224 },
|
|
127
|
+
},
|
|
128
|
+
'4k': {
|
|
129
|
+
// 1:1 and 16:9 sit EXACTLY on the 8,294,400 ceiling — 3840×2160 is one of
|
|
130
|
+
// fal's own priced sizes, so the bound is inclusive. 4:3 / 3:4 are the two
|
|
131
|
+
// that had to move; see the constraint note above.
|
|
132
|
+
'1:1': { width: 2880, height: 2880 },
|
|
133
|
+
'16:9': { width: 3840, height: 2160 },
|
|
134
|
+
'9:16': { width: 2160, height: 3840 },
|
|
135
|
+
'4:3': { width: 3264, height: 2448 },
|
|
136
|
+
'3:4': { width: 2448, height: 3264 },
|
|
137
|
+
},
|
|
138
|
+
};
|
|
139
|
+
/**
|
|
140
|
+
* fal's named ~1MP presets per aspect ratio, with custom dims where fal has no
|
|
141
|
+
* preset. The `1k` rung of every non-GPT image model resolves through this.
|
|
142
|
+
*/
|
|
143
|
+
export const FAL_1MP_SIZES = {
|
|
144
|
+
'1:1': 'square_hd',
|
|
145
|
+
'4:3': 'landscape_4_3',
|
|
146
|
+
'3:4': 'portrait_4_3',
|
|
147
|
+
'16:9': 'landscape_16_9',
|
|
148
|
+
'9:16': 'portrait_16_9',
|
|
149
|
+
'2:3': { width: 832, height: 1248 },
|
|
150
|
+
'3:2': { width: 1248, height: 832 },
|
|
151
|
+
'4:5': { width: 896, height: 1120 },
|
|
152
|
+
'5:4': { width: 1120, height: 896 },
|
|
153
|
+
'21:9': { width: 1344, height: 576 },
|
|
154
|
+
};
|
|
155
|
+
/** Pixel dims for a megapixel target at an aspect ratio, rounded to multiples of 8. */
|
|
156
|
+
export function computeFalDimensions(aspectRatio, targetMP) {
|
|
157
|
+
const parts = aspectRatio.split(':').map(Number);
|
|
158
|
+
const w = parts[0] || 16;
|
|
159
|
+
const h = parts[1] || 9;
|
|
160
|
+
const ratio = w / h;
|
|
161
|
+
const targetPixels = targetMP * 1_000_000;
|
|
162
|
+
return {
|
|
163
|
+
width: Math.round(Math.sqrt(targetPixels * ratio) / 8) * 8,
|
|
164
|
+
height: Math.round(Math.sqrt(targetPixels / ratio) / 8) * 8,
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
/**
|
|
168
|
+
* The `image_size` a non-GPT fal image request carries, for one model × aspect ×
|
|
169
|
+
* resolution rung.
|
|
170
|
+
*
|
|
171
|
+
* 🚨 THIS IS A BILLING INPUT, WHICH IS WHY IT LIVES HERE (moved out of
|
|
172
|
+
* slate/src/main/api/fal.ts, 2026-09-10). The resolution rung is a segment of
|
|
173
|
+
* every image cost key, and nothing in the REQUEST names it — fal is told pixel
|
|
174
|
+
* dimensions, not "2k". The proxy therefore recovers the rung by running this
|
|
175
|
+
* function over the model's declared `imageResolutions` × `aspectRatios` and
|
|
176
|
+
* matching the body's `image_size`, exactly as it recovers a GPT Image rung from
|
|
177
|
+
* `GPT_IMAGE_25_SIZES`. A second copy of this arithmetic would mean the desktop
|
|
178
|
+
* and the server could disagree about what a request is worth, silently.
|
|
179
|
+
*
|
|
180
|
+
* GPT Image does NOT come through here — that family carries explicit pixel
|
|
181
|
+
* classes in `GPT_IMAGE_25_SIZES` and an explicit `quality` rung.
|
|
182
|
+
*/
|
|
183
|
+
export function falImageSize(model, aspectRatio, resolution) {
|
|
184
|
+
const ar = aspectRatio || '16:9';
|
|
185
|
+
const isSeedream5 = model === 'seedream-5-lite';
|
|
186
|
+
const res = resolution || (isSeedream5 ? '2k' : '1k');
|
|
187
|
+
// 1K (~1MP): fal's named presets, or small custom dims where there is none.
|
|
188
|
+
if (res === '1k') {
|
|
189
|
+
return FAL_1MP_SIZES[ar] || 'landscape_16_9';
|
|
190
|
+
}
|
|
191
|
+
// Seedream 5 Lite: custom dims must be ≥3.69MP (2560×1440) and ≤9.44MP
|
|
192
|
+
// (3072×3072), so its three rungs target 4 / 7 / 9 MP — all inside that band.
|
|
193
|
+
if (isSeedream5) {
|
|
194
|
+
if (res === '4k')
|
|
195
|
+
return computeFalDimensions(ar, 9);
|
|
196
|
+
return res === '3k' ? computeFalDimensions(ar, 7) : computeFalDimensions(ar, 4);
|
|
197
|
+
}
|
|
198
|
+
if (res === '2k')
|
|
199
|
+
return computeFalDimensions(ar, 2);
|
|
200
|
+
return computeFalDimensions(ar, 4);
|
|
201
|
+
}
|
|
103
202
|
/**
|
|
104
203
|
* The provider every AGENT generation actually lands on for Kling and Veo.
|
|
105
204
|
*
|
|
@@ -122,14 +221,22 @@ export const AGENT_ROUTE_PROVIDER = 'fal';
|
|
|
122
221
|
export const MODEL_CAPABILITIES = {
|
|
123
222
|
// ── Image models ───────────────────────────────────────────────────────────
|
|
124
223
|
'nano-banana-2': {
|
|
224
|
+
imageResolutions: ['1k', '2k', '4k'],
|
|
225
|
+
// fal's nano-banana-2 schema caps `num_images` at 4 (read 2026-09-09). This
|
|
226
|
+
// is the ONLY model that batches: the MCP's headless path (no projectId) asks
|
|
227
|
+
// fal for one batch, and every other route — desktop and agent alike — fires
|
|
228
|
+
// N separate single-image generations. The proxy bills the batch size.
|
|
229
|
+
maxBatchImages: 4,
|
|
125
230
|
aspectRatios: FULL_ASPECT_RATIOS,
|
|
126
231
|
maxRefImages: 14,
|
|
127
232
|
},
|
|
128
233
|
'nano-banana-2-lite': {
|
|
234
|
+
imageResolutions: ['1k'],
|
|
129
235
|
aspectRatios: FULL_ASPECT_RATIOS,
|
|
130
236
|
maxRefImages: 4, // fal edit endpoint caps input images at 4
|
|
131
237
|
},
|
|
132
238
|
'nano-banana-pro': {
|
|
239
|
+
imageResolutions: ['1k', '2k', '4k'],
|
|
133
240
|
aspectRatios: FULL_ASPECT_RATIOS,
|
|
134
241
|
maxRefImages: 14,
|
|
135
242
|
},
|
|
@@ -169,18 +276,22 @@ export const MODEL_CAPABILITIES = {
|
|
|
169
276
|
// limits either: the MCP's 4,000-character prompt against fal's 32,000, and
|
|
170
277
|
// image quantity, which is a fan-out and has no provider ceiling at all.
|
|
171
278
|
'gpt-image-2-5-flare': {
|
|
279
|
+
imageResolutions: ['2k', '3k', '4k'],
|
|
172
280
|
aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
|
|
173
281
|
maxRefImages: 16,
|
|
174
282
|
},
|
|
175
283
|
'gpt-image-2-5-sunburst': {
|
|
284
|
+
imageResolutions: ['2k', '3k', '4k'],
|
|
176
285
|
aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
|
|
177
286
|
maxRefImages: 16,
|
|
178
287
|
},
|
|
179
288
|
'flux-2-max': {
|
|
289
|
+
imageResolutions: ['1k', '2k', '4k'],
|
|
180
290
|
aspectRatios: FULL_ASPECT_RATIOS,
|
|
181
291
|
maxRefImages: 4,
|
|
182
292
|
},
|
|
183
293
|
'seedream-5-lite': {
|
|
294
|
+
imageResolutions: ['2k', '3k', '4k'],
|
|
184
295
|
aspectRatios: FULL_ASPECT_RATIOS,
|
|
185
296
|
maxRefImages: 10,
|
|
186
297
|
},
|
|
@@ -377,6 +488,9 @@ export const MODEL_CAPABILITIES = {
|
|
|
377
488
|
// the base row's branch — a different ladder AND a different price at the one
|
|
378
489
|
// tier they share. Every lookup downstream is an exact-id map, not a prefix.
|
|
379
490
|
'minimax-h3': {
|
|
491
|
+
// fal reference-to-video schema, 2026-09-09: each audio clip is 2-15s.
|
|
492
|
+
referenceAudioDuration: { min: 2, max: 15 },
|
|
493
|
+
referenceVideoDuration: { min: 2, max: 15 },
|
|
380
494
|
aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
|
|
381
495
|
// The full ladder. 480p/768p are NATIVE generation modes; 2K and 4K upscale
|
|
382
496
|
// a 768p base result through H3-Regenerate-2K, which is API-only and not in
|
|
@@ -408,6 +522,9 @@ export const MODEL_CAPABILITIES = {
|
|
|
408
522
|
maxReferenceAudioSeconds: 15,
|
|
409
523
|
},
|
|
410
524
|
'minimax-h3-max': {
|
|
525
|
+
// fal reference-to-video schema, 2026-09-09: each audio clip is 2-15s.
|
|
526
|
+
referenceAudioDuration: { min: 2, max: 15 },
|
|
527
|
+
referenceVideoDuration: { min: 2, max: 15 },
|
|
411
528
|
aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
|
|
412
529
|
// 🚨 REFERENCES LANDED 2026-09-09, AFTER A FALSE CLAIM WAS RETIRED. This row
|
|
413
530
|
// shipped from v1.5.5 declaring zero reference capacity because a comment
|
|
@@ -431,15 +548,7 @@ export const MODEL_CAPABILITIES = {
|
|
|
431
548
|
// arms — quoted off this endpoint, not inherited.
|
|
432
549
|
maxReferenceVideoSeconds: 15,
|
|
433
550
|
maxReferenceAudioSeconds: 15,
|
|
434
|
-
|
|
435
|
-
// resolution enum is ["480P","768P","1080P"] on all three h3-max endpoints.
|
|
436
|
-
// 2K/4K genuinely are absent: the H3-Regenerate-2K upscaler is API-only and
|
|
437
|
-
// is not in the open weights fal self-hosts, which is the actual mechanism
|
|
438
|
-
// behind the shorter ladder — 1080p was never part of that story.
|
|
439
|
-
//
|
|
440
|
-
// DEFAULT stays 768p: it is the tier the model natively generates, and
|
|
441
|
-
// 1080p is a 2x price step ($0.160/s against $0.080/s).
|
|
442
|
-
videoResolution: { options: ['480p', '768p', '1080p'], default: '768p' },
|
|
551
|
+
videoResolution: { options: ['480p', '768p'], default: '768p' },
|
|
443
552
|
duration: { min: 5, max: 15, mode: 'continuous' },
|
|
444
553
|
},
|
|
445
554
|
// ── LTX-2.5 (both seats on fal — added 2026-08-29) ─────────────────────────
|
|
@@ -817,4 +926,28 @@ export function describeReferenceImageCaps(models) {
|
|
|
817
926
|
return n === 0 ? '0 (prompt + source clip only)' : String(n);
|
|
818
927
|
});
|
|
819
928
|
}
|
|
929
|
+
/** H3 Max reference accounting, fal's worked tables read 2026-09-09.
|
|
930
|
+
* https://fal.ai/models/minimax/h3-max/reference-to-video
|
|
931
|
+
* 1080p video-reference pricing is unpublished; never infer it from output rates.
|
|
932
|
+
*/
|
|
933
|
+
export const MINIMAX_MAX_REFERENCE = {
|
|
934
|
+
freeTokens: 4096,
|
|
935
|
+
imagePixelsPerToken: 1024,
|
|
936
|
+
normalizedImageEdge: 1024,
|
|
937
|
+
audioTokensPerSecond: 80,
|
|
938
|
+
videoTokensPerSecond: { '480p': 2886, '768p': 7459.2 },
|
|
939
|
+
};
|
|
940
|
+
export function minimaxMaxReferenceTokens(input) {
|
|
941
|
+
const rate = MINIMAX_MAX_REFERENCE.videoTokensPerSecond[input.resolution];
|
|
942
|
+
if (input.videoSeconds > 0 && rate === undefined) {
|
|
943
|
+
throw new Error(`H3 Max video-reference pricing is unavailable at ${input.resolution}; choose a priced resolution.`);
|
|
944
|
+
}
|
|
945
|
+
for (const n of [input.imagePixels, input.videoSeconds, input.audioSeconds]) {
|
|
946
|
+
if (!Number.isFinite(n) || n < 0)
|
|
947
|
+
throw new Error('Reference metadata must be finite and nonnegative');
|
|
948
|
+
}
|
|
949
|
+
return Math.max(0, Math.ceil(input.imagePixels / MINIMAX_MAX_REFERENCE.imagePixelsPerToken +
|
|
950
|
+
input.videoSeconds * (rate ?? 0) + input.audioSeconds * MINIMAX_MAX_REFERENCE.audioTokensPerSecond -
|
|
951
|
+
MINIMAX_MAX_REFERENCE.freeTokens));
|
|
952
|
+
}
|
|
820
953
|
//# sourceMappingURL=model-capabilities.js.map
|
|
@@ -14,7 +14,21 @@ export interface ReferenceGroup {
|
|
|
14
14
|
/** Display + citation name: 'Marcus' | 'the cafe' | 'noir'. Used verbatim. */
|
|
15
15
|
name: string;
|
|
16
16
|
kind: ReferenceKind;
|
|
17
|
-
/**
|
|
17
|
+
/**
|
|
18
|
+
* A group can carry several images for workflows that genuinely need them.
|
|
19
|
+
*
|
|
20
|
+
* 🚨 A `character` GROUP MAY ALSO CARRY ONE `audio` MEDIUM — that character's
|
|
21
|
+
* assigned VOICE (2026-09-09). It is the same idea as the identity image, on
|
|
22
|
+
* the other axis of identity: the mention attaches what the character IS, and
|
|
23
|
+
* a voice is part of that. The audio takes its number from the audio counter,
|
|
24
|
+
* so adding one renumbers no image, and it is cited INLINE beside her name
|
|
25
|
+
* ("Sarah (image 1, voice timbre from audio 1)") rather than as the neutral
|
|
26
|
+
* `audio-ref` sentence — step 3e holds the receipt for why that is legal.
|
|
27
|
+
*
|
|
28
|
+
* A character group with a voice and NO identity image is a real state, not a
|
|
29
|
+
* defect: `voice-without-photo` is legal on any model that reads audio alone
|
|
30
|
+
* (Seedance 2.5). It cites no image and the timbre line carries the binding.
|
|
31
|
+
*/
|
|
18
32
|
media: ReferenceMedia[];
|
|
19
33
|
/**
|
|
20
34
|
* What is SAID in this group's reference audio, typed by the user.
|
|
@@ -37,6 +37,23 @@ function citeImages(nums) {
|
|
|
37
37
|
const noun = nums.length === 1 ? 'image' : 'images';
|
|
38
38
|
return `${noun} ${joinNums(nums)}`;
|
|
39
39
|
}
|
|
40
|
+
/**
|
|
41
|
+
* "voice timbre from audio 1" — a character's VOICE, cited inline beside her
|
|
42
|
+
* name (lowercase, for inline use, exactly like `citeImages`).
|
|
43
|
+
*
|
|
44
|
+
* 🚨 THE ROLE WORDS ARE THE LOAD-BEARING HALF, not decoration. A bare
|
|
45
|
+
* "(image 1, audio 1)" would be the UNROLED state: BytePlus's capability table
|
|
46
|
+
* gives an audio reference five possible jobs — "music, dialogue, voice, tone,
|
|
47
|
+
* or timbre" — and an unroled clip falls back to DIALOGUE, so the model
|
|
48
|
+
* transcribes it and speaks ITS words instead of the prompt's. That is the
|
|
49
|
+
* shipped defect where a supplied take came back as "a map called Slates" for
|
|
50
|
+
* "an app called Slates" (2026-08-28). "voice timbre" is the vendor's own
|
|
51
|
+
* phrase for the half we want: the sound of her, not her words.
|
|
52
|
+
*/
|
|
53
|
+
function citeVoice(nums) {
|
|
54
|
+
const noun = nums.length === 1 ? 'audio' : 'audios';
|
|
55
|
+
return `voice timbre from ${noun} ${joinNums(nums)}`;
|
|
56
|
+
}
|
|
40
57
|
function joinNums(nums) {
|
|
41
58
|
if (nums.length === 1)
|
|
42
59
|
return String(nums[0]);
|
|
@@ -140,7 +157,12 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
140
157
|
videoNums.push(videoNum);
|
|
141
158
|
orderedVideoPaths.push(m.path);
|
|
142
159
|
}
|
|
143
|
-
else if (m.mediaKind === 'audio' && g.kind === 'audio-ref') {
|
|
160
|
+
else if (m.mediaKind === 'audio' && (g.kind === 'audio-ref' || g.kind === 'character')) {
|
|
161
|
+
// ONE audio counter across hand-attached clips and a CHARACTER'S VOICE,
|
|
162
|
+
// for the same reason the video counter is shared: the two are the same
|
|
163
|
+
// numbered space on the wire, and a second counter would emit two
|
|
164
|
+
// "Audio 1"s the moment a request carried both. Which SENTENCE names
|
|
165
|
+
// the clip is what differs (step 3 vs step 3e), never the number.
|
|
144
166
|
audioNum += 1;
|
|
145
167
|
audioNums.push(audioNum);
|
|
146
168
|
orderedAudioPaths.push(m.path);
|
|
@@ -208,7 +230,30 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
208
230
|
return ''; // styles never inline — trailing clause only
|
|
209
231
|
if (!seenFirst.has(key)) {
|
|
210
232
|
seenFirst.add(key);
|
|
211
|
-
|
|
233
|
+
// 🚨 ONE BINDING SITE PER ENTITY, CARRYING EVERY MEDIUM SHE OWNS
|
|
234
|
+
// (2026-09-09). The mention attaches her face AND her voice, so both are
|
|
235
|
+
// cited where her name appears rather than one inline and the other in a
|
|
236
|
+
// preamble sentence above the user's own words. That split was the first
|
|
237
|
+
// shape this shipped in, and it read backwards: a two-speaker prompt made
|
|
238
|
+
// you hold two name→audio mappings in your head before you reached the
|
|
239
|
+
// sentence, and a character with a voice and no photo said her name twice
|
|
240
|
+
// while citing nothing.
|
|
241
|
+
//
|
|
242
|
+
// It is also closer to the vendor, not further. BytePlus's binding
|
|
243
|
+
// example is ONE sentence covering both media — "Image 1 depicts the
|
|
244
|
+
// protagonist John and uses the voice timbre from Audio 1." — and the
|
|
245
|
+
// preamble form had already split it in half.
|
|
246
|
+
//
|
|
247
|
+
// 🚨 EMPTY MEANS OMITTED, NEVER AN EMPTY PARENTHESIS. A voice-only
|
|
248
|
+
// character has no `imageNums` and `citeImages([])` would compose the
|
|
249
|
+
// literal "images " — a citation pointing at nothing, inside the one
|
|
250
|
+
// function whose whole job is that citations point at what is sent.
|
|
251
|
+
const cites = [];
|
|
252
|
+
if (g.imageNums.length > 0)
|
|
253
|
+
cites.push(citeImages(g.imageNums));
|
|
254
|
+
if (g.audioNums.length > 0)
|
|
255
|
+
cites.push(citeVoice(g.audioNums));
|
|
256
|
+
return cites.length > 0 ? `${g.name} (${cites.join(', ')})` : g.name;
|
|
212
257
|
}
|
|
213
258
|
return g.name;
|
|
214
259
|
});
|
|
@@ -239,25 +284,6 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
239
284
|
topKeys.push(`${noun} ${joinNums(g.videoNums)} ${tail}`);
|
|
240
285
|
}
|
|
241
286
|
}
|
|
242
|
-
// Reference audio ("Audio 1 is a provided reference."), plus THE WORDS when
|
|
243
|
-
// the user has typed them — see ReferenceGroup.spokenText for why the words
|
|
244
|
-
// have to travel as text as well as audio.
|
|
245
|
-
for (const g of numbered) {
|
|
246
|
-
if (g.kind === 'audio-ref' && g.audioNums.length > 0) {
|
|
247
|
-
const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
|
|
248
|
-
const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
|
|
249
|
-
topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
|
|
250
|
-
// Trimmed, never rewritten: the words between the quotes are the user's
|
|
251
|
-
// exactly as typed. The delimiters are CURLY on purpose — a straight
|
|
252
|
-
// quote inside the user's own line then sits beside them without
|
|
253
|
-
// colliding, so nothing has to be escaped and nothing is edited.
|
|
254
|
-
const spoken = (g.spokenText ?? '').trim();
|
|
255
|
-
if (spoken) {
|
|
256
|
-
const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
|
|
257
|
-
topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: “${spoken}”`);
|
|
258
|
-
}
|
|
259
|
-
}
|
|
260
|
-
}
|
|
261
287
|
// 🚨 A PINNED REFERENCE IMAGE GETS NO KEY LINE, DELIBERATELY (2026-08-10).
|
|
262
288
|
// It used to emit "Image 1 is a provided reference." — the only branch here
|
|
263
289
|
// that assigns NO role, and therefore says nothing: every image in the request
|
|
@@ -295,6 +321,104 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
|
295
321
|
}
|
|
296
322
|
}
|
|
297
323
|
}
|
|
324
|
+
// ── Every AUDIO line, in one pass, in audio-NUMBER order ────────────────
|
|
325
|
+
//
|
|
326
|
+
// 🚨 AFTER the subject lines, deliberately. A key line that names a subject
|
|
327
|
+
// ("Image 1 is Marcus.") has to come before a line that gives that subject's
|
|
328
|
+
// media a job, or the prompt describes a voice before it says whose face it
|
|
329
|
+
// belongs to. Binding is carried by the SENTENCE rather than by adjacency —
|
|
330
|
+
// the vendor states that outright — so nothing on the wire depends on this;
|
|
331
|
+
// what depends on it is whether the composed preview can be read top to
|
|
332
|
+
// bottom, and that preview is the surface the transparency invariant rests
|
|
333
|
+
// on.
|
|
334
|
+
//
|
|
335
|
+
// 🚨 ONE LOOP OVER THE GROUPS, NOT ONE LOOP PER KIND, AND THAT IS THE WHOLE
|
|
336
|
+
// POINT OF ITS SHAPE. There are two audio sentences — a hand-attached clip's
|
|
337
|
+
// neutral "Audio N is a provided reference." and a character's roled
|
|
338
|
+
// "Sarah uses the voice timbre from Audio N." — and they draw their numbers
|
|
339
|
+
// from the SAME counter walking THIS list. Emitting them in two passes
|
|
340
|
+
// printed them in kind order instead of number order, so a request with two
|
|
341
|
+
// voices and one room-tone clip opened with "Audio 3 is a provided
|
|
342
|
+
// reference." and named Audio 1 and Audio 2 after it. Nothing was wrong on
|
|
343
|
+
// the wire — binding is carried by the sentence, not by adjacency, which the
|
|
344
|
+
// vendor states outright — but a prompt that counts backwards is a prompt
|
|
345
|
+
// nobody can proofread, and the composed preview is the surface the whole
|
|
346
|
+
// transparency invariant rests on. One pass over `numbered` IS number order,
|
|
347
|
+
// because the numbers were assigned by the same walk.
|
|
348
|
+
for (const g of numbered) {
|
|
349
|
+
if (g.audioNums.length === 0)
|
|
350
|
+
continue;
|
|
351
|
+
const noun = g.audioNums.length === 1 ? 'Audio' : 'Audios';
|
|
352
|
+
if (g.kind === 'character') {
|
|
353
|
+
// A CHARACTER'S VOICE. When her token appears in the prompt the binding
|
|
354
|
+
// rides INLINE on her name (step 2) and there is nothing to say up here.
|
|
355
|
+
// This is the same duality her IMAGE already has — "Sarah (image 1)"
|
|
356
|
+
// inline versus "Image 1 is Sarah." when the prompt never names her — so
|
|
357
|
+
// it is the existing pattern rather than a second grammar.
|
|
358
|
+
//
|
|
359
|
+
// Emitted from THIS pass, not from a block of its own, so it keeps its
|
|
360
|
+
// place in audio-NUMBER order among the neutral lines. See the header.
|
|
361
|
+
const namedInPrompt = g.token && matchedInPrompt.has(normToken(g.token));
|
|
362
|
+
if (!namedInPrompt) {
|
|
363
|
+
topKeys.push(`${g.name} uses the ${citeVoice(g.audioNums)}.`);
|
|
364
|
+
}
|
|
365
|
+
continue;
|
|
366
|
+
}
|
|
367
|
+
if (g.kind !== 'audio-ref')
|
|
368
|
+
continue;
|
|
369
|
+
// A clip the user dragged on declared no role, so it keeps the neutral
|
|
370
|
+
// line, plus THE WORDS when the user has typed them — see
|
|
371
|
+
// ReferenceGroup.spokenText for why the words have to travel as text as
|
|
372
|
+
// well as audio.
|
|
373
|
+
const tail = g.audioNums.length === 1 ? 'is a provided reference.' : 'are provided references.';
|
|
374
|
+
topKeys.push(`${noun} ${joinNums(g.audioNums)} ${tail}`);
|
|
375
|
+
// Trimmed, never rewritten: the words between the quotes are the user's
|
|
376
|
+
// exactly as typed. The delimiters are CURLY on purpose — a straight
|
|
377
|
+
// quote inside the user's own line then sits beside them without
|
|
378
|
+
// colliding, so nothing has to be escaped and nothing is edited.
|
|
379
|
+
const spoken = (g.spokenText ?? '').trim();
|
|
380
|
+
if (spoken) {
|
|
381
|
+
const lower = g.audioNums.length === 1 ? 'audio' : 'audios';
|
|
382
|
+
topKeys.push(`The words spoken in ${lower} ${joinNums(g.audioNums)} are exactly: “${spoken}”`);
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
// ── 3e. WHY A CHARACTER'S VOICE GETS A ROLE AT ALL (2026-09-09) ──────────
|
|
386
|
+
//
|
|
387
|
+
// The binding itself is composed INLINE beside her name (step 2), or as a
|
|
388
|
+
// fallback line in the audio pass above when the prompt never names her.
|
|
389
|
+
// This is the receipt for why composing a role is legal at all.
|
|
390
|
+
//
|
|
391
|
+
// 🚨 AUDIO IS THE ONE MODALITY WHERE THE NEUTRAL LINE UNDER-SPECIFIES, and
|
|
392
|
+
// this is the sentence that closes it. An image is definitionally a
|
|
393
|
+
// reference and a video has two possible roles, so both are settled by a
|
|
394
|
+
// neutral line. An audio attachment has FIVE — BytePlus's own capability
|
|
395
|
+
// table lists "music, dialogue, voice, tone, or timbre" — so
|
|
396
|
+
// "Audio 1 is a provided reference." distinguishes a clip from nothing while
|
|
397
|
+
// leaving four roles open, and an unroled clip falls back to DIALOGUE: the
|
|
398
|
+
// model re-transcribes it and speaks ITS words. That is the shipped defect
|
|
399
|
+
// where a supplied take came back as "a map called Slates" for "an app
|
|
400
|
+
// called Slates" (2026-08-28).
|
|
401
|
+
//
|
|
402
|
+
// The wording is the vendor's, not ours. BytePlus's own binding sentence is
|
|
403
|
+
// "Image 1 depicts the protagonist John and uses the voice timbre from
|
|
404
|
+
// Audio 1."; MiniMax builds the same primitive into H3's notation
|
|
405
|
+
// ("<Audio 1> is the voice-timbre reference for <Subject 1>"). Two vendors,
|
|
406
|
+
// independently. The DIALOGUE therefore comes from the prompt and the clip
|
|
407
|
+
// carries only the voice — receipts and line refs:
|
|
408
|
+
// second-brain/business/projects/slates/research/model-prompting-research.md
|
|
409
|
+
// § 2026-09-09 Multimodal reference GRAMMAR, facts 2 and 3.
|
|
410
|
+
//
|
|
411
|
+
// 🚨 IT IS LEGAL COMPOSITION ONLY BECAUSE THE ROLE WAS DECLARED. Assigning a
|
|
412
|
+
// voice to a character IS the declaration; a clip dragged onto the rail is
|
|
413
|
+
// not, and keeps the neutral line above. Inferring a role nobody declared
|
|
414
|
+
// stays forbidden (`slate/.claude/rules/prompt-surface.md`).
|
|
415
|
+
//
|
|
416
|
+
// The citation is lowercase (`voice timbre from audio 1`) like every other
|
|
417
|
+
// inline citation this composer emits. An earlier draft capitalised it to
|
|
418
|
+
// match the vendor's example verbatim, which left a single capitalised
|
|
419
|
+
// `Audio 1` sitting mid-sentence among lowercase `image 1`s; moving the
|
|
420
|
+
// binding inline removed the reason for the exception along with the
|
|
421
|
+
// exception.
|
|
298
422
|
// ── 4. Style trailing clause (one, at the end — style reads best last) ──
|
|
299
423
|
const styleNums = [];
|
|
300
424
|
for (const g of numbered) {
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type GptQuality, type GptBackground } from './model-capabilities.js';
|
|
1
2
|
/**
|
|
2
3
|
* Role an attachment carries in the composer tray. User-set, never inferred.
|
|
3
4
|
*
|
|
@@ -62,13 +63,15 @@ export interface ShotParams {
|
|
|
62
63
|
quality?: string;
|
|
63
64
|
/** GPT Image 2.5's tier. Always sent explicitly: fal's own default is
|
|
64
65
|
* `high`, which is the third of five rungs, not the top of two. */
|
|
65
|
-
gptQuality?:
|
|
66
|
+
gptQuality?: GptQuality;
|
|
66
67
|
/** GPT Image's alpha switch. `auto` is fal's default and ours; `transparent`
|
|
67
68
|
* asks for a real alpha channel rather than a painted backdrop. Costs
|
|
68
69
|
* nothing — fal prices this family on size × quality only, so it is NOT a
|
|
69
70
|
* cost-key segment. Named `gptBackground` because `background` already
|
|
70
71
|
* means "generate asynchronously" on every op that carries a Shot. */
|
|
71
|
-
gptBackground?:
|
|
72
|
+
gptBackground?: GptBackground;
|
|
73
|
+
/** Character voices explicitly removed from this recipe. */
|
|
74
|
+
detachedVoiceCharacterIds?: string[];
|
|
72
75
|
duration?: number;
|
|
73
76
|
imageQuantity?: number;
|
|
74
77
|
gridMode?: 'off' | '2x2' | '3x3';
|
|
@@ -1,22 +1,4 @@
|
|
|
1
|
-
|
|
2
|
-
//
|
|
3
|
-
// THE PRINCIPLE: a generation's full recipe already exists (the desktop writes
|
|
4
|
-
// `referenceGroups` into every `settings_json`), but only as a byproduct of
|
|
5
|
-
// spending money on it. This module gives that structure a NAME, so it can be
|
|
6
|
-
// listed, forked, agent-authored and restored without loss — before anything
|
|
7
|
-
// has been generated.
|
|
8
|
-
//
|
|
9
|
-
// This is the canonical implementation. It is mirrored byte-for-byte into the
|
|
10
|
-
// desktop app's `slate/src/shared/shotSpec.ts` (the desktop installs the
|
|
11
|
-
// published @slatesvideo/shared from npm and cannot file-import this source, so
|
|
12
|
-
// the mirror carries a header pointing here — the same rule
|
|
13
|
-
// `reference-composer.ts` follows). `slate/scripts/composer-mirror-check.mjs`
|
|
14
|
-
// asserts the two agree; do not invent a second sync mechanism.
|
|
15
|
-
//
|
|
16
|
-
// 🚨 KEEP THIS A DEPENDENCY-FREE LEAF. It imports nothing, in either repo. The
|
|
17
|
-
// desktop's renderer bundles its mirror, the desktop's MAIN process reads it,
|
|
18
|
-
// and the op surface here builds Zod schemas from it — a single `node:` import
|
|
19
|
-
// would break the first of those.
|
|
1
|
+
import { GPT_BACKGROUNDS } from './model-capabilities.js';
|
|
20
2
|
/**
|
|
21
3
|
* Emission ORDER of the ordered roles — the order `buildReferenceGroups` pushes
|
|
22
4
|
* them in, which is the order the composer numbers them in, which is the order
|
|
@@ -214,7 +196,10 @@ function readParams(v) {
|
|
|
214
196
|
raw.gptQuality === 'max') {
|
|
215
197
|
out.gptQuality = raw.gptQuality;
|
|
216
198
|
}
|
|
217
|
-
if (
|
|
199
|
+
if (Array.isArray(raw.detachedVoiceCharacterIds)) {
|
|
200
|
+
out.detachedVoiceCharacterIds = strArray(raw.detachedVoiceCharacterIds);
|
|
201
|
+
}
|
|
202
|
+
if (GPT_BACKGROUNDS.includes(raw.gptBackground)) {
|
|
218
203
|
out.gptBackground = raw.gptBackground;
|
|
219
204
|
}
|
|
220
205
|
if (raw.gridMode === 'off' || raw.gridMode === '2x2' || raw.gridMode === '3x3')
|