@slatesvideo/shared 0.5.9 β 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -0
- package/dist/index.js +6 -0
- package/dist/operations/index.d.ts +34 -13
- package/dist/operations/index.js +426 -77
- package/dist/prompts/index.d.ts +1 -0
- package/dist/prompts/index.js +4 -0
- package/dist/prompts/model-capabilities.d.ts +124 -0
- package/dist/prompts/model-capabilities.js +624 -0
- package/dist/prompts/model-facts.d.ts +11 -3
- package/dist/prompts/model-facts.js +87 -51
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +64 -2
- package/dist/skills/content.js +8 -6
- package/exports/slates-prompt-builder/generated/SKILL.md +1 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +5 -5
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +5 -1
- package/skills/slates-direct-response-ad.md +2 -0
- package/skills/slates-model-selection.md +17 -11
- package/skills/slates-one-prompt-film.md +3 -2
- package/skills/slates-prompting-gpt-image-2.md +70 -40
- package/skills/slates-prompting-minimax-h3.md +287 -0
- package/skills/slates-prompting-seedance-2-5.md +23 -17
- package/skills/slates-prompting-veo-3.md +2 -2
- package/skills/slates-ugc-influencer-ad.md +307 -0
package/dist/operations/index.js
CHANGED
|
@@ -16,6 +16,16 @@ import { SKILLS } from '../skills/content.js';
|
|
|
16
16
|
// Reference-capacity prose is DERIVED, never hand-typed β root CLAUDE.md:
|
|
17
17
|
// "never hand-type a fact an LLM will read". These helpers read MODEL_FACTS.
|
|
18
18
|
import { multimodalRefSummary, multimodalRefModels, seedanceTaskIntentWords } from '../prompts/model-facts.js';
|
|
19
|
+
// π¨ MODEL CAPABILITY SSOT. Aspect ratios, video resolutions and durations are
|
|
20
|
+
// VALIDATED against and DESCRIBED from `MODEL_CAPABILITIES` β this file must
|
|
21
|
+
// never re-state one of those constraints as a literal enum or a sentence
|
|
22
|
+
// again. It did until 2026-08-16, and every axis had drifted: an invented
|
|
23
|
+
// `9:21`, "Kling/Seedance support all" (Seedance takes 6 of 11 and has no
|
|
24
|
+
// `4:5`), "Veo locks to 16:9" (two on fal), "Kling: 5-15" (min is 3), a
|
|
25
|
+
// `videoResolution` line that never mentioned Kling, and 4s quoted for Veo at
|
|
26
|
+
// 1080p. A customer burned a round trip on `4:5` + Seedance: accepted here,
|
|
27
|
+
// queued, credits reserved, rejected by the provider asynchronously.
|
|
28
|
+
import { AGENT_ROUTE_PROVIDER, aspectRatioUnion, videoResolutionUnion, durationBounds, aspectRatiosFor, getModelCapability, defaultVideoResolutionFor, checkAspectRatio, checkVideoResolution, checkDuration, describeAspectRatios, describeVideoResolutions, describeDurations, describeReferenceImageCaps, } from '../prompts/model-capabilities.js';
|
|
19
29
|
export function defaultContext() {
|
|
20
30
|
return {
|
|
21
31
|
cloud: () => new SlatesCloudClient(),
|
|
@@ -23,6 +33,18 @@ export function defaultContext() {
|
|
|
23
33
|
};
|
|
24
34
|
}
|
|
25
35
|
// ββ Helpers βββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
|
36
|
+
/**
|
|
37
|
+
* `z.enum` over a GENERATED list. Zod's signature wants a non-empty tuple
|
|
38
|
+
* literal, which is exactly what you can't write when the values come from the
|
|
39
|
+
* capability SSOT β and writing them out is how the enums drifted in the first
|
|
40
|
+
* place. Callers must pass a non-empty array; every call site derives from
|
|
41
|
+
* `MODEL_CAPABILITIES`, which never yields an empty union for a real model set.
|
|
42
|
+
*/
|
|
43
|
+
function zEnum(values) {
|
|
44
|
+
if (values.length === 0)
|
|
45
|
+
throw new Error('zEnum: empty value list');
|
|
46
|
+
return z.enum(values);
|
|
47
|
+
}
|
|
26
48
|
function ok(data, text) {
|
|
27
49
|
return {
|
|
28
50
|
// Compact JSON β pretty-printing (indent 2) cost ~10-15% extra tokens
|
|
@@ -133,19 +155,146 @@ export const listAvailableModels = {
|
|
|
133
155
|
};
|
|
134
156
|
},
|
|
135
157
|
};
|
|
158
|
+
// ββ Video model vocabulary ββββββββββββββββββββββββββββββββββββββ
|
|
159
|
+
//
|
|
160
|
+
// Declared HERE, above every op, because both `slates_estimate_generation_cost`
|
|
161
|
+
// and `slates_generate_video` build their Zod schemas from it at module load β
|
|
162
|
+
// and a schema is evaluated in source order, so a list declared below the first
|
|
163
|
+
// op that reads it is a temporal-dead-zone crash, not a lint nit.
|
|
164
|
+
// Exported: the exact `model` ids slates_generate_video accepts β consumed
|
|
165
|
+
// by the desktop Studio Agent system prompt (SSOT; never restate these ids
|
|
166
|
+
// in prose that can drift).
|
|
167
|
+
export const VIDEO_MODELS = [
|
|
168
|
+
'kling-v3.0-std',
|
|
169
|
+
'kling-v3.0-pro',
|
|
170
|
+
'kling-v3.0-omni',
|
|
171
|
+
'veo-3.1-fast',
|
|
172
|
+
'veo-3.1-standard',
|
|
173
|
+
'seedance-2',
|
|
174
|
+
// Seedance 2.5 is a SECOND SEAT, not a replacement: 30s takes, 30 image
|
|
175
|
+
// references, audio-only references, up to 1080p (2026-08-24) β but no 4K,
|
|
176
|
+
// and dearer than 2.0 at every shared tier, so 2.0 stays the default. Its
|
|
177
|
+
// EDIT row is not here; edit models live on slates_edit_video, same as the
|
|
178
|
+
// Kling and Omni Flash ones.
|
|
179
|
+
'seedance-2.5',
|
|
180
|
+
'omni-flash',
|
|
181
|
+
// MiniMax H3, two seats in one family (2026-08-27). Base H3 is the AUTHORED-
|
|
182
|
+
// AUDIO seat β three directable sound layers in one pass, declared reference
|
|
183
|
+
// relationships, 480p to 4K, and the cheapest 768-class second we sell.
|
|
184
|
+
// H3 Max is fal's self-hosted post-train: faster, capped at 768p, takes NO
|
|
185
|
+
// references, and costs MORE than base H3 at the tier they share β a
|
|
186
|
+
// premium-speed seat, never a cheap H3.
|
|
187
|
+
//
|
|
188
|
+
// NEVER PREFIX-MATCH: 'minimax-h3-max' starts with 'minimax-h3'. Every
|
|
189
|
+
// branch keyed on these ids matches EXACTLY; a prefix test silently bills
|
|
190
|
+
// the Max row at base rates and offers it 2K/4K it cannot render.
|
|
191
|
+
'minimax-h3',
|
|
192
|
+
'minimax-h3-max',
|
|
193
|
+
];
|
|
194
|
+
// ββ Capability-derived param vocabulary + guard βββββββββββββββββ
|
|
195
|
+
//
|
|
196
|
+
// π¨ EVERYTHING BELOW IS GENERATED FROM `MODEL_CAPABILITIES`. Not one aspect
|
|
197
|
+
// ratio, resolution or duration is typed into this file any more, and none may
|
|
198
|
+
// be again. The enums are the UNION of what the video models accept (so the
|
|
199
|
+
// JSON schema still advertises the legal universe to the LLM) and
|
|
200
|
+
// `assertVideoCapabilities` does the PER-MODEL narrowing.
|
|
201
|
+
/** Union of the ratios every video model accepts on the agent's route. */
|
|
202
|
+
const VIDEO_ASPECT_RATIOS = aspectRatioUnion(VIDEO_MODELS, AGENT_ROUTE_PROVIDER);
|
|
203
|
+
/** Union of the resolutions every video model renders at. */
|
|
204
|
+
const VIDEO_RESOLUTIONS = videoResolutionUnion(VIDEO_MODELS);
|
|
205
|
+
/** Widest legal duration window across every video model. */
|
|
206
|
+
const VIDEO_DURATION_BOUNDS = durationBounds(VIDEO_MODELS);
|
|
207
|
+
/** Omni Flash's combined reference-image cap, read from the SSOT (7). */
|
|
208
|
+
const omniFlashRefCap = getModelCapability('omni-flash')?.maxIngredientImages ?? 0;
|
|
209
|
+
/** Every resolution token any video model speaks, for the forgiving cost-key
|
|
210
|
+
* parser. Derived, so 768p and 2k arrived with the models rather than with a
|
|
211
|
+
* later bug report. */
|
|
212
|
+
const VIDEO_RESOLUTION_VOCAB = VIDEO_RESOLUTIONS;
|
|
213
|
+
// ββ MiniMax H3: the reference-image surcharge, as a KEY DIMENSION ββββββββββββ
|
|
214
|
+
//
|
|
215
|
+
// fal charges $0.080 per reference image PAST THE FIRST FIVE on
|
|
216
|
+
// `minimax/h3/reference-to-video`, and nowhere else β not on image-to-video
|
|
217
|
+
// start/end frames (those are FL2VA inputs, not Ref2VA references), and not on
|
|
218
|
+
// h3-max, which has no reference endpoint at all. Left unmodelled it inverts
|
|
219
|
+
// the margin: a 10s 768p clip earns $0.30 and four extra images cost $0.32.
|
|
220
|
+
//
|
|
221
|
+
// `/proxy/generate` resolves ONE key to ONE integer, so a surcharge has to live
|
|
222
|
+
// IN the key β exactly the shape Kling's `-audio` suffix uses. The bare key
|
|
223
|
+
// stays identical to every other model's; the suffix appears only when the
|
|
224
|
+
// option costs money.
|
|
225
|
+
//
|
|
226
|
+
// The rounding is EXACT and worth knowing before anyone changes the rate:
|
|
227
|
+
// $0.080 x 1.5 x 100 = 12 cents, and 12 is divisible by CENTS_PER_CREDIT (3),
|
|
228
|
+
// so every extra image is 4 credits at every resolution and every duration with
|
|
229
|
+
// zero drift. A rate that is not a multiple of 2 cents breaks that property.
|
|
230
|
+
const MINIMAX_MODELS = new Set(['minimax-h3', 'minimax-h3-max']);
|
|
231
|
+
/** Reference images fal does not charge for. */
|
|
232
|
+
const MINIMAX_FREE_REF_IMAGES = 5;
|
|
233
|
+
/** K for the `-ref{K}` suffix: images past the free five, capped by the model's
|
|
234
|
+
* own declared ceiling. 0 for h3-max (no reference transport) and for anything
|
|
235
|
+
* that is not a MiniMax row. Mirrors refImageSurchargeCount() in
|
|
236
|
+
* slate/src/shared/pricing.ts. */
|
|
237
|
+
/** Reference IMAGES a generate_video call carries β the four free-reference
|
|
238
|
+
* arrays, combined, exactly as the desktop composer counts them. */
|
|
239
|
+
function minimaxRefImageCount(input) {
|
|
240
|
+
return ((input.ingredientAssetIds?.length ?? 0) +
|
|
241
|
+
(input.characterAssetIds?.length ?? 0) +
|
|
242
|
+
(input.environmentAssetIds?.length ?? 0) +
|
|
243
|
+
(input.styleAssetIds?.length ?? 0));
|
|
244
|
+
}
|
|
245
|
+
function minimaxRefSurchargeCount(model, referenceImages) {
|
|
246
|
+
if (!MINIMAX_MODELS.has(model))
|
|
247
|
+
return 0;
|
|
248
|
+
const cap = getModelCapability(model)?.maxIngredientImages ?? 0;
|
|
249
|
+
const n = Math.floor(referenceImages ?? 0);
|
|
250
|
+
if (!Number.isFinite(n) || n <= 0)
|
|
251
|
+
return 0;
|
|
252
|
+
return Math.max(0, Math.min(n, cap) - MINIMAX_FREE_REF_IMAGES);
|
|
253
|
+
}
|
|
254
|
+
/** The exact `model` ids `slates_edit_video` accepts. Edit rows are deliberately
|
|
255
|
+
* NOT in VIDEO_MODELS β they take a source clip, not frames. */
|
|
256
|
+
export const EDIT_VIDEO_MODELS = [
|
|
257
|
+
'kling-v3.0-omni-edit',
|
|
258
|
+
'kling-v3.0-omni-pro-edit',
|
|
259
|
+
'omni-flash-edit',
|
|
260
|
+
'seedance-2.5-edit',
|
|
261
|
+
];
|
|
262
|
+
/**
|
|
263
|
+
* Source-clip length an edit engine accepts, read from the SSOT.
|
|
264
|
+
*
|
|
265
|
+
* On an edit row the output length FOLLOWS the source clip, so the model's
|
|
266
|
+
* declared duration window is exactly the legal clip window β Kling 3-15s,
|
|
267
|
+
* Omni Flash 3-10s, Seedance 2.5 4-30s. These three pairs used to be typed
|
|
268
|
+
* inline as `isSeedanceEdit ? 4 : 3` / `isSeedanceEdit ? 30 : isOmniFlashEdit ?
|
|
269
|
+
* 10 : 15`, i.e. a fourth copy of registry data keyed on model-id comparisons.
|
|
270
|
+
*/
|
|
271
|
+
function editClipBounds(model) {
|
|
272
|
+
const d = getModelCapability(model)?.duration;
|
|
273
|
+
if (!d)
|
|
274
|
+
throw new Error(`editClipBounds: no duration capability for "${model}"`);
|
|
275
|
+
return { min: d.min, max: d.max };
|
|
276
|
+
}
|
|
277
|
+
/** Resolutions any edit engine exposes as a PARAM. Only Seedance's edit price
|
|
278
|
+
* moves with resolution; the Kling and Omni Flash rows are fixed to the source. */
|
|
279
|
+
const EDIT_VIDEO_RESOLUTIONS = videoResolutionUnion(['seedance-2.5-edit']);
|
|
136
280
|
export const estimateGenerationCost = {
|
|
137
281
|
id: 'slates_estimate_generation_cost',
|
|
138
282
|
description: 'Pre-flight cost estimate. Call before any generate_* op so the user sees "this will cost N credits" up front. Takes the SAME base model ids as the generate ops (video: "seedance-2" + duration + videoResolution; image: "nano-banana-2" + resolution) β exact registry cost keys also work. Pairs with the confirm gate.',
|
|
139
283
|
input: z.object({
|
|
140
284
|
model: z.string().describe('Base model id as passed to the generate op (e.g. "seedance-2", "kling-v3.0-std", "nano-banana-2") or an exact registry cost key ("nano-banana-2-2k", "seedance-2-1080p-8s")'),
|
|
141
285
|
quantity: z.number().int().min(1).max(10).optional().describe('Number of generations (default 1)'),
|
|
142
|
-
|
|
143
|
-
|
|
286
|
+
// Video windows and resolutions GENERATED from MODEL_CAPABILITIES. The
|
|
287
|
+
// hand-typed "Video 3-15" here was wrong the day seedance-2.5 (4-30s)
|
|
288
|
+
// shipped, and "Seedance defaults to 1080p" was wrong for 2.5, which has no
|
|
289
|
+
// 1080p at all.
|
|
290
|
+
duration: z.number().int().min(1).max(360).optional().describe(`Seconds; cost scales linearly. Required with a video or audio base id. Video per model: ${describeDurations(VIDEO_MODELS)}. Audio: seed-audio 3-120 (β οΈ the requested duration IS the bill), eleven-sfx 1-22.`),
|
|
291
|
+
videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe(`Video only. Omitted, each model quotes at its own default. Per model: ${describeVideoResolutions(VIDEO_MODELS)}`),
|
|
144
292
|
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k = gpt-image-2 1440p class).'),
|
|
145
293
|
quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only β quality tier (default medium).'),
|
|
146
294
|
sound: z.boolean().optional().describe('Veo only β audio flag changes the cost key.'),
|
|
147
295
|
seedanceFace: z.boolean().optional().describe('Seedance AI-face route (pricier key).'),
|
|
148
296
|
seedanceRealFace: z.boolean().optional().describe('Seedance consented real-face route (premium key).'),
|
|
297
|
+
referenceImages: z.number().int().min(0).optional().describe(`minimax-h3 only β how many reference IMAGES the generation will carry. The first ${MINIMAX_FREE_REF_IMAGES} are free and each one after that is a paid dimension of the cost key, so a quote that omits this UNDER-REPORTS a reference-heavy job. Ignored by every other model (minimax-h3-max takes no references at all).`),
|
|
149
298
|
}),
|
|
150
299
|
async run(input, ctx) {
|
|
151
300
|
const registry = await ctx.cloud().get('/api/agent/models');
|
|
@@ -218,20 +367,31 @@ export const estimateGenerationCost = {
|
|
|
218
367
|
message: `"${resolved.model}" cost scales with duration β pass duration (seconds) to estimate.`,
|
|
219
368
|
});
|
|
220
369
|
}
|
|
370
|
+
// Refuse an out-of-range combination rather than quoting a key that does
|
|
371
|
+
// not exist β the generate op gates the same way, from the same data, so
|
|
372
|
+
// a quote can never promise something generation then refuses.
|
|
373
|
+
const capErr = assertVideoCapabilities({
|
|
374
|
+
model: resolved.model,
|
|
375
|
+
videoResolution: input.videoResolution ?? resolved.videoResolution,
|
|
376
|
+
duration,
|
|
377
|
+
});
|
|
378
|
+
if (capErr)
|
|
379
|
+
return ok(capErr);
|
|
221
380
|
key = videoCostKey({
|
|
222
381
|
model: resolved.model,
|
|
223
382
|
duration,
|
|
224
383
|
videoResolution: input.videoResolution ??
|
|
225
384
|
resolved.videoResolution ??
|
|
226
385
|
// Seedance quotes are resolution-scaled, so a missing resolution has
|
|
227
|
-
// to fall back to the model's OWN default β
|
|
228
|
-
// and a blanket
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
386
|
+
// to fall back to the model's OWN default β the ladders differ per
|
|
387
|
+
// row (2.5 has no 4K) and a blanket literal quotes a key that does
|
|
388
|
+
// not exist. Read from the SSOT; this used to be a hand-typed
|
|
389
|
+
// model-id ternary.
|
|
390
|
+
defaultVideoResolutionFor(resolved.model),
|
|
232
391
|
sound: input.sound ?? resolved.sound,
|
|
233
392
|
seedanceFace: input.seedanceFace ?? resolved.seedanceFace,
|
|
234
393
|
seedanceRealFace: input.seedanceRealFace,
|
|
394
|
+
referenceImages: input.referenceImages ?? resolved.referenceImages,
|
|
235
395
|
});
|
|
236
396
|
}
|
|
237
397
|
}
|
|
@@ -872,6 +1032,30 @@ async function previewAssets(ctx, refs) {
|
|
|
872
1032
|
}
|
|
873
1033
|
return out;
|
|
874
1034
|
}
|
|
1035
|
+
/** The exact `model` ids `slates_generate_image` accepts. */
|
|
1036
|
+
export const IMAGE_MODELS = [
|
|
1037
|
+
'nano-banana-2',
|
|
1038
|
+
'nano-banana-2-lite',
|
|
1039
|
+
'nano-banana-pro',
|
|
1040
|
+
'gpt-image-2',
|
|
1041
|
+
'flux-2-max',
|
|
1042
|
+
'seedream-5-lite',
|
|
1043
|
+
];
|
|
1044
|
+
/**
|
|
1045
|
+
* The aspect ratios an image generation can carry β the UNION over what these
|
|
1046
|
+
* models declare in `MODEL_CAPABILITIES`, generated so the enum cannot hold a
|
|
1047
|
+
* value no model accepts. It used to be a hand-typed eleven-value list whose
|
|
1048
|
+
* `9:21` exists in ZERO models; that phantom is gone by construction.
|
|
1049
|
+
*
|
|
1050
|
+
* β οΈ STILL A UNION, NOT A PER-MODEL CHECK. `gpt-image-2` takes five of these
|
|
1051
|
+
* ten and the other five would be accepted here. The image param surface has
|
|
1052
|
+
* not been audited (aspect ratios, resolution classes, per-model reference
|
|
1053
|
+
* caps) β that audit is the named follow-up in
|
|
1054
|
+
* `slate/docs/plan-docs/2026-08-16-MODEL-CAPABILITY-SSOT.md` Β§5. The machinery
|
|
1055
|
+
* to close it already exists: call `checkAspectRatio(model, ratio)` the way
|
|
1056
|
+
* `assertVideoCapabilities` does below.
|
|
1057
|
+
*/
|
|
1058
|
+
const IMAGE_ASPECT_RATIOS = aspectRatioUnion(IMAGE_MODELS);
|
|
875
1059
|
// Registry cost-key for an image model+resolution. Mirrors imageCreditKey()
|
|
876
1060
|
// in slate/src/shared/pricing.ts β MUST byte-match it (the same hard rule as
|
|
877
1061
|
// videoCostKey): NB2/NB Pro price per resolution, FLUX.2 Max prices per
|
|
@@ -895,11 +1079,11 @@ export const generateImage = {
|
|
|
895
1079
|
description: 'Generate an image via Slates credits. Models: nano-banana-2 (default), nano-banana-2-lite (fast/cheap drafts, 1K only), nano-banana-pro (hero-frame/typography premium), gpt-image-2 (sharp text / character sheets / grids; quality medium|high), flux-2-max (photoreal, less censored), seedream-5-lite (cheapest flat, less censored). Which model for which job: read the slates-model-selection skill. Pass projectId to save into a Slates project (recommended β asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing β defaults waste credits). Cost > $0.50 returns requires_confirm β pass confirm=true after explicit user OK. MCP/CLI generation always charges credits. No skill files installed? Call slates_get_prompting_guide with the model\'s topic (and \'slates-cost-discipline\') before first use.',
|
|
896
1080
|
input: z.object({
|
|
897
1081
|
prompt: z.string().min(1).max(4000),
|
|
898
|
-
model:
|
|
1082
|
+
model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
|
|
899
1083
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Renderer refreshes live. Required for every model except nano-banana-2.'),
|
|
900
1084
|
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Pick deliberately: 1k drafts, 2k hero shots, 4k print/final. nano-banana-2-lite is 1k-only. gpt-image-2 classes: 1k=1024Β², 2k=1080p, 3k=1440p, 4k=2160p (3k is gpt-image-2 only). Never default this.'),
|
|
901
1085
|
quality: z.enum(['medium', 'high']).optional().describe('gpt-image-2 only. medium (default) = sharp text, fast, the value seat; high = max text precision + reasoning at ~4Γ the price. Ignored by other models.'),
|
|
902
|
-
aspectRatio:
|
|
1086
|
+
aspectRatio: zEnum(IMAGE_ASPECT_RATIOS).optional().describe(`Pick deliberately from the use case. Cinematic β 16:9. TikTok/Reels/Story β 9:16. IG square β 1:1. Ultra-wide β 21:9. Ask the user when ambiguous. Per model: ${describeAspectRatios(IMAGE_MODELS)}`),
|
|
903
1087
|
count: z.number().int().min(1).max(4).optional(),
|
|
904
1088
|
referenceImageUrls: z.array(z.string().url()).max(14).optional().describe('Headless (no-projectId) nano-banana-2 only: up to 14 ref URLs. For projectId runs, upload refs with slates_upload_reference_image first. Always label each image\'s role in the prompt text.'),
|
|
905
1089
|
referenceAssetIds: z.array(z.string()).max(14).optional().describe("Project assets to use as reference/ingredient images β asset UUIDs or badge codes (\"IMG-A8\"); codes resolve against the project at call time. Requires projectId. For nano-banana-2 up to 14 refs; FLUX/Seedream route to their edit endpoints with lower per-model caps. Label each reference's role in the prompt text."),
|
|
@@ -922,7 +1106,8 @@ export const generateImage = {
|
|
|
922
1106
|
missing,
|
|
923
1107
|
message: `Missing required field(s): ${missing.join(', ')}. ` +
|
|
924
1108
|
`Read the slates-prompting-nano-banana-2 + slates-cost-discipline skills, ` +
|
|
925
|
-
|
|
1109
|
+
// Generated from MODEL_CAPABILITIES β never retype a ratio list.
|
|
1110
|
+
`or ask the user. Aspect ratio by model: ${describeAspectRatios(IMAGE_MODELS)}. ` +
|
|
926
1111
|
`Resolution options: 1k 2k 4k (same price band β pick by need, not cost).`,
|
|
927
1112
|
});
|
|
928
1113
|
}
|
|
@@ -1310,23 +1495,52 @@ export const editImage = {
|
|
|
1310
1495
|
},
|
|
1311
1496
|
};
|
|
1312
1497
|
// ββ Generate video ββββββββββββββββββββββββββββββββββββββββββββββ
|
|
1313
|
-
//
|
|
1314
|
-
//
|
|
1315
|
-
//
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
|
|
1320
|
-
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1498
|
+
//
|
|
1499
|
+
// VIDEO_MODELS and the capability-derived param vocabulary are declared near the
|
|
1500
|
+
// top of this file: `slates_estimate_generation_cost`'s schema is evaluated at
|
|
1501
|
+
// module load and reads them, so they have to initialise before it.
|
|
1502
|
+
/**
|
|
1503
|
+
* Reject an illegal aspectRatio / videoResolution / duration BEFORE submit,
|
|
1504
|
+
* naming the legal set. Returns a `requires_clarification` payload or null.
|
|
1505
|
+
*
|
|
1506
|
+
* β οΈ WHY THIS IS A RUN()-TIME GUARD AND NOT A SCHEMA `superRefine`. The op
|
|
1507
|
+
* accepts registry COST-KEY spellings as `model` ("seedance-2-1080p-8s") and
|
|
1508
|
+
* `resolveVideoModel()` unpacks them into base id + duration + resolution at
|
|
1509
|
+
* the top of `run()`. A schema-level refinement runs BEFORE that, so it would
|
|
1510
|
+
* see an unnormalised model id and a `duration` that is still undefined β it
|
|
1511
|
+
* would reject legal calls and wave illegal ones through. The check has to sit
|
|
1512
|
+
* after normalisation, which is here. It also lets the failure arrive as the
|
|
1513
|
+
* same teaching `requires_clarification` shape as every other gate in this op,
|
|
1514
|
+
* rather than a raw Zod error the agent has to guess its way out of.
|
|
1515
|
+
*
|
|
1516
|
+
* `promptMode` matters: Veo's reference-to-video endpoint is 8s only, declared
|
|
1517
|
+
* as `duration.modeOverrides.ingredients`. Free reference images with no
|
|
1518
|
+
* first/last frame IS ingredients mode β the same condition
|
|
1519
|
+
* `buildFalVeoRequest` uses to pick the ref2v endpoint.
|
|
1520
|
+
*/
|
|
1521
|
+
function assertVideoCapabilities(input) {
|
|
1522
|
+
const freeRefs = (input.ingredientAssetIds?.length ?? 0) +
|
|
1523
|
+
(input.characterAssetIds?.length ?? 0) +
|
|
1524
|
+
(input.environmentAssetIds?.length ?? 0) +
|
|
1525
|
+
(input.styleAssetIds?.length ?? 0);
|
|
1526
|
+
const promptMode = freeRefs > 0 && !input.firstFrameAssetId && !input.lastFrameAssetId ? 'ingredients' : undefined;
|
|
1527
|
+
const ratioErr = checkAspectRatio(input.model, input.aspectRatio, AGENT_ROUTE_PROVIDER);
|
|
1528
|
+
if (ratioErr)
|
|
1529
|
+
return { requires_clarification: true, missing: ['aspectRatio'], message: ratioErr };
|
|
1530
|
+
const resErr = checkVideoResolution(input.model, input.videoResolution);
|
|
1531
|
+
if (resErr)
|
|
1532
|
+
return { requires_clarification: true, missing: ['videoResolution'], message: resErr };
|
|
1533
|
+
// Duration LAST: its legal window depends on the resolution, so a bad
|
|
1534
|
+
// resolution must be reported first or the duration message names a window
|
|
1535
|
+
// derived from a value the model never had.
|
|
1536
|
+
const durErr = checkDuration(input.model, input.duration, {
|
|
1537
|
+
videoResolution: input.videoResolution,
|
|
1538
|
+
promptMode,
|
|
1539
|
+
});
|
|
1540
|
+
if (durErr)
|
|
1541
|
+
return { requires_clarification: true, missing: ['duration'], message: durErr };
|
|
1542
|
+
return null;
|
|
1543
|
+
}
|
|
1330
1544
|
// Model β registry cost-key. Each provider's keys ship with their own
|
|
1331
1545
|
// shape (verified against /api/agent/models):
|
|
1332
1546
|
// Kling: kling-v3-{standard|pro|omni}-{N}s β note the user-facing
|
|
@@ -1347,6 +1561,13 @@ const KLING_TIER_MAP = {
|
|
|
1347
1561
|
// Exported for scripts/pricing-consistency-check.mjs (slates-api repo), which
|
|
1348
1562
|
// asserts this builder byte-matches the desktop's klingCreditKey/seedanceCreditKey.
|
|
1349
1563
|
export function videoCostKey(input) {
|
|
1564
|
+
// EXACT-ID MAP, NEVER A PREFIX: 'minimax-h3-max' starts with 'minimax-h3'.
|
|
1565
|
+
// Mirrors minimaxCreditKey() in slate/src/shared/pricing.ts.
|
|
1566
|
+
if (MINIMAX_MODELS.has(input.model)) {
|
|
1567
|
+
const res = input.videoResolution ?? defaultVideoResolutionFor(input.model);
|
|
1568
|
+
const k = minimaxRefSurchargeCount(input.model, input.referenceImages);
|
|
1569
|
+
return `${input.model}-${res}-${input.duration}s${k > 0 ? `-ref${k}` : ''}`;
|
|
1570
|
+
}
|
|
1350
1571
|
if (input.model.startsWith('seedance')) {
|
|
1351
1572
|
// Mirrors seedanceCreditKey() in slate/src/shared/pricing.ts (version Γ face
|
|
1352
1573
|
// Γ vref Γ res Γ duration). AI-face route bills the `-face-` key (~45% over
|
|
@@ -1354,10 +1575,11 @@ export function videoCostKey(input) {
|
|
|
1354
1575
|
// (fal partner endpoint). A reference video flips to `-vref-{res}-{T}s`,
|
|
1355
1576
|
// T = in + out.
|
|
1356
1577
|
//
|
|
1357
|
-
// β οΈ EVERY BOUND HERE IS VERSION-SCOPED. 2.5
|
|
1358
|
-
// and takes references to 30s combined β so its vref total
|
|
1359
|
-
// 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would
|
|
1360
|
-
// fraction of the real bill.
|
|
1578
|
+
// β οΈ EVERY BOUND HERE IS VERSION-SCOPED. 2.5 runs 480p/720p/1080p (no 4K),
|
|
1579
|
+
// reaches 30s, and takes references to 30s combined β so its vref total
|
|
1580
|
+
// reaches 60, DOUBLE 2.0's ceiling of 30. Clamping a 2.5 quote at 30 would
|
|
1581
|
+
// quote a real key at a fraction of the real bill. The ladder itself lives in
|
|
1582
|
+
// MODEL_CAPABILITIES; only the per-version DEFAULT is stated below.
|
|
1361
1583
|
const v25 = input.model.startsWith('seedance-2.5');
|
|
1362
1584
|
const res = input.videoResolution ?? (v25 ? '720p' : '1080p');
|
|
1363
1585
|
const face = input.seedanceRealFace ? '-realface' : input.seedanceFace ? '-face' : '';
|
|
@@ -1419,9 +1641,11 @@ export function omniFlashEditCostKey(duration) {
|
|
|
1419
1641
|
* 2.0: refs 2β15s + output 4β15s. 2.5: refs to 30s + output to 30s. */
|
|
1420
1642
|
const SEEDANCE_20_VREF_MAX_TOTAL = 30;
|
|
1421
1643
|
const SEEDANCE_25_VREF_MAX_TOTAL = 60;
|
|
1422
|
-
/** Seedance 2.5 source-clip bounds for the edit task type
|
|
1423
|
-
|
|
1424
|
-
|
|
1644
|
+
/** Seedance 2.5 source-clip bounds for the edit task type β READ from the
|
|
1645
|
+
* capability SSOT, not typed. Kept as named exports because published builds
|
|
1646
|
+
* import them. */
|
|
1647
|
+
export const SEEDANCE_25_EDIT_MIN_SECONDS = editClipBounds('seedance-2.5-edit').min;
|
|
1648
|
+
export const SEEDANCE_25_EDIT_MAX_SECONDS = editClipBounds('seedance-2.5-edit').max;
|
|
1425
1649
|
// Seedance 2.5 video-edit cost key β mirrors seedanceCreditKey() in
|
|
1426
1650
|
// slate/src/shared/pricing.ts for the edit row (must byte-match; checked by the
|
|
1427
1651
|
// slates-api pricing-consistency script). Unlike the Kling and Omni Flash edit
|
|
@@ -1515,10 +1739,22 @@ function resolveVideoModel(raw) {
|
|
|
1515
1739
|
out.duration = parseInt(dur[1], 10);
|
|
1516
1740
|
s = s.replace(/-(\d+)s\b/, '');
|
|
1517
1741
|
}
|
|
1518
|
-
|
|
1742
|
+
// MiniMax H3's paid-reference suffix, stripped like every other key dimension
|
|
1743
|
+
// so a pasted `minimax-h3-768p-10s-ref2` resolves instead of erroring. The
|
|
1744
|
+
// number it carries is K (images PAST the free five), so it is converted back
|
|
1745
|
+
// to a TOTAL before anything can re-surcharge it.
|
|
1746
|
+
const ref = /-ref(\d+)\b/.exec(s);
|
|
1747
|
+
if (ref) {
|
|
1748
|
+
out.referenceImages = MINIMAX_FREE_REF_IMAGES + parseInt(ref[1], 10);
|
|
1749
|
+
s = s.replace(/-ref(\d+)\b/, '');
|
|
1750
|
+
}
|
|
1751
|
+
// The RESOLUTION vocabulary is GENERATED from MODEL_CAPABILITIES β the
|
|
1752
|
+
// hand-typed list that stood here went stale the day 768p and 2k shipped.
|
|
1753
|
+
const resRe = new RegExp(`-(${VIDEO_RESOLUTION_VOCAB.join('|')})\\b`);
|
|
1754
|
+
const res = resRe.exec(s);
|
|
1519
1755
|
if (res) {
|
|
1520
1756
|
out.videoResolution = res[1];
|
|
1521
|
-
s = s.replace(
|
|
1757
|
+
s = s.replace(resRe, '');
|
|
1522
1758
|
}
|
|
1523
1759
|
if (/-audio\b/.test(s)) {
|
|
1524
1760
|
out.sound = true;
|
|
@@ -1557,6 +1793,19 @@ function resolveVideoModel(raw) {
|
|
|
1557
1793
|
'gemini-omni-flash': 'omni-flash',
|
|
1558
1794
|
'gemini-omni-flash-preview': 'omni-flash',
|
|
1559
1795
|
'omni-flash-preview': 'omni-flash',
|
|
1796
|
+
// The MAX spellings must come out as MAX. Bare `minimax`, `h3` and
|
|
1797
|
+
// `hailuo-3` all mean the BASE row β it holds the full ladder and the
|
|
1798
|
+
// references, and it is cheaper at the tier they share.
|
|
1799
|
+
'minimax-h3-max': 'minimax-h3-max',
|
|
1800
|
+
'minimax-h3max': 'minimax-h3-max',
|
|
1801
|
+
'h3-max': 'minimax-h3-max',
|
|
1802
|
+
'hailuo-3-max': 'minimax-h3-max',
|
|
1803
|
+
minimax: 'minimax-h3',
|
|
1804
|
+
'minimax-h3': 'minimax-h3',
|
|
1805
|
+
h3: 'minimax-h3',
|
|
1806
|
+
'hailuo-3': 'minimax-h3',
|
|
1807
|
+
'hailuo-3.0': 'minimax-h3',
|
|
1808
|
+
hailuo: 'minimax-h3',
|
|
1560
1809
|
};
|
|
1561
1810
|
if (aliases[s]) {
|
|
1562
1811
|
out.model = aliases[s];
|
|
@@ -1582,21 +1831,39 @@ function promptingSkillFor(model) {
|
|
|
1582
1831
|
return 'slates-prompting-seedance';
|
|
1583
1832
|
if (model.startsWith('omni-flash'))
|
|
1584
1833
|
return 'slates-prompting-omni-flash';
|
|
1834
|
+
// ONE skill covers both H3 seats β the prompt grammar is identical and only
|
|
1835
|
+
// the ladder and the reference transport differ β so a prefix is right here.
|
|
1836
|
+
if (model.startsWith('minimax-h3'))
|
|
1837
|
+
return 'slates-prompting-minimax-h3';
|
|
1585
1838
|
return 'slates-cost-discipline';
|
|
1586
1839
|
}
|
|
1587
1840
|
export const generateVideo = {
|
|
1588
1841
|
id: 'slates_generate_video',
|
|
1589
|
-
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3) β video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm β pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
|
|
1842
|
+
description: 'Generate video via Slates credits. REQUIRED before calling: read slates-model-selection (the routing doctrine), slates-cost-discipline, and the matching per-model prompting skill (slates-prompting-seedance / slates-prompting-seedance-2-5 / slates-prompting-kling-v3 / slates-prompting-veo-3 / slates-prompting-minimax-h3) β video models prompt very differently; load them via slates_get_prompting_guide if no skill files are installed. Read slates-content-policy when the scene involves conflict, creatures, crowds, destruction, weapons, or young characters. projectId, aspectRatio, and duration are required (requires_clarification otherwise). Cost > $0.50 returns requires_confirm β pass confirm=true after explicit user OK. Image-to-video via firstFrameAssetId; first+last frames = Veo/Seedance only; ingredients via ingredientAssetIds (Kling Omni / Seedance). Asset params take UUIDs or badge codes ("IMG-A8").',
|
|
1590
1843
|
input: z.object({
|
|
1591
1844
|
prompt: z.string().min(1).max(4000),
|
|
1592
|
-
|
|
1845
|
+
// ROUTING doctrine only. Every capability number was stripped on 2026-08-16
|
|
1846
|
+
// and now lives in the aspectRatio / duration / videoResolution descriptions
|
|
1847
|
+
// below, generated from MODEL_CAPABILITIES. "Veo = 16:9 only" and
|
|
1848
|
+
// "seedance-2.5 480p/720p" were both stated here AND there, and the two
|
|
1849
|
+
// copies disagreed β and the second of those went stale on 2026-08-24 when
|
|
1850
|
+
// 2.5 gained 1080p, which is exactly the failure mode generating it fixes.
|
|
1851
|
+
model: z.string().describe(`One of: ${VIDEO_MODELS.join(' | ')}. Pass the BASE id β duration and videoResolution are separate params (registry cost keys like "kling-v3-standard-8s" auto-resolve). Route per the slates-model-selection skill: Kling std = general-purpose DEFAULT, Seedance 2 = premium physics/effects/hero tier, seedance-2.5 = a SECOND SEAT beside it (longer takes, far more references, audio-only refs β but no 4K, and dearer than seedance-2 at every shared resolution, so stay on seedance-2 unless length or reference count is the point), Veo = native-synced-audio niche only, never the default, omni-flash = cheap tier with audio included (t2v, single-start-frame i2v, or reference images; no last frame, no video/audio refs), minimax-h3 = the AUTHORED-AUDIO seat (dialogue, scene sound and score directed as three separate layers in one pass, plus declared reference relationships; reference images past the fifth are a PAID key dimension β pass referenceImages when quoting), minimax-h3-max = the same model post-trained by fal for SPEED, capped at 768p, no references of any kind, and DEARER than minimax-h3 at the tier they share β a deliberate pick, never a default and never the cheap H3. All are VIDEO-only. Each model's legal aspect ratios, durations and resolutions are in those params' own descriptions β read them there, not from memory. For per-call cost, call slates_estimate_generation_cost.`),
|
|
1593
1852
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Strongly recommended β the desktop UI shows a progress card live and the asset appears when complete.'),
|
|
1594
|
-
|
|
1595
|
-
|
|
1596
|
-
|
|
1853
|
+
// π¨ THESE THREE DESCRIPTIONS ARE GENERATED FROM `MODEL_CAPABILITIES`.
|
|
1854
|
+
// Never hand-write a ratio, resolution or duration into them again β every
|
|
1855
|
+
// one of the hand-written claims that stood here had drifted, and an
|
|
1856
|
+
// out-of-set value is accepted by the client and rejected by the provider
|
|
1857
|
+
// ASYNCHRONOUSLY, after the job queues and credits reserve.
|
|
1858
|
+
aspectRatio: zEnum(VIDEO_ASPECT_RATIOS).optional().describe(`NOT every model takes every value β an out-of-set ratio is REFUSED before submit, not silently ignored. Per model: ${describeAspectRatios(VIDEO_MODELS, AGENT_ROUTE_PROVIDER)}`),
|
|
1859
|
+
duration: z.number().int().min(VIDEO_DURATION_BOUNDS.min).max(VIDEO_DURATION_BOUNDS.max).optional().describe(`Seconds. Cost scales linearly β a 30s seedance-2.5 take is several hundred credits, so be explicit rather than defaulting. Per model: ${describeDurations(VIDEO_MODELS)}`),
|
|
1860
|
+
videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe(`An unsupported resolution is REJECTED, never downgraded. Per model: ${describeVideoResolutions(VIDEO_MODELS)}. 4K video is Pro-only (the server returns PRO_REQUIRED for a base-tier account).`),
|
|
1597
1861
|
firstFrameAssetId: z.string().optional().describe('Starting frame for image-to-video: asset UUID or badge code ("IMG-A8") β codes resolve against the project at call time, so a code the user just spoke is always safe to pass.'),
|
|
1598
1862
|
lastFrameAssetId: z.string().optional().describe('Ending frame (UUID or badge code). Veo and Seedance only. Pairs with firstFrameAssetId for guided transitions.'),
|
|
1599
|
-
ingredientAssetIds: z.array(z.string()).max(30).optional().describe(
|
|
1863
|
+
ingredientAssetIds: z.array(z.string()).max(30).optional().describe(
|
|
1864
|
+
// Caps DERIVED from MODEL_CAPABILITIES β the hand-typed list omitted
|
|
1865
|
+
// kling-v3.0-omni-pro entirely and read as if 7 were an Omni Flash-only rule.
|
|
1866
|
+
`Visual reference / ingredient assets (UUIDs or badge codes). Cap per model (combined across ingredient/character/environment/style params): ${describeReferenceImageCaps(VIDEO_MODELS)}. More is not better: 2-4 strong references beat both extremes, and past 4 reference PEOPLE output stability drops on Seedance regardless of the cap.`),
|
|
1600
1867
|
characterAssetIds: z.array(z.string()).optional().describe('Character sheet assets (UUIDs or badge codes) β keeps a character consistent across the shot.'),
|
|
1601
1868
|
environmentAssetIds: z.array(z.string()).optional().describe('Environment reference assets (UUIDs or badge codes) β keeps a location/setting consistent across the shot.'),
|
|
1602
1869
|
styleAssetIds: z.array(z.string()).optional().describe('Style reference assets (UUIDs or badge codes) β locks the visual style of the shot.'),
|
|
@@ -1662,22 +1929,28 @@ export const generateVideo = {
|
|
|
1662
1929
|
missing,
|
|
1663
1930
|
message: `Missing required field(s): ${missing.join(', ')}. ` +
|
|
1664
1931
|
`Read the slates-cost-discipline + ${promptingSkillFor(input.model)} skills, ` +
|
|
1665
|
-
|
|
1666
|
-
|
|
1932
|
+
// Generated from MODEL_CAPABILITIES. The prose that stood here claimed
|
|
1933
|
+
// "Veo locks to 16:9", "Kling/Seedance support 1:1 16:9 9:16 4:3 3:4
|
|
1934
|
+
// 21:9" and "Kling 5-15s" β three wrong claims in one sentence.
|
|
1935
|
+
`or ask the user. ` +
|
|
1936
|
+
`Aspect ratio for ${input.model}: ${aspectRatiosFor(input.model, AGENT_ROUTE_PROVIDER).join(', ')}. ` +
|
|
1937
|
+
`Duration: ${describeDurations([input.model])}. ` +
|
|
1938
|
+
`Cost scales linearly with duration.`,
|
|
1667
1939
|
});
|
|
1668
1940
|
}
|
|
1669
|
-
//
|
|
1670
|
-
//
|
|
1671
|
-
//
|
|
1672
|
-
//
|
|
1941
|
+
// π¨ CAPABILITY GATE β the one that stops a client-side accept of a
|
|
1942
|
+
// server-side reject. Registry-driven and model-keyed: an aspect ratio,
|
|
1943
|
+
// resolution or duration the chosen model does not take is refused HERE,
|
|
1944
|
+
// before the job queues and credits reserve. Runs after normalisation so it
|
|
1945
|
+
// sees the resolved model and any params a cost key back-filled.
|
|
1946
|
+
const capErr = assertVideoCapabilities(input);
|
|
1947
|
+
if (capErr)
|
|
1948
|
+
return ok(capErr);
|
|
1949
|
+
// Omni Flash's SHAPE constraints β t2v, single-start-frame i2v, or ref2v
|
|
1950
|
+
// with reference IMAGES. No last frame, no video/audio refs. (Its duration
|
|
1951
|
+
// and resolution are covered by the capability gate above; the 3-10s check
|
|
1952
|
+
// that used to live here was a hand-typed duplicate of registry data.)
|
|
1673
1953
|
if (input.model === 'omni-flash') {
|
|
1674
|
-
if (input.duration < 3 || input.duration > 10) {
|
|
1675
|
-
return ok({
|
|
1676
|
-
requires_clarification: true,
|
|
1677
|
-
missing: ['duration'],
|
|
1678
|
-
message: `Omni Flash supports 3-10 seconds (you passed ${input.duration}s). Pick a duration in that range.`,
|
|
1679
|
-
});
|
|
1680
|
-
}
|
|
1681
1954
|
if (input.lastFrameAssetId ||
|
|
1682
1955
|
input.videoReferenceAssetId || input.audioReferenceAssetId ||
|
|
1683
1956
|
(input.videoReferenceAssetIds?.length ?? 0) > 0 ||
|
|
@@ -1685,40 +1958,98 @@ export const generateVideo = {
|
|
|
1685
1958
|
return ok({
|
|
1686
1959
|
requires_clarification: true,
|
|
1687
1960
|
missing: [],
|
|
1688
|
-
message:
|
|
1961
|
+
message: `Omni Flash takes a prompt, an optional start frame, and up to ${omniFlashRefCap} reference IMAGES β last frames and video/audio references are not supported. Drop those, or switch to seedance-2 (video/audio refs, last frame) or veo (last frame).`,
|
|
1689
1962
|
});
|
|
1690
1963
|
}
|
|
1691
1964
|
const refCount = (input.ingredientAssetIds?.length ?? 0) +
|
|
1692
1965
|
(input.characterAssetIds?.length ?? 0) +
|
|
1693
1966
|
(input.environmentAssetIds?.length ?? 0) +
|
|
1694
1967
|
(input.styleAssetIds?.length ?? 0);
|
|
1695
|
-
if (refCount >
|
|
1968
|
+
if (refCount > omniFlashRefCap) {
|
|
1696
1969
|
return ok({
|
|
1697
1970
|
requires_clarification: true,
|
|
1698
1971
|
missing: [],
|
|
1699
|
-
message: `Omni Flash takes at most
|
|
1972
|
+
message: `Omni Flash takes at most ${omniFlashRefCap} reference images combined (you passed ${refCount}). Trim the list.`,
|
|
1700
1973
|
});
|
|
1701
1974
|
}
|
|
1702
1975
|
}
|
|
1703
|
-
//
|
|
1704
|
-
//
|
|
1705
|
-
// a
|
|
1706
|
-
|
|
1707
|
-
|
|
1976
|
+
// MiniMax H3's SHAPE constraints. Counts and caps are read from the
|
|
1977
|
+
// capability SSOT; only the endpoint SHAPE is stated here, because it is
|
|
1978
|
+
// not a number the registry models: fal publishes text-to-video,
|
|
1979
|
+
// image-to-video and reference-to-video for `minimax/h3`, and only the
|
|
1980
|
+
// first two for `minimax/h3-max` (its reference-to-video 404s). The
|
|
1981
|
+
// reference endpoint has no frame parameters at all, so frames and
|
|
1982
|
+
// references are mutually exclusive β a shape mismatch, not a preference.
|
|
1983
|
+
if (MINIMAX_MODELS.has(input.model)) {
|
|
1984
|
+
const cap = getModelCapability(input.model);
|
|
1985
|
+
const refImages = minimaxRefImageCount(input);
|
|
1986
|
+
const refMedia = (input.videoReferenceAssetIds?.length ?? 0) +
|
|
1987
|
+
(input.audioReferenceAssetIds?.length ?? 0) +
|
|
1988
|
+
(input.videoReferenceAssetId ? 1 : 0) +
|
|
1989
|
+
(input.audioReferenceAssetId ? 1 : 0);
|
|
1990
|
+
const maxImages = cap?.maxIngredientImages ?? 0;
|
|
1991
|
+
const maxVideos = cap?.maxReferenceVideos ?? 0;
|
|
1992
|
+
const maxAudio = cap?.maxReferenceAudio ?? 0;
|
|
1993
|
+
const maxTotal = cap?.maxReferenceFilesTotal ?? 0;
|
|
1994
|
+
if (maxImages === 0 && (refImages > 0 || refMedia > 0)) {
|
|
1708
1995
|
return ok({
|
|
1709
1996
|
requires_clarification: true,
|
|
1710
|
-
missing: [
|
|
1711
|
-
message:
|
|
1997
|
+
missing: [],
|
|
1998
|
+
message: `${input.model} takes a prompt and up to two frames (start and/or end) β it has no reference endpoint at all, so reference images, video and audio cannot be sent. Drop them, or switch to minimax-h3, which reads ${getModelCapability('minimax-h3')?.maxIngredientImages ?? 9} images plus reference video and audio.`,
|
|
1712
1999
|
});
|
|
1713
2000
|
}
|
|
1714
|
-
if (
|
|
2001
|
+
if ((refImages > 0 || refMedia > 0) && (input.firstFrameAssetId || input.lastFrameAssetId)) {
|
|
1715
2002
|
return ok({
|
|
1716
2003
|
requires_clarification: true,
|
|
1717
|
-
missing: [
|
|
1718
|
-
message:
|
|
2004
|
+
missing: [],
|
|
2005
|
+
message: `${input.model} can use first/last frames OR references, not both β they are different endpoints and the reference one has no frame slots. Drop one side.`,
|
|
2006
|
+
});
|
|
2007
|
+
}
|
|
2008
|
+
if (refImages > maxImages) {
|
|
2009
|
+
return ok({
|
|
2010
|
+
requires_clarification: true,
|
|
2011
|
+
missing: [],
|
|
2012
|
+
message: `${input.model} takes at most ${maxImages} reference images combined (you passed ${refImages}). Trim the list β and note the first ${MINIMAX_FREE_REF_IMAGES} are free while each one after that adds a paid dimension to the cost key.`,
|
|
2013
|
+
});
|
|
2014
|
+
}
|
|
2015
|
+
if ((input.videoReferenceAssetIds?.length ?? 0) + (input.videoReferenceAssetId ? 1 : 0) > maxVideos) {
|
|
2016
|
+
return ok({
|
|
2017
|
+
requires_clarification: true,
|
|
2018
|
+
missing: [],
|
|
2019
|
+
message: `${input.model} takes at most ${maxVideos} reference videos.`,
|
|
2020
|
+
});
|
|
2021
|
+
}
|
|
2022
|
+
if ((input.audioReferenceAssetIds?.length ?? 0) + (input.audioReferenceAssetId ? 1 : 0) > maxAudio) {
|
|
2023
|
+
return ok({
|
|
2024
|
+
requires_clarification: true,
|
|
2025
|
+
missing: [],
|
|
2026
|
+
message: `${input.model} takes at most ${maxAudio} reference audio clips.`,
|
|
2027
|
+
});
|
|
2028
|
+
}
|
|
2029
|
+
if (maxTotal > 0 && refImages + refMedia > maxTotal) {
|
|
2030
|
+
return ok({
|
|
2031
|
+
requires_clarification: true,
|
|
2032
|
+
missing: [],
|
|
2033
|
+
message: `${input.model} takes at most ${maxTotal} reference files across all modalities (you passed ${refImages + refMedia}).`,
|
|
2034
|
+
});
|
|
2035
|
+
}
|
|
2036
|
+
// fal: "Audio cannot be the only reference input; provide at least one
|
|
2037
|
+
// reference image or video with it."
|
|
2038
|
+
const audioRefs = (input.audioReferenceAssetIds?.length ?? 0) + (input.audioReferenceAssetId ? 1 : 0);
|
|
2039
|
+
const videoRefs = (input.videoReferenceAssetIds?.length ?? 0) + (input.videoReferenceAssetId ? 1 : 0);
|
|
2040
|
+
if (audioRefs > 0 && refImages === 0 && videoRefs === 0) {
|
|
2041
|
+
return ok({
|
|
2042
|
+
requires_clarification: true,
|
|
2043
|
+
missing: [],
|
|
2044
|
+
message: `${input.model} rejects an audio-only reference set β add at least one reference image or video alongside it.`,
|
|
1719
2045
|
});
|
|
1720
2046
|
}
|
|
1721
2047
|
}
|
|
2048
|
+
// β The hand-written Veo duration block that stood here is GONE. It read
|
|
2049
|
+
// `![4,6,8].includes(duration)` plus "4K only at 8s" β and the registry
|
|
2050
|
+
// forces 8s at 1080p TOO, so it happily quoted a 4s 1080p Veo the provider
|
|
2051
|
+
// rejects. `assertVideoCapabilities` above covers both, plus the
|
|
2052
|
+
// reference-to-video 8s rule it never knew about, from the SSOT.
|
|
1722
2053
|
// Resolve every asset reference (UUID or badge code) against the
|
|
1723
2054
|
// project AT CALL TIME. Codes the user just spoke ("a8 is the one")
|
|
1724
2055
|
// resolve to whatever exists NOW β including assets created in the UI
|
|
@@ -1825,6 +2156,10 @@ export const generateVideo = {
|
|
|
1825
2156
|
sound: input.sound,
|
|
1826
2157
|
seedanceFace: input.seedanceFace,
|
|
1827
2158
|
seedanceRealFace: input.seedanceRealFace,
|
|
2159
|
+
// MiniMax H3 base: reference images past the fifth are a PAID key
|
|
2160
|
+
// dimension. Counted from the same four arrays the request sends, so the
|
|
2161
|
+
// quote and the server's own re-derivation see the same number.
|
|
2162
|
+
referenceImages: minimaxRefImageCount(input),
|
|
1828
2163
|
// Ξ£ ceil(d - 0.05) over every reference clip, both shapes β the same
|
|
1829
2164
|
// expression the desktop's estimateCost and the handler's key builder
|
|
1830
2165
|
// use. Quoting only the singular would understate a multi-clip call.
|
|
@@ -1845,7 +2180,9 @@ export const generateVideo = {
|
|
|
1845
2180
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
1846
2181
|
if (!entry) {
|
|
1847
2182
|
throw new Error(`Model variant not in registry: ${costKey}. ` +
|
|
1848
|
-
|
|
2183
|
+
// Prefixes DERIVED from the model list, so a new family cannot be
|
|
2184
|
+
// missing from the hint the way minimax-h3 was on day one.
|
|
2185
|
+
`Available video models: ${registry.models.filter((m) => VIDEO_MODELS.some((v) => m.model.startsWith(v.split('.')[0]))).map((m) => m.model).slice(0, 20).join(', ')}`);
|
|
1849
2186
|
}
|
|
1850
2187
|
const totalCents = creditCost(entry);
|
|
1851
2188
|
// Pre-flight confirm gate. Fires when:
|
|
@@ -2350,16 +2687,17 @@ export const generateMotionTransfer = {
|
|
|
2350
2687
|
// ββ Edit video (Kling O3 video-to-video) ββββββββββββββββββββββββ
|
|
2351
2688
|
export const editVideo = {
|
|
2352
2689
|
id: 'slates_edit_video',
|
|
2353
|
-
description: 'Edit an EXISTING video clip with one instruction β character swap, environment change, style transfer β in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3β15s clips, 720β3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3β10s clips, 720p output, PROMPT-ONLY β no refs, cheapest seat), or seedance-2.5-edit (4β30s clips β the ONLY engine that takes a clip over 15s;
|
|
2690
|
+
description: 'Edit an EXISTING video clip with one instruction β character swap, environment change, style transfer β in one pass, no masking. Original motion, camera, and audio are preserved; only what the prompt names changes. Use when a clip is ~90% right (fix it, don\'t re-roll it) or to AI-edit the user\'s own footage. Engines: Kling O3 edit (default; 3β15s clips, 720β3840px, subject/style refs via elements), omni-flash-edit (Gemini Omni Flash; 3β10s clips, 720p output, PROMPT-ONLY β no refs, cheapest seat), or seedance-2.5-edit (4β30s clips β the ONLY engine that takes a clip over 15s; up to 1080p, seedanceFace:true for AI-character faces). Cost = per second of OUTPUT (β clip length, rounded UP to the next second): omni-flash-edit β 19Β’/s β kling-v3.0-omni-edit β 19Β’/s, kling-v3.0-omni-pro-edit β 25Β’/s. Subjects to swap IN go as characterAssetIds (frontal + angle images become Kling elements β Kling models only); style refs as styleAssetIds; max 4 combined. seedance-2.5-edit is priced per second of output on the video-reference tier and bills roughly double a plain 2.5 generation of the same length, because every provider charges an edit on input + output seconds β always read the quote from the confirm gate rather than assuming. The edited clip saves as a NEW asset linked to its parent (chain edits freely). Routing: Kling edit is the default edit tool (element lock + audio intact); omni-flash-edit for cheap prompt-only footage-synced swaps; prefer Seedance edit/relocate only for style-transfer-heavy jobs β see slates-model-selection. Prompting: slates-prompting-kling-v3 Β§Edit / slates-prompting-omni-flash.',
|
|
2354
2691
|
input: z.object({
|
|
2355
2692
|
projectId: z.string().uuid().describe('Project the source clip lives in.'),
|
|
2356
2693
|
sourceVideoAssetId: z.string().describe('The VIDEO asset to edit β UUID or badge code ("VID-V3", bare "V3"); codes resolve against the project at call time. Kling: 3β15s clips; omni-flash-edit: 3β10s.'),
|
|
2357
2694
|
prompt: z.string().min(1).max(2500).describe('The change, not the whole scene β e.g. "replace the man with @marcus", "make it a rainy night", "turn the street into a neon Tokyo alley". Mention subjects with @name; the transport compiles them to Kling\'s @ElementN notation (Kling models). For omni-flash-edit keep it simple and add "Keep everything else the same."'),
|
|
2358
|
-
|
|
2695
|
+
// Clip windows and resolutions GENERATED from MODEL_CAPABILITIES.
|
|
2696
|
+
model: zEnum(EDIT_VIDEO_MODELS).optional().describe(`Default kling-v3.0-omni-edit. Pro (~25Β’/s vs ~19Β’/s) only for hero shots where fidelity matters. omni-flash-edit (~19Β’/s) for prompt-only edits β it takes NO character/style refs. seedance-2.5-edit is the ONLY engine that accepts a clip longer than 15s; it also takes NO character/style refs on this op. Source-clip length per engine: ${describeDurations(EDIT_VIDEO_MODELS)}. Output resolution: ${describeVideoResolutions(EDIT_VIDEO_MODELS)}`),
|
|
2359
2697
|
characterAssetIds: z.array(z.string()).max(4).optional().describe('Subject/element image assets to swap IN (UUIDs or badge codes). Each becomes a Kling element (@ElementN). KLING MODELS ONLY β rejected on omni-flash-edit.'),
|
|
2360
2698
|
styleAssetIds: z.array(z.string()).max(4).optional().describe('Style/appearance reference images (@ImageN). Max 4 combined with characterAssetIds. KLING MODELS ONLY β rejected on omni-flash-edit.'),
|
|
2361
2699
|
keepAudio: z.boolean().optional().describe('Preserve the original audio track (default true; Kling models only β omni-flash-edit and seedance-2.5-edit output carry their own audio).'),
|
|
2362
|
-
videoResolution:
|
|
2700
|
+
videoResolution: zEnum(EDIT_VIDEO_RESOLUTIONS).optional().describe(`seedance-2.5-edit ONLY (default ${defaultVideoResolutionFor('seedance-2.5-edit')}). Ignored by the Kling and Omni Flash engines, whose output follows the source clip.`),
|
|
2363
2701
|
seedanceFace: z.boolean().optional().describe('seedance-2.5-edit ONLY: set true when a CHARACTER FACE is visible in the clip. Faceless edits run on BytePlus; faces are blocked there and must route to the relaxed provider, which costs ~35% more. A face edit submitted without this flag is rejected by the provider, not silently downgraded.'),
|
|
2364
2702
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
2365
2703
|
confirm: z.boolean().optional().describe('Set true to bypass the cost confirm gate after the user OKs the spend.'),
|
|
@@ -2373,10 +2711,9 @@ export const editVideo = {
|
|
|
2373
2711
|
const model = input.model ?? 'kling-v3.0-omni-edit';
|
|
2374
2712
|
const isOmniFlashEdit = model === 'omni-flash-edit';
|
|
2375
2713
|
const isSeedanceEdit = model === 'seedance-2.5-edit';
|
|
2376
|
-
//
|
|
2377
|
-
//
|
|
2378
|
-
const minClipSeconds
|
|
2379
|
-
const maxClipSeconds = isSeedanceEdit ? SEEDANCE_25_EDIT_MAX_SECONDS : isOmniFlashEdit ? 10 : 15;
|
|
2714
|
+
// Source-clip window, read from the capability SSOT per engine β never a
|
|
2715
|
+
// model-id ternary. (Kling 3β15s, Omni Flash 3β10s, Seedance 2.5 4β30s.)
|
|
2716
|
+
const { min: minClipSeconds, max: maxClipSeconds } = editClipBounds(model);
|
|
2380
2717
|
if ((isOmniFlashEdit || isSeedanceEdit) && ((input.characterAssetIds?.length ?? 0) > 0 || (input.styleAssetIds?.length ?? 0) > 0)) {
|
|
2381
2718
|
throw new Error(`${model} takes the prompt and the source clip only on this op β no character/style reference images. Drop the refs, or switch to kling-v3.0-omni-edit which supports elements.`);
|
|
2382
2719
|
}
|
|
@@ -2419,7 +2756,10 @@ export const editVideo = {
|
|
|
2419
2756
|
const costKey = isSeedanceEdit
|
|
2420
2757
|
? seedanceEditCostKey({
|
|
2421
2758
|
duration: billedSeconds,
|
|
2422
|
-
|
|
2759
|
+
// Cast to the full union, never to the row's current two-or-three
|
|
2760
|
+
// values: MODEL_CAPABILITIES is the narrowing authority and that list
|
|
2761
|
+
// moves (1080p landed on the 2.5 edit row 2026-08-24).
|
|
2762
|
+
videoResolution: (input.videoResolution ?? defaultVideoResolutionFor('seedance-2.5-edit')),
|
|
2423
2763
|
seedanceFace: input.seedanceFace === true,
|
|
2424
2764
|
})
|
|
2425
2765
|
: isOmniFlashEdit
|
|
@@ -3152,6 +3492,15 @@ function resolveGuideTopic(topic) {
|
|
|
3152
3492
|
return 'slates-prompting-veo-3';
|
|
3153
3493
|
if (t.startsWith('omni-flash') || t.startsWith('gemini-omni') || t === 'omni flash')
|
|
3154
3494
|
return 'slates-prompting-omni-flash';
|
|
3495
|
+
// MiniMax H3 β both seats share one skill. Placed BEFORE the seed/seedance
|
|
3496
|
+
// block for the same reason seed-audio is: no prefix collision exists today,
|
|
3497
|
+
// but a `minimax-*` id must never fall through to a Seedance guide.
|
|
3498
|
+
if (t.startsWith('minimax') ||
|
|
3499
|
+
t.startsWith('hailuo') ||
|
|
3500
|
+
t === 'h3' ||
|
|
3501
|
+
t.startsWith('h3-')) {
|
|
3502
|
+
return 'slates-prompting-minimax-h3';
|
|
3503
|
+
}
|
|
3155
3504
|
if (t.startsWith('kling-mc'))
|
|
3156
3505
|
return 'slates-prompting-motion-transfer';
|
|
3157
3506
|
if (t === 'edit-video' || t === 'video-edit' || t === 'edit video' || t === 'video edit')
|