@slatesvideo/shared 0.5.9 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -0
- package/dist/index.js +6 -0
- package/dist/operations/index.d.ts +34 -13
- package/dist/operations/index.js +426 -77
- package/dist/prompts/index.d.ts +1 -0
- package/dist/prompts/index.js +4 -0
- package/dist/prompts/model-capabilities.d.ts +124 -0
- package/dist/prompts/model-capabilities.js +624 -0
- package/dist/prompts/model-facts.d.ts +11 -3
- package/dist/prompts/model-facts.js +87 -51
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +64 -2
- package/dist/skills/content.js +8 -6
- package/exports/slates-prompt-builder/generated/SKILL.md +1 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +5 -5
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +5 -1
- package/skills/slates-direct-response-ad.md +2 -0
- package/skills/slates-model-selection.md +17 -11
- package/skills/slates-one-prompt-film.md +3 -2
- package/skills/slates-prompting-gpt-image-2.md +70 -40
- package/skills/slates-prompting-minimax-h3.md +287 -0
- package/skills/slates-prompting-seedance-2-5.md +23 -17
- package/skills/slates-prompting-veo-3.md +2 -2
- package/skills/slates-ugc-influencer-ad.md +307 -0
|
@@ -0,0 +1,624 @@
|
|
|
1
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
2
|
+
// MODEL_CAPABILITIES — the SSOT for what a model will ACCEPT.
|
|
3
|
+
//
|
|
4
|
+
// Aspect ratios (including per-provider overrides), video resolutions, duration
|
|
5
|
+
// ranges and reference caps. One definition, imported by everything: the
|
|
6
|
+
// desktop's `MODEL_REGISTRY` (slate/src/shared/pricing.ts) spreads these fields
|
|
7
|
+
// into every entry, `MODEL_FACTS` derives its reference caps from them, and the
|
|
8
|
+
// MCP/CLI op surface both VALIDATES against them and GENERATES its `.describe()`
|
|
9
|
+
// prose from them.
|
|
10
|
+
//
|
|
11
|
+
// 🚨 WHY THIS FILE EXISTS. Until 2026-08-16 the op surface in
|
|
12
|
+
// `operations/index.ts` re-stated all of these constraints by hand, as flat Zod
|
|
13
|
+
// enums plus English prose, with no link of any kind back to the registry. It
|
|
14
|
+
// had drifted on every axis: an aspect-ratio enum offering `9:21` (a value that
|
|
15
|
+
// exists in NO model, invented here), "Kling/Seedance support all" when Seedance
|
|
16
|
+
// takes 6 of 11 and Kling-on-fal takes 3, "Veo locks to 16:9" when Veo on fal
|
|
17
|
+
// takes two, "Kling: 5-15" against a registry minimum of 3, and a
|
|
18
|
+
// `videoResolution` description that never mentioned Kling at all. A customer
|
|
19
|
+
// burned a round trip on 2026-08-16 passing `4:5` to Seedance: the op accepted
|
|
20
|
+
// it, the job queued, credits reserved, and the provider rejected it
|
|
21
|
+
// ASYNCHRONOUSLY. Client-side accept of a server-side reject is the worst shape
|
|
22
|
+
// a constraint bug can take.
|
|
23
|
+
//
|
|
24
|
+
// The fix is NOT a fourth mirror plus a fifth lockstep checker — a checker only
|
|
25
|
+
// proves two hand-written copies agree, it does not remove the second copy, and
|
|
26
|
+
// the second copy is the defect. So the values live HERE, once, and everything
|
|
27
|
+
// downstream imports them.
|
|
28
|
+
//
|
|
29
|
+
// 🚨 NEVER HAND-TYPE A CAPABILITY FACT AN LLM WILL READ. Op descriptions,
|
|
30
|
+
// clarification messages and skill prose all derive from the `describe*`
|
|
31
|
+
// helpers below. If you find yourself typing "4-15s" or "16:9/9:16" into a
|
|
32
|
+
// string, you are re-creating the bug this file deleted.
|
|
33
|
+
//
|
|
34
|
+
// Direction of dependency matches the settled precedent for MODEL_FACTS: this
|
|
35
|
+
// package owns the doctrine, slate DERIVES from the published package at
|
|
36
|
+
// runtime (`slate/src/main/studio-agent/context.ts` already imports
|
|
37
|
+
// `@slatesvideo/shared`). Rates, credit costs and cost-key builders deliberately
|
|
38
|
+
// did NOT move — billing stays in slate with its existing checkers.
|
|
39
|
+
//
|
|
40
|
+
// ⚠️ LEAF MODULE — no imports, no Node built-ins. The desktop RENDERER reaches
|
|
41
|
+
// this through `@slatesvideo/shared/prompts`, so anything pulled in here has to
|
|
42
|
+
// bundle for the browser.
|
|
43
|
+
//
|
|
44
|
+
// Verification that a change here is a pure relocation: the slates-api checkers
|
|
45
|
+
// read `MODEL_REGISTRY` and assert it against `CREDIT_COSTS` —
|
|
46
|
+
// `composition-matrix-check.mjs` (12,608 combinations),
|
|
47
|
+
// `reference-caps-lockstep-check.mjs`, `pricing-consistency-check.mjs`. If a
|
|
48
|
+
// value moved, they go red.
|
|
49
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
50
|
+
/**
|
|
51
|
+
* The full ten, in display order. `9:21` was in the MCP op's enum and in NO
|
|
52
|
+
* model — it was invented downstream. Do not add a ratio here that no model
|
|
53
|
+
* declares; the op's enum is generated from the union of what models accept, so
|
|
54
|
+
* a phantom entry here becomes a phantom entry an agent can pass.
|
|
55
|
+
*/
|
|
56
|
+
export const ALL_ASPECT_RATIOS = [
|
|
57
|
+
'1:1', '16:9', '9:16', '4:3', '3:4', '3:2', '2:3', '5:4', '4:5', '21:9',
|
|
58
|
+
];
|
|
59
|
+
// ── Shared ratio sets ────────────────────────────────────────────────────────
|
|
60
|
+
// Named rather than inlined because several models share a set and a set is the
|
|
61
|
+
// thing that changes (a provider adds a ratio, every model on it gains it).
|
|
62
|
+
/** Google / non-restricted models: all ten. */
|
|
63
|
+
const FULL_ASPECT_RATIOS = ALL_ASPECT_RATIOS;
|
|
64
|
+
/** Kling's DIRECT API: eight — no `5:4`, no `4:5`. */
|
|
65
|
+
const KLING_DIRECT_ASPECT_RATIOS = [
|
|
66
|
+
'1:1', '16:9', '9:16', '4:3', '3:4', '3:2', '2:3', '21:9',
|
|
67
|
+
];
|
|
68
|
+
/** Kling carried on fal: three. This is the set the CREDITS route uses. */
|
|
69
|
+
const KLING_FAL_ASPECT_RATIOS = ['16:9', '9:16', '1:1'];
|
|
70
|
+
/** Veo carried on fal: two. The credits route again — Veo direct takes all ten. */
|
|
71
|
+
const VEO_FAL_ASPECT_RATIOS = ['16:9', '9:16'];
|
|
72
|
+
/** Gemini Omni Flash (fal schema, 16:9 default): two. */
|
|
73
|
+
const OMNI_FLASH_ASPECT_RATIOS = ['16:9', '9:16'];
|
|
74
|
+
/** Seedance (both seats, and the edit row): six — notably NO `4:5`. */
|
|
75
|
+
const SEEDANCE_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
|
|
76
|
+
/**
|
|
77
|
+
* MiniMax H3, both seats: six. Read off fal's live OpenAPI 2026-08-27 for
|
|
78
|
+
* `minimax/h3/text-to-video` and `minimax/h3-max/text-to-video` — identical
|
|
79
|
+
* enums. It happens to be the same six Seedance takes; kept as its OWN constant
|
|
80
|
+
* because a provider that adds a ratio adds it to ITS family, and sharing the
|
|
81
|
+
* Seedance constant would silently move H3 the next time ByteDance moves.
|
|
82
|
+
*
|
|
83
|
+
* Two endpoint quirks the registry deliberately does not model:
|
|
84
|
+
* · `image-to-video` has NO `aspect_ratio` param at all — the output follows
|
|
85
|
+
* the start frame. The handler simply omits it there.
|
|
86
|
+
* · `reference-to-video` adds an `adaptive` value on top of these six. We
|
|
87
|
+
* never send it: the composer always has an explicit ratio, and `adaptive`
|
|
88
|
+
* is not an AspectRatio in this vocabulary.
|
|
89
|
+
*/
|
|
90
|
+
const MINIMAX_H3_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
|
|
91
|
+
/**
|
|
92
|
+
* The provider every AGENT generation actually lands on for Kling and Veo.
|
|
93
|
+
*
|
|
94
|
+
* 🚨 THIS IS WHY `providerAspectRatios` MATTERS TO THE OP. MCP/CLI/Studio-Agent
|
|
95
|
+
* generations are credits-only (BYOK is retired on the agent surface), and the
|
|
96
|
+
* credits route carries Kling and Veo on fal: `slate/src/main/agent/routes.ts`
|
|
97
|
+
* never sends `klingProvider`, so `handlers/video.ts` defaults it to `'fal'`,
|
|
98
|
+
* and `generateVeoVideo`'s proxy arm builds a fal request
|
|
99
|
+
* (`buildFalVeoRequest`). So an agent gets Kling's THREE fal ratios and Veo's
|
|
100
|
+
* TWO — not the eight and ten those models take on their direct APIs. Validating
|
|
101
|
+
* against the direct sets would accept a ratio fal rejects, which is the exact
|
|
102
|
+
* failure this module exists to delete.
|
|
103
|
+
*/
|
|
104
|
+
export const AGENT_ROUTE_PROVIDER = 'fal';
|
|
105
|
+
// ── The data ─────────────────────────────────────────────────────────────────
|
|
106
|
+
//
|
|
107
|
+
// Moved VERBATIM from `MODEL_REGISTRY` in slate/src/shared/pricing.ts on
|
|
108
|
+
// 2026-08-16. A relocation, not a re-derivation — the three slates-api checkers
|
|
109
|
+
// prove it (see the header).
|
|
110
|
+
export const MODEL_CAPABILITIES = {
|
|
111
|
+
// ── Image models ───────────────────────────────────────────────────────────
|
|
112
|
+
'nano-banana-2': {
|
|
113
|
+
aspectRatios: FULL_ASPECT_RATIOS,
|
|
114
|
+
maxRefImages: 14,
|
|
115
|
+
},
|
|
116
|
+
'nano-banana-2-lite': {
|
|
117
|
+
aspectRatios: FULL_ASPECT_RATIOS,
|
|
118
|
+
maxRefImages: 4, // fal edit endpoint caps input images at 4
|
|
119
|
+
},
|
|
120
|
+
'nano-banana-pro': {
|
|
121
|
+
aspectRatios: FULL_ASPECT_RATIOS,
|
|
122
|
+
maxRefImages: 14,
|
|
123
|
+
},
|
|
124
|
+
'gpt-image-2': {
|
|
125
|
+
// FIVE, not ten. The op's flat enum offered eleven for every image model.
|
|
126
|
+
aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
|
|
127
|
+
maxRefImages: 10,
|
|
128
|
+
},
|
|
129
|
+
'flux-2-max': {
|
|
130
|
+
aspectRatios: FULL_ASPECT_RATIOS,
|
|
131
|
+
maxRefImages: 4,
|
|
132
|
+
},
|
|
133
|
+
'seedream-5-lite': {
|
|
134
|
+
aspectRatios: FULL_ASPECT_RATIOS,
|
|
135
|
+
maxRefImages: 10,
|
|
136
|
+
},
|
|
137
|
+
// ── Kling video ────────────────────────────────────────────────────────────
|
|
138
|
+
'kling-v3.0-std': {
|
|
139
|
+
aspectRatios: KLING_DIRECT_ASPECT_RATIOS,
|
|
140
|
+
providerAspectRatios: { fal: KLING_FAL_ASPECT_RATIOS },
|
|
141
|
+
videoResolution: { options: ['1080p', '4k'] },
|
|
142
|
+
// 3, not 5. The op claimed "Kling: 5-15" and refused legal 3-4s takes.
|
|
143
|
+
duration: { min: 3, max: 15, mode: 'continuous' },
|
|
144
|
+
maxIngredientImages: 4,
|
|
145
|
+
},
|
|
146
|
+
'kling-v3.0-pro': {
|
|
147
|
+
aspectRatios: KLING_DIRECT_ASPECT_RATIOS,
|
|
148
|
+
providerAspectRatios: { fal: KLING_FAL_ASPECT_RATIOS },
|
|
149
|
+
videoResolution: { options: ['1080p', '4k'] },
|
|
150
|
+
duration: { min: 3, max: 15, mode: 'continuous' },
|
|
151
|
+
maxIngredientImages: 4,
|
|
152
|
+
},
|
|
153
|
+
'kling-v3.0-omni': {
|
|
154
|
+
aspectRatios: KLING_DIRECT_ASPECT_RATIOS,
|
|
155
|
+
providerAspectRatios: { fal: KLING_FAL_ASPECT_RATIOS },
|
|
156
|
+
videoResolution: { options: ['1080p', '4k'] },
|
|
157
|
+
duration: { min: 3, max: 15, mode: 'continuous' },
|
|
158
|
+
maxIngredientImages: 4,
|
|
159
|
+
},
|
|
160
|
+
'kling-v3.0-omni-pro': {
|
|
161
|
+
aspectRatios: KLING_DIRECT_ASPECT_RATIOS,
|
|
162
|
+
providerAspectRatios: { fal: KLING_FAL_ASPECT_RATIOS },
|
|
163
|
+
videoResolution: { options: ['1080p', '4k'] },
|
|
164
|
+
duration: { min: 3, max: 15, mode: 'continuous' },
|
|
165
|
+
maxIngredientImages: 4,
|
|
166
|
+
},
|
|
167
|
+
// Kling O3 video-to-video edit (fal-only; the source clip is the canvas, so
|
|
168
|
+
// aspect/resolution/duration all follow it).
|
|
169
|
+
'kling-v3.0-omni-edit': {
|
|
170
|
+
aspectRatios: KLING_FAL_ASPECT_RATIOS,
|
|
171
|
+
videoResolution: { options: ['1080p'], fixed: '1080p' },
|
|
172
|
+
duration: { min: 3, max: 15, mode: 'continuous' },
|
|
173
|
+
maxIngredientImages: 4, // elements + style refs combined (fal cap)
|
|
174
|
+
},
|
|
175
|
+
'kling-v3.0-omni-pro-edit': {
|
|
176
|
+
aspectRatios: KLING_FAL_ASPECT_RATIOS,
|
|
177
|
+
videoResolution: { options: ['1080p'], fixed: '1080p' },
|
|
178
|
+
duration: { min: 3, max: 15, mode: 'continuous' },
|
|
179
|
+
maxIngredientImages: 4,
|
|
180
|
+
},
|
|
181
|
+
// ── Gemini Omni Flash ──────────────────────────────────────────────────────
|
|
182
|
+
'omni-flash': {
|
|
183
|
+
aspectRatios: OMNI_FLASH_ASPECT_RATIOS,
|
|
184
|
+
videoResolution: { options: ['720p'], fixed: '720p' },
|
|
185
|
+
duration: { min: 3, max: 10, mode: 'continuous' },
|
|
186
|
+
maxIngredientImages: 7,
|
|
187
|
+
},
|
|
188
|
+
'omni-flash-edit': {
|
|
189
|
+
aspectRatios: OMNI_FLASH_ASPECT_RATIOS,
|
|
190
|
+
videoResolution: { options: ['720p'], fixed: '720p' },
|
|
191
|
+
duration: { min: 3, max: 10, mode: 'continuous' },
|
|
192
|
+
maxIngredientImages: 0,
|
|
193
|
+
},
|
|
194
|
+
// ── Veo ────────────────────────────────────────────────────────────────────
|
|
195
|
+
'veo-3.1-fast': {
|
|
196
|
+
aspectRatios: FULL_ASPECT_RATIOS,
|
|
197
|
+
providerAspectRatios: { fal: VEO_FAL_ASPECT_RATIOS },
|
|
198
|
+
videoResolution: { options: ['720p', '1080p', '4k'] },
|
|
199
|
+
duration: {
|
|
200
|
+
min: 4, max: 8, mode: 'discrete',
|
|
201
|
+
values: [4, 6, 8],
|
|
202
|
+
// BOTH 1080p and 4k force 8s. The op said "4K only at 8s" and quoted 4s
|
|
203
|
+
// at 1080p, which the provider rejects.
|
|
204
|
+
resolutionOverrides: {
|
|
205
|
+
'1080p': { min: 8, max: 8, mode: 'discrete', values: [8] },
|
|
206
|
+
'4k': { min: 8, max: 8, mode: 'discrete', values: [8] },
|
|
207
|
+
},
|
|
208
|
+
modeOverrides: {
|
|
209
|
+
ingredients: { min: 8, max: 8, mode: 'discrete', values: [8] },
|
|
210
|
+
},
|
|
211
|
+
},
|
|
212
|
+
maxIngredientImages: 3,
|
|
213
|
+
},
|
|
214
|
+
'veo-3.1-standard': {
|
|
215
|
+
aspectRatios: FULL_ASPECT_RATIOS,
|
|
216
|
+
providerAspectRatios: { fal: VEO_FAL_ASPECT_RATIOS },
|
|
217
|
+
videoResolution: { options: ['720p', '1080p', '4k'] },
|
|
218
|
+
duration: {
|
|
219
|
+
min: 4, max: 8, mode: 'discrete',
|
|
220
|
+
values: [4, 6, 8],
|
|
221
|
+
resolutionOverrides: {
|
|
222
|
+
'1080p': { min: 8, max: 8, mode: 'discrete', values: [8] },
|
|
223
|
+
'4k': { min: 8, max: 8, mode: 'discrete', values: [8] },
|
|
224
|
+
},
|
|
225
|
+
modeOverrides: {
|
|
226
|
+
ingredients: { min: 8, max: 8, mode: 'discrete', values: [8] },
|
|
227
|
+
},
|
|
228
|
+
},
|
|
229
|
+
maxIngredientImages: 3,
|
|
230
|
+
},
|
|
231
|
+
// ── Seedance ───────────────────────────────────────────────────────────────
|
|
232
|
+
'seedance-2': {
|
|
233
|
+
aspectRatios: SEEDANCE_ASPECT_RATIOS,
|
|
234
|
+
videoResolution: { options: ['480p', '720p', '1080p', '4k'], default: '1080p' },
|
|
235
|
+
duration: { min: 4, max: 15, mode: 'continuous' },
|
|
236
|
+
maxIngredientImages: 9,
|
|
237
|
+
maxReferenceVideos: 3,
|
|
238
|
+
maxReferenceAudio: 3,
|
|
239
|
+
// 12, and it is the FAL ceiling — not a BytePlus one. SETTLED 2026-08-10
|
|
240
|
+
// against both providers' primary sources; do not "correct" it to 15.
|
|
241
|
+
//
|
|
242
|
+
// BytePlus ModelArk (first party, docs → Multimodal reference):
|
|
243
|
+
// "You can combine the following modal content as needed…
|
|
244
|
+
// Images: 0–9 images · Videos: 0–3 videos · Audio: 0–3 audios"
|
|
245
|
+
// Per-arm ranges, combined AS NEEDED. No total is stated anywhere, so
|
|
246
|
+
// on BytePlus the effective maximum really is 9+3+3 = 15.
|
|
247
|
+
// fal live OpenAPI (bytedance/seedance-2.0/reference-to-video):
|
|
248
|
+
// same per-arm maxItems 9/3/3, PLUS an explicit
|
|
249
|
+
// "Total files across all modalities must not exceed 12."
|
|
250
|
+
// EvoLink (the third route): publishes NO numeric reference limits at
|
|
251
|
+
// all — checked 2026-08-10. Genuinely unknown, not assumed to be 15.
|
|
252
|
+
//
|
|
253
|
+
// So the providers that DO state a total disagree, and 12 binds because
|
|
254
|
+
// **a single generation can change providers after the user has approved
|
|
255
|
+
// it**: the real-face consent cascade resubmits an EvoLink rejection to fal
|
|
256
|
+
// mid-flight. A 15-file composition would be quoted, accepted, rejected by
|
|
257
|
+
// ByteDance's real-person classifier, re-quoted through the consent
|
|
258
|
+
// interstitial, and only THEN refused by fal for a reason the user was
|
|
259
|
+
// never shown. 12 is the minimum of the two documented ceilings, with the
|
|
260
|
+
// third unknown — so it is a floor on what is safe, not a proven optimum.
|
|
261
|
+
//
|
|
262
|
+
// The older "the modalities trade against each other" reading was a guess
|
|
263
|
+
// at why fal states 12; it is not what BytePlus documents. The "15" in the
|
|
264
|
+
// 2.5 plan's capability table was a SUM, not a figure anyone read.
|
|
265
|
+
// (2.5 is unaffected: fal states 50 and 30+10+10 = 50, so both agree.)
|
|
266
|
+
maxReferenceFilesTotal: 12,
|
|
267
|
+
maxReferenceVideoSeconds: 15,
|
|
268
|
+
maxReferenceAudioSeconds: 15,
|
|
269
|
+
},
|
|
270
|
+
'seedance-2.5': {
|
|
271
|
+
aspectRatios: SEEDANCE_ASPECT_RATIOS,
|
|
272
|
+
// 1080p landed 2026-08-24 on ALL THREE rails — BytePlus and EvoLink publish
|
|
273
|
+
// 1080p rate rows and fal's live OpenAPI enum reads
|
|
274
|
+
// ['480p','720p','1080p']. There is still NO 4K on 2.5 (2.0 is the only
|
|
275
|
+
// Seedance with one), which is what keeps `is4kVideoKey` version-blind.
|
|
276
|
+
//
|
|
277
|
+
// DEFAULT STAYS 720p, deliberately: a 30s take at 1080p is ~614 credits
|
|
278
|
+
// against a 1,000-credit welcome grant, and that is at the promotional
|
|
279
|
+
// 1080p rate — it rises when the promo lapses. Reaching a tier and
|
|
280
|
+
// defaulting to it are different decisions.
|
|
281
|
+
videoResolution: { options: ['480p', '720p', '1080p'], default: '720p' },
|
|
282
|
+
duration: { min: 4, max: 30, mode: 'continuous' },
|
|
283
|
+
maxIngredientImages: 30,
|
|
284
|
+
maxReferenceVideos: 10,
|
|
285
|
+
maxReferenceAudio: 10,
|
|
286
|
+
// fal states 50 and 30+10+10 = 50, so both documented ceilings agree here.
|
|
287
|
+
maxReferenceFilesTotal: 50,
|
|
288
|
+
maxReferenceVideoSeconds: 30,
|
|
289
|
+
maxReferenceAudioSeconds: 30,
|
|
290
|
+
},
|
|
291
|
+
'seedance-2.5-edit': {
|
|
292
|
+
aspectRatios: SEEDANCE_ASPECT_RATIOS,
|
|
293
|
+
// Same ladder as the generation row (1080p added 2026-08-24). EvoLink's
|
|
294
|
+
// rate card carries 1080p on the edit/extend row and BytePlus's video-input
|
|
295
|
+
// column runs the full tier list; an edit bills that tier × 2.
|
|
296
|
+
videoResolution: { options: ['480p', '720p', '1080p'], default: '720p' },
|
|
297
|
+
duration: { min: 4, max: 30, mode: 'continuous' },
|
|
298
|
+
// 🚨 ZERO, AND IT MUST MATCH WHAT THE HANDLER SENDS. The model's edit task
|
|
299
|
+
// type does accept reference images, but slate's
|
|
300
|
+
// `generation/handlers/edit-video.ts` sends the prompt and the source clip
|
|
301
|
+
// and NOTHING ELSE on this row — no `image_urls` on the EvoLink call, no
|
|
302
|
+
// `image_url` items in the BytePlus content array. This declared 30 while
|
|
303
|
+
// the handler sent 0, so attaching references produced no error, no
|
|
304
|
+
// warning, and no images in the request: a silent drop, which is the one
|
|
305
|
+
// outcome `validateComposition` exists to prevent. Both sibling edit rows
|
|
306
|
+
// already model this correctly (Omni Flash Edit is 0 and warns "takes the
|
|
307
|
+
// prompt + source clip only"; Kling O3 Edit is 4 and actually sends them).
|
|
308
|
+
//
|
|
309
|
+
// Raising it is a HANDLER change first: wire the refs, then move the cap.
|
|
310
|
+
maxIngredientImages: 0,
|
|
311
|
+
// NO multimodal reference caps, deliberately: on an edit row the clip IS the
|
|
312
|
+
// canvas and arrives through `sourceVideo`, not as a reference.
|
|
313
|
+
},
|
|
314
|
+
// ── MiniMax H3 (both seats on fal — added 2026-08-27) ──────────────────────
|
|
315
|
+
//
|
|
316
|
+
// Every value below is READ OFF fal's live OpenAPI, fetched 2026-08-27:
|
|
317
|
+
// minimax/h3/{text-to-video,image-to-video,reference-to-video}
|
|
318
|
+
// minimax/h3-max/{text-to-video,image-to-video}
|
|
319
|
+
// `minimax/h3-max/reference-to-video` returns 404 — it does not exist, which
|
|
320
|
+
// is why the Max row declares no reference capacity at all.
|
|
321
|
+
//
|
|
322
|
+
// 🚨 NEVER PREFIX-MATCH THESE TWO IDS. `minimax-h3-max` starts with
|
|
323
|
+
// `minimax-h3`, so any `startsWith('minimax-h3')` swallows the Max row into
|
|
324
|
+
// the base row's branch — a different ladder AND a different price at the one
|
|
325
|
+
// tier they share. Every lookup downstream is an exact-id map, not a prefix.
|
|
326
|
+
'minimax-h3': {
|
|
327
|
+
aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
|
|
328
|
+
// The full ladder. 480p/768p are NATIVE generation modes; 2K and 4K upscale
|
|
329
|
+
// a 768p base result through H3-Regenerate-2K, which is API-only and not in
|
|
330
|
+
// the open weights — that is why fal can undercut list at the bottom two
|
|
331
|
+
// tiers and matches it exactly at the top two.
|
|
332
|
+
//
|
|
333
|
+
// DEFAULT 768p, NOT fal's own default of 2K. 768p is the tier the model was
|
|
334
|
+
// trained to output and the one every benchmark quotes; 2K is a 2.2x price
|
|
335
|
+
// step and 4K a 2.7x step, and reaching a tier is a different decision from
|
|
336
|
+
// defaulting to it (same reasoning that keeps Seedance 2.5 on 720p).
|
|
337
|
+
videoResolution: { options: ['480p', '768p', '2k', '4k'], default: '768p' },
|
|
338
|
+
// 5, not 4. MiniMax's own model card says 4-15s; fal's schema — which is
|
|
339
|
+
// what our request actually hits — says `minimum: 5`. The endpoint wins.
|
|
340
|
+
duration: { min: 5, max: 15, mode: 'continuous' },
|
|
341
|
+
// Ref2VA omni-reference caps, verbatim from the reference-to-video schema:
|
|
342
|
+
// reference_image_urls maxItems 9, reference_video_urls maxItems 3,
|
|
343
|
+
// reference_audio_urls maxItems 3, and in every one of the three
|
|
344
|
+
// descriptions: "Reference images, videos, and audio clips must add up to
|
|
345
|
+
// at most 12 files."
|
|
346
|
+
maxIngredientImages: 9,
|
|
347
|
+
maxReferenceVideos: 3,
|
|
348
|
+
maxReferenceAudio: 3,
|
|
349
|
+
maxReferenceFilesTotal: 12,
|
|
350
|
+
// COMBINED, not per clip. fal states "2-15 seconds each, combined duration
|
|
351
|
+
// at most 15 seconds" for both media arms — so the per-clip floor of 2s is
|
|
352
|
+
// the shared reference-video minimum already enforced by the composer, and
|
|
353
|
+
// 15 is the sum these fields have always meant.
|
|
354
|
+
maxReferenceVideoSeconds: 15,
|
|
355
|
+
maxReferenceAudioSeconds: 15,
|
|
356
|
+
},
|
|
357
|
+
'minimax-h3-max': {
|
|
358
|
+
aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
|
|
359
|
+
// 480p/768p ONLY — fal's post-train of the open weights, and the 2K
|
|
360
|
+
// upscaler was never open-sourced. Declaring the shorter ladder here IS the
|
|
361
|
+
// whole Max-seat mechanism: `assertVideoCapabilities` refuses 2K/4K on this
|
|
362
|
+
// id, the desktop picker renders only what this entry declares, and the
|
|
363
|
+
// agent's Zod enum stays the union while the per-model guard narrows.
|
|
364
|
+
// Anything shaped like "disable the higher tiers when Max is selected" is
|
|
365
|
+
// re-implementing a guard that already exists.
|
|
366
|
+
videoResolution: { options: ['480p', '768p'], default: '768p' },
|
|
367
|
+
duration: { min: 5, max: 15, mode: 'continuous' },
|
|
368
|
+
// NO reference caps, deliberately: fal publishes text-to-video and
|
|
369
|
+
// image-to-video for h3-max and NOTHING else (reference-to-video 404s), so
|
|
370
|
+
// there is no transport for a reference of any modality. A cap declared
|
|
371
|
+
// above what the handler sends is a SILENT DROP — the exact failure
|
|
372
|
+
// `seedance-2.5-edit` shipped with. Absent means the composer refuses.
|
|
373
|
+
},
|
|
374
|
+
// ── Audio ──────────────────────────────────────────────────────────────────
|
|
375
|
+
//
|
|
376
|
+
// `aspectRatios: []` is deliberate, not an oversight: audio has no frame, and
|
|
377
|
+
// an empty list is what makes the desktop composer HIDE the ratio control
|
|
378
|
+
// instead of offering a meaningless one. Duration for these two lives in
|
|
379
|
+
// `MODEL_REGISTRY.audio.durationSeconds` alongside the billing bounds, which
|
|
380
|
+
// are mirrored in three repos and locked by `pricing-consistency-check.mjs` §4
|
|
381
|
+
// — moving them here would split one clamp across two files.
|
|
382
|
+
'seed-audio': {
|
|
383
|
+
aspectRatios: [],
|
|
384
|
+
// ONE image XOR up to 3 audio clips; the XOR is enforced by
|
|
385
|
+
// `validateComposition`, this is only the image arm.
|
|
386
|
+
maxRefImages: 1,
|
|
387
|
+
},
|
|
388
|
+
'eleven-sfx': {
|
|
389
|
+
aspectRatios: [],
|
|
390
|
+
},
|
|
391
|
+
};
|
|
392
|
+
// ── Queries ──────────────────────────────────────────────────────────────────
|
|
393
|
+
export function getModelCapability(model) {
|
|
394
|
+
return MODEL_CAPABILITIES[model];
|
|
395
|
+
}
|
|
396
|
+
/** Aspect ratios a model accepts, honouring the provider override. */
|
|
397
|
+
export function aspectRatiosFor(model, provider) {
|
|
398
|
+
const cap = MODEL_CAPABILITIES[model];
|
|
399
|
+
if (!cap)
|
|
400
|
+
return ALL_ASPECT_RATIOS;
|
|
401
|
+
if (provider && cap.providerAspectRatios?.[provider])
|
|
402
|
+
return cap.providerAspectRatios[provider];
|
|
403
|
+
return cap.aspectRatios;
|
|
404
|
+
}
|
|
405
|
+
/** Video resolutions a model accepts. A FIXED model reports exactly its one value. */
|
|
406
|
+
export function videoResolutionsFor(model) {
|
|
407
|
+
const vr = MODEL_CAPABILITIES[model]?.videoResolution;
|
|
408
|
+
if (!vr)
|
|
409
|
+
return [];
|
|
410
|
+
return vr.fixed ? [vr.fixed] : vr.options;
|
|
411
|
+
}
|
|
412
|
+
/** The resolution a model would actually run at. Fixed wins; else keep a legal
|
|
413
|
+
* current value; else the model's own default. Mirrors `clampVideoResolution`. */
|
|
414
|
+
export function defaultVideoResolutionFor(model) {
|
|
415
|
+
const vr = MODEL_CAPABILITIES[model]?.videoResolution;
|
|
416
|
+
if (!vr)
|
|
417
|
+
return undefined;
|
|
418
|
+
return vr.fixed ?? vr.default ?? vr.options[0];
|
|
419
|
+
}
|
|
420
|
+
/**
|
|
421
|
+
* Duration constraints after applying overrides.
|
|
422
|
+
*
|
|
423
|
+
* ⚠️ ORDER IS LOAD-BEARING and mirrors `getAvailableDurations` in
|
|
424
|
+
* slate/src/shared/pricing.ts EXACTLY: mode override first (more specific),
|
|
425
|
+
* resolution override only if no mode override applied. Reversing them would
|
|
426
|
+
* make the desktop and the agent disagree about the same generation.
|
|
427
|
+
*/
|
|
428
|
+
export function durationsFor(model, opts = {}) {
|
|
429
|
+
const base = MODEL_CAPABILITIES[model]?.duration;
|
|
430
|
+
if (!base)
|
|
431
|
+
return undefined;
|
|
432
|
+
if (opts.promptMode && base.modeOverrides?.[opts.promptMode]) {
|
|
433
|
+
return { ...base, ...base.modeOverrides[opts.promptMode] };
|
|
434
|
+
}
|
|
435
|
+
if (opts.videoResolution && base.resolutionOverrides?.[opts.videoResolution]) {
|
|
436
|
+
return { ...base, ...base.resolutionOverrides[opts.videoResolution] };
|
|
437
|
+
}
|
|
438
|
+
return base;
|
|
439
|
+
}
|
|
440
|
+
/** Every legal whole-second duration. Mirrors `getAvailableDurations`. */
|
|
441
|
+
export function durationValuesFor(model, opts = {}) {
|
|
442
|
+
const d = durationsFor(model, opts);
|
|
443
|
+
if (!d)
|
|
444
|
+
return [];
|
|
445
|
+
if (d.mode === 'discrete' && d.values)
|
|
446
|
+
return d.values;
|
|
447
|
+
const out = [];
|
|
448
|
+
for (let i = d.min; i <= d.max; i++)
|
|
449
|
+
out.push(i);
|
|
450
|
+
return out;
|
|
451
|
+
}
|
|
452
|
+
/** Union of every ratio the given models accept — the legal universe for an enum. */
|
|
453
|
+
export function aspectRatioUnion(models, provider) {
|
|
454
|
+
const seen = new Set();
|
|
455
|
+
for (const m of models)
|
|
456
|
+
for (const r of aspectRatiosFor(m, provider))
|
|
457
|
+
seen.add(r);
|
|
458
|
+
// Emit in ALL_ASPECT_RATIOS order so the enum is stable regardless of input order.
|
|
459
|
+
return ALL_ASPECT_RATIOS.filter((r) => seen.has(r));
|
|
460
|
+
}
|
|
461
|
+
/** Union of every resolution the given models accept. */
|
|
462
|
+
export function videoResolutionUnion(models) {
|
|
463
|
+
// Ascending by output height, so an enum reads as a ladder. 768p sits between
|
|
464
|
+
// 720p and 1080p; 2k (≈2560×1440) between 1080p and 4k.
|
|
465
|
+
const order = ['480p', '720p', '768p', '1080p', '2k', '4k'];
|
|
466
|
+
const seen = new Set();
|
|
467
|
+
for (const m of models)
|
|
468
|
+
for (const r of videoResolutionsFor(m))
|
|
469
|
+
seen.add(r);
|
|
470
|
+
return order.filter((r) => seen.has(r));
|
|
471
|
+
}
|
|
472
|
+
/** Widest legal duration window across the given models, overrides included. */
|
|
473
|
+
export function durationBounds(models) {
|
|
474
|
+
let min = Infinity;
|
|
475
|
+
let max = -Infinity;
|
|
476
|
+
for (const m of models) {
|
|
477
|
+
for (const v of durationValuesFor(m)) {
|
|
478
|
+
if (v < min)
|
|
479
|
+
min = v;
|
|
480
|
+
if (v > max)
|
|
481
|
+
max = v;
|
|
482
|
+
}
|
|
483
|
+
// Overrides can only narrow, never widen — but read them anyway so a future
|
|
484
|
+
// widening override cannot silently fall outside the enum's bounds.
|
|
485
|
+
const base = MODEL_CAPABILITIES[m]?.duration;
|
|
486
|
+
for (const o of [
|
|
487
|
+
...Object.values(base?.resolutionOverrides ?? {}),
|
|
488
|
+
...Object.values(base?.modeOverrides ?? {}),
|
|
489
|
+
]) {
|
|
490
|
+
if (o.min < min)
|
|
491
|
+
min = o.min;
|
|
492
|
+
if (o.max > max)
|
|
493
|
+
max = o.max;
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
return Number.isFinite(min) ? { min, max } : { min: 0, max: 0 };
|
|
497
|
+
}
|
|
498
|
+
// ── Validation ───────────────────────────────────────────────────────────────
|
|
499
|
+
//
|
|
500
|
+
// Each returns an ACTIONABLE message naming the legal set, or null when the
|
|
501
|
+
// value is fine. The message is generated, so it can never name a set the data
|
|
502
|
+
// does not contain.
|
|
503
|
+
export function checkAspectRatio(model, aspectRatio, provider) {
|
|
504
|
+
if (!aspectRatio)
|
|
505
|
+
return null;
|
|
506
|
+
const legal = aspectRatiosFor(model, provider);
|
|
507
|
+
if (legal.length === 0 || legal.includes(aspectRatio))
|
|
508
|
+
return null;
|
|
509
|
+
return `${model} does not accept aspectRatio "${aspectRatio}". It accepts ${legal.join(', ')}. Pick one of those, or switch to a model that takes the shape you want.`;
|
|
510
|
+
}
|
|
511
|
+
export function checkVideoResolution(model, videoResolution) {
|
|
512
|
+
if (!videoResolution)
|
|
513
|
+
return null;
|
|
514
|
+
const legal = videoResolutionsFor(model);
|
|
515
|
+
if (legal.length === 0)
|
|
516
|
+
return null;
|
|
517
|
+
if (legal.includes(videoResolution))
|
|
518
|
+
return null;
|
|
519
|
+
const vr = MODEL_CAPABILITIES[model]?.videoResolution;
|
|
520
|
+
if (vr?.fixed) {
|
|
521
|
+
return `${model} renders at ${vr.fixed} only — it has no resolution parameter, so videoResolution "${videoResolution}" cannot apply. Drop the param.`;
|
|
522
|
+
}
|
|
523
|
+
return `${model} does not render at ${videoResolution}. It offers ${legal.join(', ')}. Pick one of those, or switch models.`;
|
|
524
|
+
}
|
|
525
|
+
export function checkDuration(model, duration, opts = {}) {
|
|
526
|
+
if (duration == null)
|
|
527
|
+
return null;
|
|
528
|
+
const d = durationsFor(model, opts);
|
|
529
|
+
if (!d)
|
|
530
|
+
return null;
|
|
531
|
+
const legal = durationValuesFor(model, opts);
|
|
532
|
+
if (legal.includes(duration))
|
|
533
|
+
return null;
|
|
534
|
+
// Name WHY the window narrowed, and how to widen it again — otherwise the
|
|
535
|
+
// message reads as a contradiction of the model's own advertised range.
|
|
536
|
+
const base = MODEL_CAPABILITIES[model]?.duration;
|
|
537
|
+
let why = '';
|
|
538
|
+
let escape = '';
|
|
539
|
+
if (opts.promptMode && base?.modeOverrides?.[opts.promptMode]) {
|
|
540
|
+
why = ' with reference images attached';
|
|
541
|
+
escape = `, or drop the reference images to get back to ${fmtWindow(base)}`;
|
|
542
|
+
}
|
|
543
|
+
else if (opts.videoResolution && base?.resolutionOverrides?.[opts.videoResolution]) {
|
|
544
|
+
why = ` at ${opts.videoResolution}`;
|
|
545
|
+
escape = `, or pick a resolution without that restriction (${fmtWindow(base)} at the unrestricted ones)`;
|
|
546
|
+
}
|
|
547
|
+
const allowed = legal.length === 1 ? `${legal[0]}s only` : d.mode === 'discrete' ? `${legal.join('s, ')}s` : `${d.min}-${d.max}s`;
|
|
548
|
+
return `${model}${why} accepts ${allowed} — ${duration}s is not legal. Pick a duration in range${escape}.`;
|
|
549
|
+
}
|
|
550
|
+
// ── Generated prose ──────────────────────────────────────────────────────────
|
|
551
|
+
//
|
|
552
|
+
// Every `.describe()` string the LLM reads about these three params is built
|
|
553
|
+
// here, so prose CANNOT contradict the data. Grouping models that share a value
|
|
554
|
+
// keeps the desktop's cached token prefix small.
|
|
555
|
+
function groupBy(models, fn) {
|
|
556
|
+
const groups = [];
|
|
557
|
+
for (const m of models) {
|
|
558
|
+
const value = fn(m);
|
|
559
|
+
if (!value)
|
|
560
|
+
continue;
|
|
561
|
+
const existing = groups.find((g) => g.value === value);
|
|
562
|
+
if (existing)
|
|
563
|
+
existing.models.push(m);
|
|
564
|
+
else
|
|
565
|
+
groups.push({ value, models: [m] });
|
|
566
|
+
}
|
|
567
|
+
return groups.map((g) => `${g.models.join('/')}: ${g.value}`).join(' · ');
|
|
568
|
+
}
|
|
569
|
+
/** e.g. "kling-v3.0-std/kling-v3.0-pro: 16:9, 9:16, 1:1 · seedance-2: 21:9, …" */
|
|
570
|
+
export function describeAspectRatios(models, provider) {
|
|
571
|
+
return groupBy(models, (m) => aspectRatiosFor(m, provider).join(', '));
|
|
572
|
+
}
|
|
573
|
+
/** e.g. "seedance-2: 480p, 720p, 1080p, 4k (default 1080p) · omni-flash: 720p only (fixed)" */
|
|
574
|
+
export function describeVideoResolutions(models) {
|
|
575
|
+
return groupBy(models, (m) => {
|
|
576
|
+
const vr = MODEL_CAPABILITIES[m]?.videoResolution;
|
|
577
|
+
if (!vr)
|
|
578
|
+
return '';
|
|
579
|
+
if (vr.fixed)
|
|
580
|
+
return `${vr.fixed} only (fixed — do not pass videoResolution)`;
|
|
581
|
+
const def = vr.default ?? vr.options[0];
|
|
582
|
+
return `${vr.options.join(', ')} (default ${def})`;
|
|
583
|
+
});
|
|
584
|
+
}
|
|
585
|
+
function fmtWindow(d) {
|
|
586
|
+
return d.mode === 'discrete' && d.values ? `${d.values.join('s/')}s` : `${d.min}-${d.max}s`;
|
|
587
|
+
}
|
|
588
|
+
/** e.g. "kling-v3.0-std: 3-15s · veo-3.1-fast: 4s/6s/8s (1080p/4k: 8s only; with reference images: 8s only)" */
|
|
589
|
+
export function describeDurations(models) {
|
|
590
|
+
return groupBy(models, (m) => {
|
|
591
|
+
const d = MODEL_CAPABILITIES[m]?.duration;
|
|
592
|
+
if (!d)
|
|
593
|
+
return '';
|
|
594
|
+
const clauses = [];
|
|
595
|
+
const resGroups = [];
|
|
596
|
+
for (const [res, o] of Object.entries(d.resolutionOverrides ?? {})) {
|
|
597
|
+
const value = fmtWindow(o);
|
|
598
|
+
const hit = resGroups.find((g) => g.value === value);
|
|
599
|
+
if (hit)
|
|
600
|
+
hit.keys.push(res);
|
|
601
|
+
else
|
|
602
|
+
resGroups.push({ value, keys: [res] });
|
|
603
|
+
}
|
|
604
|
+
for (const g of resGroups)
|
|
605
|
+
clauses.push(`${g.keys.join('/')}: ${g.value} only`);
|
|
606
|
+
if (d.modeOverrides?.ingredients) {
|
|
607
|
+
clauses.push(`with reference images: ${fmtWindow(d.modeOverrides.ingredients)} only`);
|
|
608
|
+
}
|
|
609
|
+
return `${fmtWindow(d)}${clauses.length ? ` (${clauses.join('; ')})` : ''}`;
|
|
610
|
+
});
|
|
611
|
+
}
|
|
612
|
+
/** e.g. "seedance-2: 9 · seedance-2.5: 30 · omni-flash: 7 · seedance-2.5-edit: 0 (prompt + source clip only)" */
|
|
613
|
+
export function describeReferenceImageCaps(models) {
|
|
614
|
+
return groupBy(models, (m) => {
|
|
615
|
+
const cap = MODEL_CAPABILITIES[m];
|
|
616
|
+
if (!cap)
|
|
617
|
+
return '';
|
|
618
|
+
const n = cap.maxIngredientImages ?? cap.maxRefImages;
|
|
619
|
+
if (n == null)
|
|
620
|
+
return '';
|
|
621
|
+
return n === 0 ? '0 (prompt + source clip only)' : String(n);
|
|
622
|
+
});
|
|
623
|
+
}
|
|
624
|
+
//# sourceMappingURL=model-capabilities.js.map
|
|
@@ -6,9 +6,9 @@ export interface ModelFact {
|
|
|
6
6
|
maxRefImages: number | null;
|
|
7
7
|
/** Max ingredient images (video models) — null if not applicable. */
|
|
8
8
|
maxIngredients: number | null;
|
|
9
|
-
/** Reference VIDEOS accepted in one generation. null
|
|
9
|
+
/** Reference VIDEOS accepted in one generation. null = none. */
|
|
10
10
|
maxReferenceVideos?: number | null;
|
|
11
|
-
/** Reference AUDIO clips accepted in one generation. null
|
|
11
|
+
/** Reference AUDIO clips accepted in one generation. null = none. */
|
|
12
12
|
maxReferenceAudio?: number | null;
|
|
13
13
|
/** Combined seconds across every reference video. */
|
|
14
14
|
maxReferenceVideoSeconds?: number | null;
|
|
@@ -16,7 +16,15 @@ export interface ModelFact {
|
|
|
16
16
|
maxReferenceAudioSeconds?: number | null;
|
|
17
17
|
/** Ceiling on TOTAL reference files across all modalities. */
|
|
18
18
|
maxReferenceFilesTotal?: number | null;
|
|
19
|
-
/**
|
|
19
|
+
/**
|
|
20
|
+
* An audio reference needs at least one image or video reference alongside.
|
|
21
|
+
*
|
|
22
|
+
* Still declared here rather than derived: it is a BEHAVIOURAL rule, not a
|
|
23
|
+
* cap, and its runtime home is the registry FEATURE flag
|
|
24
|
+
* `features.audioRefNeedsCompanion` (which gates composer affordances). Only
|
|
25
|
+
* Seedance 2.0 sets it; 2.5 allowing audio-only references is one of the
|
|
26
|
+
* things the second seat buys.
|
|
27
|
+
*/
|
|
20
28
|
audioRefNeedsCompanion?: boolean;
|
|
21
29
|
notes: string;
|
|
22
30
|
}
|