@slatesvideo/shared 0.5.9 → 0.5.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,9 +6,9 @@ export interface ModelFact {
6
6
  maxRefImages: number | null;
7
7
  /** Max ingredient images (video models) — null if not applicable. */
8
8
  maxIngredients: number | null;
9
- /** Reference VIDEOS accepted in one generation. null/absent = none. */
9
+ /** Reference VIDEOS accepted in one generation. null = none. */
10
10
  maxReferenceVideos?: number | null;
11
- /** Reference AUDIO clips accepted in one generation. null/absent = none. */
11
+ /** Reference AUDIO clips accepted in one generation. null = none. */
12
12
  maxReferenceAudio?: number | null;
13
13
  /** Combined seconds across every reference video. */
14
14
  maxReferenceVideoSeconds?: number | null;
@@ -16,7 +16,15 @@ export interface ModelFact {
16
16
  maxReferenceAudioSeconds?: number | null;
17
17
  /** Ceiling on TOTAL reference files across all modalities. */
18
18
  maxReferenceFilesTotal?: number | null;
19
- /** An audio reference needs at least one image or video reference alongside. */
19
+ /**
20
+ * An audio reference needs at least one image or video reference alongside.
21
+ *
22
+ * Still declared here rather than derived: it is a BEHAVIOURAL rule, not a
23
+ * cap, and its runtime home is the registry FEATURE flag
24
+ * `features.audioRefNeedsCompanion` (which gates composer affordances). Only
25
+ * Seedance 2.0 sets it; 2.5 allowing audio-only references is one of the
26
+ * things the second seat buys.
27
+ */
20
28
  audioRefNeedsCompanion?: boolean;
21
29
  notes: string;
22
30
  }
@@ -1,14 +1,48 @@
1
- // Per-model prompting facts — reference/ingredient limits and the prompt
2
- // formula, as KNOWLEDGE (for skills + lead-magnet + desktop tooltips). The
3
- // RUNTIME source of truth for limits is slate/src/shared/pricing.ts
4
- // (MODEL_REGISTRY.maxRefImages / maxIngredientImages); these mirror it for
5
- // documentation. Code-verified 2026-06-25.
1
+ // Per-model prompting facts — routing doctrine and the prompt formula, as
2
+ // KNOWLEDGE (for skills + lead-magnet + op descriptions).
3
+ //
4
+ // 🚨 THE REFERENCE CAPS ARE NO LONGER TYPED HERE (2026-08-16). They are DERIVED
5
+ // from `MODEL_CAPABILITIES` in ./model-capabilities.ts, which is now the single
6
+ // definition — the same module `slate/src/shared/pricing.ts` spreads into every
7
+ // MODEL_REGISTRY entry. This file used to hand-mirror those numbers "for
8
+ // documentation"; a hand-mirror in the same package as the source is exactly the
9
+ // defect the capability SSOT exists to delete, so `caps()` below does the lookup
10
+ // and a wrong id throws at module load instead of shipping a stale number.
6
11
  //
7
12
  // Prose that ALSO appears in a skill or the tips card comes from
8
13
  // skills/_partials/*.md via PARTIALS — never restated here. A `notes` string is
9
14
  // a third rendering of a fact, and a third rendering is a third thing that can
10
15
  // survive a doctrine reversal the other two got.
11
16
  import { PARTIALS } from './partials.generated.js';
17
+ import { MODEL_CAPABILITIES } from './model-capabilities.js';
18
+ /**
19
+ * Reference caps for a fact, read out of the capability SSOT.
20
+ *
21
+ * The argument is a `MODEL_CAPABILITIES` key, which is a REGISTRY model id — and
22
+ * three facts here are FAMILY-level (`kling-v3`, `kling-v3-edit`, `veo-3.1`)
23
+ * with no registry row of their own, so they name a representative variant.
24
+ * That is safe because the caps are identical across the variants in each
25
+ * family (std/pro/omni/omni-pro all take 4; both edit rows take 4; both Veo
26
+ * seats take 3) — and if a future variant diverges, this becomes a wrong number
27
+ * rather than a crash, so split the fact rather than picking a side.
28
+ *
29
+ * Throws on an unknown id: a typo must fail the build, not ship as `null` caps
30
+ * that quietly tell an agent a model reads no references.
31
+ */
32
+ function caps(capabilityId) {
33
+ const c = MODEL_CAPABILITIES[capabilityId];
34
+ if (!c)
35
+ throw new Error(`MODEL_FACTS: no MODEL_CAPABILITIES entry for "${capabilityId}"`);
36
+ return {
37
+ maxRefImages: c.maxRefImages ?? null,
38
+ maxIngredients: c.maxIngredientImages ?? null,
39
+ maxReferenceVideos: c.maxReferenceVideos ?? null,
40
+ maxReferenceAudio: c.maxReferenceAudio ?? null,
41
+ maxReferenceVideoSeconds: c.maxReferenceVideoSeconds ?? null,
42
+ maxReferenceAudioSeconds: c.maxReferenceAudioSeconds ?? null,
43
+ maxReferenceFilesTotal: c.maxReferenceFilesTotal ?? null,
44
+ };
45
+ }
12
46
  /**
13
47
  * One sentence of multimodal-reference capacity for a model, derived. Returns
14
48
  * an empty string for a model that takes none, so a caller can append it
@@ -79,61 +113,50 @@ export const MODEL_FACTS = [
79
113
  // (gemini-3-pro-image-preview); do not conflate them.
80
114
  label: 'Nano Banana 2 (Gemini 3.1 Flash Image)',
81
115
  kind: 'image',
82
- maxRefImages: 14, // 10 object-fidelity + 4 character-consistency; categories don't trade.
83
- maxIngredients: null,
116
+ // 14 = 10 object-fidelity + 4 character-consistency; the categories don't trade.
117
+ ...caps('nano-banana-2'),
84
118
  notes: 'Default image model. 14 refs hard cap (10 object + 4 character). Brief it like a creative director, not tag soup. No negativePrompt field — use positive reframing. Best image start-frame for legible text. Knowledge cutoff Jan 2025.',
85
119
  },
86
120
  {
87
121
  id: 'nano-banana-2-lite',
88
122
  label: 'Nano Banana 2 Lite',
89
123
  kind: 'image',
90
- maxRefImages: 4,
91
- maxIngredients: null,
124
+ ...caps('nano-banana-2-lite'),
92
125
  notes: 'FAST/DRAFT image tier — ~half the price of NB2 full, ~2.7× faster, 1K output ONLY. Same Gemini content filter as NB2. Route here for iteration volume and drafts where 1K is fine; keep NB2 full for final 2K/4K. Character consistency + legible text hold up.',
93
126
  },
94
127
  {
95
128
  id: 'nano-banana-pro',
96
129
  label: 'Nano Banana Pro',
97
130
  kind: 'image',
98
- maxRefImages: 14,
99
- maxIngredients: null,
131
+ ...caps('nano-banana-pro'),
100
132
  notes: 'HERO-FRAME / typography PREMIUM image tier (Gemini 3 Pro backbone; ~2× NB2 price). NB2 ≈ 95% of Pro — route here only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element reasoning must be perfect. Up to 14 reference images (character locking, multi-subject fusion). Native 16:9 + 4K.',
101
133
  },
102
134
  {
103
135
  id: 'gpt-image-2',
104
136
  label: 'GPT Image 2',
105
137
  kind: 'image',
106
- maxRefImages: 10,
107
- maxIngredients: null,
138
+ ...caps('gpt-image-2'),
108
139
  notes: 'TEXT/DIAGRAM/PANEL king — near-perfect character-level text, ordered panels, exact placement (~3s gens). Route here for character sheets, shot grids, and text-bearing panels. Quality tiers: medium (default, the value seat — half NB2 price at 1080p) / high (~4×, max text precision). Third filter regime (OpenAI moderate). 4K is API-only — even paid ChatGPT can\'t render it. Photoreal/character-locked/edit-heavy → Banana line instead.',
109
140
  },
110
141
  {
111
142
  id: 'flux-2-max',
112
143
  label: 'FLUX.2 Max',
113
144
  kind: 'image',
114
- maxRefImages: 4,
115
- maxIngredients: null,
145
+ ...caps('flux-2-max'),
116
146
  notes: 'Photoreal, less censored, up to ~4MP. Auto-routes to its edit endpoint when references are present. Lower ref cap than NB2.',
117
147
  },
118
148
  {
119
149
  id: 'seedream-5-lite',
120
150
  label: 'Seedream 5 Lite',
121
151
  kind: 'image',
122
- maxRefImages: 10,
123
- maxIngredients: null,
152
+ ...caps('seedream-5-lite'),
124
153
  notes: 'Cheapest image model (~flat price). Less censored. Routes to its edit endpoint with references.',
125
154
  },
126
155
  {
127
156
  id: 'seedance-2',
128
157
  label: 'Seedance 2.0',
129
158
  kind: 'video',
130
- maxRefImages: null,
131
- maxIngredients: 9, // ingredient images per video gen
132
- maxReferenceVideos: 3,
133
- maxReferenceAudio: 3,
134
- maxReferenceVideoSeconds: 15,
135
- maxReferenceAudioSeconds: 15,
136
- maxReferenceFilesTotal: 12,
159
+ ...caps('seedance-2'),
137
160
  audioRefNeedsCompanion: true,
138
161
  notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds. 2.0 STAYS THE DEFAULT over 2.5 because it is the only Seedance with 1080p and 4K.',
139
162
  },
@@ -141,13 +164,7 @@ export const MODEL_FACTS = [
141
164
  id: 'seedance-2.5',
142
165
  label: 'Seedance 2.5',
143
166
  kind: 'video',
144
- maxRefImages: null,
145
- maxIngredients: 30, // 30 image refs; the model also takes 10 video + 10 audio (50 total)
146
- maxReferenceVideos: 10,
147
- maxReferenceAudio: 10,
148
- maxReferenceVideoSeconds: 30,
149
- maxReferenceAudioSeconds: 30,
150
- maxReferenceFilesTotal: 50,
167
+ ...caps('seedance-2.5'),
151
168
  // No companion requirement — audio-only references are one of the things
152
169
  // the second seat actually buys.
153
170
  notes: `A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is 480p or 720p on every route Slates offers: no 1080p, no 4K. Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), TIMED BEATS, or tighter prompt adherence. Pick 2.0 when resolution matters at all. VIDEO-ONLY. TIMESTAMPS: ${PARTIALS['seedance-25-timestamps-short']} Multi-view subject reference images are also supported on 2.5 (up to 5 subjects) where 2.0 wanted one view per subject. 🚨 COST DISCIPLINE: 720p STOPS READING AS "THE CHEAP ONE" HERE. A 30s 720p clip on the real-face route is 710 credits and on the AI-face route 484 — more than a 15s 1080p Seedance 2.0 face generation (411), against a 1,000-credit welcome grant. Always quote with slates_estimate_generation_cost before a long take, and explore at SHORT LENGTH (4-8s) rather than at low resolution — a 480p pass does not de-risk a 720p render, because generation is stochastic and the 720p run is a different take, not the same shot rendered better. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.`,
@@ -156,64 +173,63 @@ export const MODEL_FACTS = [
156
173
  id: 'seedance-2.5-edit',
157
174
  label: 'Seedance 2.5 Edit',
158
175
  kind: 'video',
159
- maxRefImages: null,
160
- maxIngredients: 0, // prompt + source clip only on slates_edit_video
176
+ // 0 ingredients: prompt + source clip only on slates_edit_video.
177
+ ...caps('seedance-2.5-edit'),
161
178
  notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s), though ByteDance recommends staying inside 20s for quality. That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p output, native audio. Prompt + source clip only on this op — no reference images (the MODEL takes 1-5 reference images on an edit; Slates has not wired that path). Phrase the change as "from A to B", and TIMESTAMP a partial edit ("…from 4-6 seconds…") — 2.5 reads whole-second timestamps on edits, and without a range the instruction applies to the whole clip. AUDIO is editable on this same row: change a line, change an accent, translate dialogue with re-fitted lips, strip or replace BGM and sound effects. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
162
179
  },
163
180
  {
164
181
  id: 'kling-v3',
165
182
  label: 'Kling 3.0',
166
183
  kind: 'video',
167
- maxRefImages: null,
168
- maxIngredients: 4,
184
+ // Family-level fact — caps are identical across std/pro/omni/omni-pro.
185
+ ...caps('kling-v3.0-std'),
169
186
  notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools (MC std/pro, lip-sync, avatar) — those two tools are Kling-only.',
170
187
  },
171
188
  {
172
189
  id: 'kling-v3-edit',
173
190
  label: 'Kling O3 Video Edit',
174
191
  kind: 'video',
175
- maxRefImages: null,
176
- maxIngredients: 4, // combined subject elements + style refs per edit
192
+ // Family-level fact; 4 = combined subject elements + style refs per edit.
193
+ ...caps('kling-v3.0-omni-edit'),
177
194
  notes: 'VIDEO-TO-VIDEO EDIT — the REF-DRIVEN edit tool: takes an EXISTING 3–15s clip and changes what the prompt names, with element/style reference images (@ElementN = frontal + angles) locking subject identity; max 4 combined refs. keep_audio preserves the ORIGINAL audio verbatim (spoken words cannot drift) — but video lips can drift slightly against it, and multi-beat instructions get under-executed (7/09 receipt: missed a second action beat Omni Flash edit landed) — ONE beat per pass. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the 7/09 fidelity head-to-head. Billed per second of output (≈ clip length, rounded up). Seedance edit/relocate is the alternative for style-transfer-heavy jobs.',
178
195
  },
179
196
  {
180
197
  id: 'veo-3.1',
181
198
  label: 'Veo 3.1',
182
199
  kind: 'video',
183
- maxRefImages: null,
184
- maxIngredients: 3,
200
+ // Family-level fact — fast and standard declare the same caps.
201
+ ...caps('veo-3.1-fast'),
185
202
  notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one gen. 16:9 only, 4/6/8s only. Otherwise Kling (default) or Seedance (physics/premium) win.',
186
203
  },
187
204
  {
188
205
  id: 'omni-flash',
189
206
  label: 'Gemini Omni Flash',
190
207
  kind: 'video',
191
- maxRefImages: null,
192
- maxIngredients: 7, // ref2v image_urls; 7 mirrors Google's own reference limit
208
+ // 7 ref2v image_urls — mirrors Google's own reference limit.
209
+ ...caps('omni-flash'),
193
210
  notes: 'CHEAP 720p tier with native synced audio included — t2v, single-start-frame i2v, or reference-to-video with up to 7 reference images. 3-10s, 16:9/9:16 only. No last frame, no video/audio references. VIDEO-ONLY. New seat: quality vs Kling/Seedance unproven pending comparison gens — do not route hero shots here; use it for cheap drafts, audio-in-one-gen at low cost, ref2v character consistency trials, and its edit variant.',
194
211
  },
195
212
  {
196
213
  id: 'omni-flash-edit',
197
214
  label: 'Omni Flash Edit',
198
215
  kind: 'video',
199
- maxRefImages: null,
200
- maxIngredients: 0, // prompt + source clip ONLY — no element/style refs on this endpoint
216
+ // 0: prompt + source clip ONLY — no element/style refs on this endpoint.
217
+ ...caps('omni-flash-edit'),
201
218
  notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (7/09 head-to-head vs Kling edit on real talking footage: lips held perfectly, audio near-identical, both action beats landed). Takes an EXISTING 3-10s clip and changes what the prompt names, footage-synced (prop/effect/environment/lighting swaps). Fidelity is EARNED by prompt discipline: ONE short instruction + "Keep everything else the same." — long descriptive prompts DESTROY it (Google-documented + 7/09 receipt). Never name objects as metaphors ("candle-like" → literal candle). Quirk: occasional tail jitter/doubled last speech beat — trim the tail. NO reference images (identity swaps needing refs → Kling edit); bit-exact audio needs → Kling keep_audio or segment-splice. 720p output, cheapest edit seat (~2/3 of Kling edit Std).',
202
219
  },
203
220
  {
204
221
  id: 'seed-audio',
205
222
  label: 'Seed Audio 1.0',
206
223
  kind: 'audio',
207
- maxRefImages: 1, // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
208
- maxIngredients: null,
224
+ // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
225
+ ...caps('seed-audio'),
209
226
  notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
210
227
  },
211
228
  {
212
229
  id: 'eleven-sfx',
213
230
  label: 'ElevenLabs Sound Effects v2',
214
231
  kind: 'audio',
215
- maxRefImages: null,
216
- maxIngredients: null,
232
+ ...caps('eleven-sfx'),
217
233
  notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
218
234
  },
219
235
  ];