@slatesvideo/shared 0.6.7 → 0.6.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,83 +1,83 @@
1
- {
2
- "name": "@slatesvideo/shared",
3
- "version": "0.6.7",
4
- "description": "Shared operations layer for the Slates MCP server and CLI: auth, cloud/desktop clients, and the single tool surface both consume. Most users want @slatesvideo/mcp-server or @slatesvideo/cli instead.",
5
- "license": "MIT",
6
- "type": "module",
7
- "main": "./dist/index.js",
8
- "types": "./dist/index.d.ts",
9
- "exports": {
10
- ".": {
11
- "import": "./dist/index.js",
12
- "types": "./dist/index.d.ts"
13
- },
14
- "./auth": "./dist/auth.js",
15
- "./clients/cloud": "./dist/clients/cloud.js",
16
- "./clients/desktop": "./dist/clients/desktop.js",
17
- "./operations": "./dist/operations/index.js",
18
- "./prompts": {
19
- "import": "./dist/prompts/index.js",
20
- "types": "./dist/prompts/index.d.ts"
21
- },
22
- "./model-capabilities": {
23
- "types": "./dist/prompts/model-capabilities.d.ts",
24
- "default": "./dist/prompts/model-capabilities.js"
25
- },
26
- "./shot-grammar": {
27
- "types": "./dist/prompts/shot-grammar.d.ts",
28
- "default": "./dist/prompts/shot-grammar.js"
29
- },
30
- "./asset-label": {
31
- "types": "./dist/prompts/asset-label.d.ts",
32
- "default": "./dist/prompts/asset-label.js"
33
- }
34
- },
35
- "files": [
36
- "dist",
37
- "!dist/**/*.map",
38
- "skills",
39
- "exports/slates-prompt-builder/generated",
40
- "README.md"
41
- ],
42
- "scripts": {
43
- "sync-partials": "node scripts/sync-partials.mjs",
44
- "build-prompt-builder": "node scripts/build-prompt-builder.mjs",
45
- "check-prompt-builder": "node scripts/build-prompt-builder.mjs --check",
46
- "build": "node scripts/sync-partials.mjs --check && node scripts/build-prompt-builder.mjs --check && node scripts/embed-skills.mjs && tsc && node scripts/render-capability-partials.mjs --check",
47
- "typecheck": "node scripts/sync-partials.mjs --check && node scripts/build-prompt-builder.mjs --check && node scripts/embed-skills.mjs && tsc --noEmit",
48
- "prepublishOnly": "npm run build",
49
- "render-partials": "node scripts/render-capability-partials.mjs"
50
- },
51
- "repository": {
52
- "type": "git",
53
- "url": "git+https://github.com/EricDisero/slates-mcp.git",
54
- "directory": "packages/shared"
55
- },
56
- "homepage": "https://slates.video",
57
- "bugs": {
58
- "url": "https://github.com/EricDisero/slates-mcp/issues"
59
- },
60
- "keywords": [
61
- "slates",
62
- "mcp",
63
- "model-context-protocol",
64
- "ai-video",
65
- "video-generation",
66
- "image-generation",
67
- "claude",
68
- "veo",
69
- "kling",
70
- "seedance"
71
- ],
72
- "engines": {
73
- "node": ">=18"
74
- },
75
- "dependencies": {
76
- "zod": "^3.23.0",
77
- "zod-to-json-schema": "^3.23.0"
78
- },
79
- "devDependencies": {
80
- "fflate": "^0.8.3",
81
- "typescript": "^5.7.0"
82
- }
83
- }
1
+ {
2
+ "name": "@slatesvideo/shared",
3
+ "version": "0.6.9",
4
+ "description": "Shared operations layer for the Slates MCP server and CLI: auth, cloud/desktop clients, and the single tool surface both consume. Most users want @slatesvideo/mcp-server or @slatesvideo/cli instead.",
5
+ "license": "MIT",
6
+ "type": "module",
7
+ "main": "./dist/index.js",
8
+ "types": "./dist/index.d.ts",
9
+ "exports": {
10
+ ".": {
11
+ "import": "./dist/index.js",
12
+ "types": "./dist/index.d.ts"
13
+ },
14
+ "./auth": "./dist/auth.js",
15
+ "./clients/cloud": "./dist/clients/cloud.js",
16
+ "./clients/desktop": "./dist/clients/desktop.js",
17
+ "./operations": "./dist/operations/index.js",
18
+ "./prompts": {
19
+ "import": "./dist/prompts/index.js",
20
+ "types": "./dist/prompts/index.d.ts"
21
+ },
22
+ "./model-capabilities": {
23
+ "types": "./dist/prompts/model-capabilities.d.ts",
24
+ "default": "./dist/prompts/model-capabilities.js"
25
+ },
26
+ "./shot-grammar": {
27
+ "types": "./dist/prompts/shot-grammar.d.ts",
28
+ "default": "./dist/prompts/shot-grammar.js"
29
+ },
30
+ "./asset-label": {
31
+ "types": "./dist/prompts/asset-label.d.ts",
32
+ "default": "./dist/prompts/asset-label.js"
33
+ }
34
+ },
35
+ "files": [
36
+ "dist",
37
+ "!dist/**/*.map",
38
+ "skills",
39
+ "exports/slates-prompt-builder/generated",
40
+ "README.md"
41
+ ],
42
+ "scripts": {
43
+ "sync-partials": "node scripts/sync-partials.mjs",
44
+ "build-prompt-builder": "node scripts/build-prompt-builder.mjs",
45
+ "check-prompt-builder": "node scripts/build-prompt-builder.mjs --check",
46
+ "build": "node scripts/sync-partials.mjs --check && node scripts/build-prompt-builder.mjs --check && node scripts/embed-skills.mjs && tsc && node scripts/render-capability-partials.mjs --check",
47
+ "typecheck": "node scripts/sync-partials.mjs --check && node scripts/build-prompt-builder.mjs --check && node scripts/embed-skills.mjs && tsc --noEmit",
48
+ "prepublishOnly": "npm run build",
49
+ "render-partials": "node scripts/render-capability-partials.mjs"
50
+ },
51
+ "repository": {
52
+ "type": "git",
53
+ "url": "git+https://github.com/EricDisero/slates-mcp.git",
54
+ "directory": "packages/shared"
55
+ },
56
+ "homepage": "https://slates.video",
57
+ "bugs": {
58
+ "url": "https://github.com/EricDisero/slates-mcp/issues"
59
+ },
60
+ "keywords": [
61
+ "slates",
62
+ "mcp",
63
+ "model-context-protocol",
64
+ "ai-video",
65
+ "video-generation",
66
+ "image-generation",
67
+ "claude",
68
+ "veo",
69
+ "kling",
70
+ "seedance"
71
+ ],
72
+ "engines": {
73
+ "node": ">=18"
74
+ },
75
+ "dependencies": {
76
+ "zod": "^3.23.0",
77
+ "zod-to-json-schema": "^3.23.0"
78
+ },
79
+ "devDependencies": {
80
+ "fflate": "^0.8.3",
81
+ "typescript": "^5.7.0"
82
+ }
83
+ }
@@ -31,7 +31,7 @@ The tables below are a snapshot. This roster churns constantly (NB2 Lite, Omni F
31
31
  | **One take longer than 15 seconds**, or a shot needing more than 9 image references, or an AUDIO-ONLY reference, or **beats that have to land at a named second** | **Seedance 2.5** | A SECOND SEAT beside 2.0, never an upgrade: 4–30s in one take, 30 image + 10 video + 10 audio references, audio-only refs, and the only Seedance seat that **acts on timestamps** (rules in `slates-prompting-seedance-2-5` § Timestamps) — 480p / 720p / 1080p, **no 4K**, and **dearer than 2.0 at every shared resolution** (720p $0.231/s vs $0.15/s, +54%). If you want 4K, or the same resolution cheaper, stay on 2.0. 🚨 Two live hazards: (a) with references attached, the words *add / remove / replace / change / extend / continue* make it reclassify the request as a video EDIT and fail AFTER the job queues — describe the finished frame, or use `seedance-2.5-edit`; (b) LENGTH is the price dial, not resolution — a 30s 720p face gen is 489 credits and a 30s 1080p faceless gen is 614, against a 1,000-credit welcome grant. Quote before any take over ~10s. |
32
32
  | **The SOUND has to be directed, not just present** — a specific line delivered a specific way, scene sound that has to sit under it, and score that must stay out of the characters' world | **MiniMax H3** | The only seat where audio is authored in three separate layers in ONE pass (synchronised events in the body, ambience in a soundscape section, audience-only score in its own) rather than toggled on. 5–15s, 480p / 768p / 2K / 4K, 24fps, 32kHz stereo, 11 languages. Rules in `slates-prompting-minimax-h3`. |
33
33
  | **A reference has to keep a DECLARED amount of itself** — especially moving one subject's characteristic onto a *different* subject | **MiniMax H3** | The only seat that understands a stated retention relationship (kept whole / kept in part / transferred onto another subject / loose echo). 9 images + 3 video + 3 audio, 12 files total. 🚨 The first 5 reference images are free and every one after that costs 4 credits — pass `referenceImages` to `slates_estimate_generation_cost` before a reference-heavy job. |
34
- | **Turnaround is the requirement** on a text-to-video or start-frame shot at 480p/768p | **MiniMax H3 Max** | fal's self-hosted post-train of H3. **Measured 2026-08-27: a 5s 768p clip finished in 4.8s against 57s on base H3 — about 12x faster**, same prompt, queue to file. When turnaround is the requirement this is not a marginal win. 🚨 It is the PREMIUM seat, not a cheap H3 — $0.080/s at 768p against base H3's $0.060/s, 33% more, and it tops out at 768p. It still animates a start frame and an end frame — image-to-video is one of the two things it is for — but it has no reference-to-video endpoint, so the omni-reference set (9 images + video + audio) is base-H3 only. Never the default; never reach for it to save money. |
34
+ | **Turnaround is the requirement** on a text-to-video or start-frame shot at 480p/768p | **MiniMax H3 Max** | fal's self-hosted post-train of H3. **Measured 2026-08-27: a 5s 768p clip finished in 4.8s against 57s on base H3 — about 12x faster**, same prompt, queue to file. When turnaround is the requirement this is not a marginal win. 🚨 It is the PREMIUM seat, not a cheap H3 — $0.080/s at 768p against base H3's $0.060/s, 33% more, and it tops out at 768p. It still animates a start frame and an end frame — image-to-video is one of the two things it is for — and since 2026-09-09 it takes the full omni-reference set too (9 images + 3 video + 3 audio), so the seats now differ on ladder and price rather than on what they accept. Never the default; never reach for it to save money. |
35
35
  | Native synchronized audio (dialogue + SFX generated WITH the video in one gen), 16:9, ≤8s | Veo 3.1 | Narrow, and now narrower: if the sound needs DIRECTING rather than merely existing, MiniMax H3 is the better seat. |
36
36
 
37
37
  ### Named Seedance escalation triggers
@@ -66,7 +66,7 @@ Fork each bound frame's image Shot with `slates_duplicate_shot` (`model:` the vi
66
66
  **Model mixing — route per `slates-model-selection`** (details in the per-model guides):
67
67
  - **Kling V3** (`slates-prompting-kling-v3`): the DEFAULT for most shots — 16:9 / 9:16 / 1:1, 3-15s, strong start-frame adherence; std is the workhorse, Omni for multi-character dialogue.
68
68
  - **Seedance 2** (`slates-prompting-seedance`): the PREMIUM tier — any shot where physics/effects/scale remotely matter, plus the hero shot; audio included, first+last frame guidance, native 4K (4K video is Pro-only).
69
- - **MiniMax H3** (`slates-prompting-minimax-h3`): route here when a shot's SOUND is part of the writing — a line delivered a particular way, scene sound under it, score that must stay outside the characters' world. It authors all three in one pass, which **collapses a shot's audio pass into its video pass** and removes the separate `slates_generate_audio` step for that shot. 5-15s, 480p/768p/2K/4K. Its sibling `minimax-h3-max` is faster but capped at 768p, takes no references, and costs MORE at 768p — a deliberate speed pick, never a saving.
69
+ - **MiniMax H3** (`slates-prompting-minimax-h3`): route here when a shot's SOUND is part of the writing — a line delivered a particular way, scene sound under it, score that must stay outside the characters' world. It authors all three in one pass, which **collapses a shot's audio pass into its video pass** and removes the separate `slates_generate_audio` step for that shot. 5-15s, 480p/768p/2K/4K. Its sibling `minimax-h3-max` is faster, tops out at 1080p, takes the same references, and costs MORE at 768p — a deliberate speed pick, never a saving.
70
70
  - **Veo 3.1** (`slates-prompting-veo-3`): niche, never the default — only when native synced audio must generate WITH the video in one gen; 16:9 or 9:16, 4/6/8s (8s only at 1080p/4K or with reference images).
71
71
 
72
72
  Failed gen? The run continues past it and **nothing is retried automatically**. Read the per-Shot error in the result, fix that Shot with `slates_update_shot`, and re-fire only it (a retry beyond the plan = announce the delta cost).
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: slates-prompting-minimax-h3
3
- description: How to prompt MiniMax H3 and MiniMax H3 Max. Read before calling slates_generate_video with model minimax-h3 or minimax-h3-max. H3 is the only Slates video seat where AUDIO IS AUTHORED rather than toggled — synchronised dialogue, scene sound and an audience-only score are three separate sections of the prompt, generated in one pass — and the only one where a reference carries a DECLARED RELATIONSHIP (kept whole, partly kept, transferred onto a different subject, or a loose echo). Base minimax-h3 runs 480p/768p/2K/4K and reads 9 images + 3 video + 3 audio references; minimax-h3-max is fal's faster post-train, capped at 768p, and costs MORE than base H3 at 768p — a deliberate speed pick, never the default and never the cheap one; it still animates start and end frames, but it has no reference-to-video endpoint, so the omni-reference set is base-H3 only. Two hazards live here: reference images past the fifth cost 4 credits each on the base row, and audio written into the wrong section is dropped or duplicated.
3
+ description: How to prompt MiniMax H3 and MiniMax H3 Max. Read before calling slates_generate_video with model minimax-h3 or minimax-h3-max. H3 is the only Slates video seat where AUDIO IS AUTHORED rather than toggled — synchronised dialogue, scene sound and an audience-only score are three separate sections of the prompt, generated in one pass — and the only one where a reference carries a DECLARED RELATIONSHIP (kept whole, partly kept, transferred onto a different subject, or a loose echo). Base minimax-h3 runs 480p/768p/2K/4K and reads 9 images + 3 video + 3 audio references; minimax-h3-max is fal's faster post-train, capped at 768p, and costs MORE than base H3 at 768p — a deliberate speed pick, never the default and never the cheap one; it animates start and end frames AND takes the same 9+3+3 omni-reference set (corrected 2026-09-09), so the seats differ on ladder and price, not on what they accept. Two hazards live here: reference images past the free allowance are billed (5 free then +4 credits on base H3; 4 free then +1 on Max), and audio written into the wrong section is dropped or duplicated.
4
4
  ---
5
5
 
6
6
  # MiniMax H3 — prompting
@@ -28,7 +28,7 @@ description: How to prompt MiniMax H3 and MiniMax H3 Max. Read before calling sl
28
28
  - `A woman sits still at a kitchen table for a beat, then looks up. She says in English, "You said Tuesday." Scene sound: a fridge hum, a spoon set down on formica. Score: none.`
29
29
  - `Two mechanics either side of an open bonnet. The younger one wipes his hands, waits, then speaks in Spanish, "No es el alternador." Scene sound: a socket wrench, a radio two bays over. Score: a low sustained cello under the last three seconds, audience only.`
30
30
 
31
- **Hard constraint:** the two seats differ in what the ENDPOINT accepts, not in grammar. Base H3 reaches 2K/4K and takes references; `minimax-h3-max` is capped at 768p, has NO reference endpoint at all (start and end frames still work), and costs MORE at the tier they share — it is a speed pick, never the cheap one. H3's top two resolution tiers are UPSCALES of the native render: judge at native. Reference images past the fifth are a paid dimension of the cost key — declare the count when quoting.
31
+ **Hard constraint:** the two seats differ in what the ENDPOINT accepts, not in grammar. Base H3 reaches 2K/4K and takes references; `minimax-h3-max` tops out at 1080p rather than 4K, takes the same 9+3+3 references, and costs MORE at the tier they share — it is a speed pick, never the cheap one. H3's top two resolution tiers are UPSCALES of the native render: judge at native. Reference images past the free allowance are a paid dimension of the cost key (5 free on base, 4 on Max) — declare the count when quoting.
32
32
  <!-- @card:end -->
33
33
 
34
34
  <!-- @banned:start -->
@@ -55,8 +55,8 @@ endpoint accepts:
55
55
 
56
56
  | | `minimax-h3` | `minimax-h3-max` |
57
57
  |---|---|---|
58
- | Resolution | 480p / 768p / **2K / 4K** | 480p / 768p |
59
- | References | 9 images + 3 video + 3 audio (12 files) | **none — no reference endpoint exists.** Frames still work; see the row below |
58
+ | Resolution | 480p / 768p / **2K / 4K** | 480p / 768p / **1080p** |
59
+ | References | 9 images + 3 video + 3 audio (12 files) | 9 images + 3 video + 3 audio (12 files) |
60
60
  | Frames | start and/or end | start and/or end |
61
61
  | Price at 768p | **$0.060/s** | $0.080/s |
62
62
  | Why pick it | resolution, references, and the cheaper second | **speed** — a 5s 768p clip in **4.8s** vs **57s** (measured) |
@@ -71,6 +71,12 @@ paying for; route to base H3 for anything needing resolution, references, or the
71
71
  4.8s wall-clock, but the order of magnitude did. For iteration loops and client-present work that gap
72
72
  is the entire reason the seat exists.
73
73
 
74
+ 🚨 **Max's known weakness: colour banding in low light (Eric, 2026-09-09).** Certain shots —
75
+ especially dark or low-key ones — come back with low-bitrate-looking banding across gradients (skies,
76
+ walls, shadow falloff). It is the one place the seat visibly gives something up. If a shot is dark
77
+ and gradient-heavy, either light it up in the prompt or route to base H3 at 768p; do not fix it by
78
+ reaching for 2K, which adds its own artifacting on top.
79
+
74
80
  ---
75
81
 
76
82
  ## The one thing that makes H3 different: audio is a THREE-LAYER instruction
@@ -186,8 +192,8 @@ original wording preserved exactly: *A red neon sign reading "Open Late" glows a
186
192
 
187
193
  ## References — H3's real differentiator is the declared RELATIONSHIP
188
194
 
189
- *(Base `minimax-h3` only. `minimax-h3-max` has no reference endpoint — Slates refuses references
190
- on that row rather than dropping them silently.)*
195
+ *(BOTH rows. `minimax-h3-max` gained the reference set on 2026-09-09; its free allowance is
196
+ FOUR images rather than the base row's five.)*
191
197
 
192
198
  <!-- @inject:references-read-literally -->
193
199
  > **The general law: the model reads a reference literally.**
@@ -254,10 +260,13 @@ exactly and say so.
254
260
  **An audio reference cannot travel alone** — H3 refuses a reference set that is audio only. Pair it
255
261
  with at least one image or video reference.
256
262
 
257
- ### 💸 Reference images past the fifth cost 4 credits each
263
+ ### 💸 Reference images past the free allowance are billed — and the two rows differ
258
264
 
259
- The first **5** reference images are free. Each additional image — the model takes **9** — adds
260
- **4 credits** to the generation, at every resolution and every length. Four extra images on a 10s
265
+ On `minimax-h3` the first **5** are free and each additional image adds **4 credits**. On
266
+ `minimax-h3-max` the first **4** are free and each additional image adds **1 credit** — fal prices
267
+ Max's references by token rather than per image, and Slates normalises every Max reference to
268
+ 1024x1024 so that per-image number is exact. Both rows take **9** images, at every resolution and
269
+ every length. Four extra images on a 10s
261
270
  768p clip add 16 credits to a 30-credit generation: **more than half again**, for references that
262
271
  often make the output worse rather than better (see the 2–4 rule above).
263
272
 
@@ -291,7 +300,9 @@ dropping one side.
291
300
  | `minimax-h3` · 2K · 10s | 65 |
292
301
  | `minimax-h3` · 4K · 10s | 80 |
293
302
  | `minimax-h3-max` · 768p · 10s | 40 |
294
- | every reference image past the fifth | **+4** |
303
+ | `minimax-h3-max` · 1080p · 10s | 80 |
304
+ | `minimax-h3` — every reference image past the **fifth** | **+4** |
305
+ | `minimax-h3-max` — every reference image past the **fourth** | **+1** |
295
306
 
296
307
  **768p is the default for a reason.** It is the tier the model natively generates.
297
308
 
@@ -302,8 +313,13 @@ cannot add information.
302
313
 
303
314
  **In our own test (2026-08-27, same prompt, same seed) the 2K pass came back with MORE artifacting
304
315
  than the 768p original it was built from**, while costing 33 credits for a 5-second take against 15,
305
- and taking nearly twice as long to return. One shot, so treat it as a warning rather than a law —
306
- but the mechanism explains it, and the burden of proof is on 2K.
316
+ and taking nearly twice as long to return.
317
+
318
+ 🚨 **Confirmed independently (Eric, 2026-09-09): 2K and 4K carry visible AI noise artifacting and
319
+ "just look bad".** That is now TWO separate observations, months apart, pointing the same way — it is
320
+ no longer a single-shot warning. The tiers stay available because a delivery spec sometimes demands
321
+ the pixels, but **do not route to 2K/4K for quality**: you are paying more, waiting longer, and
322
+ adding artifacts to a 768p render. Upscale in post from a clean 768p master instead.
307
323
 
308
324
  **So: generate at 768p and judge it at 768p.** Reach for 2K or 4K only when a delivery spec demands
309
325
  the pixels, and expect to be paying for size rather than quality — a post-production upscale from a