@slatesvideo/shared 0.6.2 → 0.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/dist/api-url.d.ts +9 -0
  2. package/dist/api-url.js +9 -0
  3. package/dist/auth.d.ts +13 -1
  4. package/dist/auth.js +9 -5
  5. package/dist/clients/cloud.d.ts +3 -0
  6. package/dist/clients/cloud.js +34 -3
  7. package/dist/clients/desktop.js +3 -0
  8. package/dist/index.d.ts +8 -2
  9. package/dist/index.js +44 -1
  10. package/dist/operations/index.d.ts +243 -31
  11. package/dist/operations/index.js +1483 -154
  12. package/dist/operations/surface.d.ts +69 -0
  13. package/dist/operations/surface.js +227 -0
  14. package/dist/prompts/agent-doctrine.d.ts +36 -0
  15. package/dist/prompts/agent-doctrine.js +201 -0
  16. package/dist/prompts/asset-label.d.ts +23 -0
  17. package/dist/prompts/asset-label.js +70 -0
  18. package/dist/prompts/banned-tokens.d.ts +40 -0
  19. package/dist/prompts/banned-tokens.js +219 -0
  20. package/dist/prompts/character-sheet.d.ts +0 -2
  21. package/dist/prompts/character-sheet.js +0 -2
  22. package/dist/prompts/craft-cards.d.ts +20 -0
  23. package/dist/prompts/craft-cards.js +82 -0
  24. package/dist/prompts/environment-sheet.js +16 -0
  25. package/dist/prompts/index.d.ts +1 -0
  26. package/dist/prompts/index.js +4 -0
  27. package/dist/prompts/model-capabilities.d.ts +65 -1
  28. package/dist/prompts/model-capabilities.js +139 -2
  29. package/dist/prompts/model-facts.d.ts +20 -4
  30. package/dist/prompts/model-facts.js +95 -27
  31. package/dist/prompts/partials.generated.js +2 -1
  32. package/dist/prompts/prompting-tips.d.ts +1 -1
  33. package/dist/prompts/prompting-tips.js +123 -0
  34. package/dist/prompts/reference-composer.d.ts +36 -7
  35. package/dist/prompts/reference-composer.js +75 -20
  36. package/dist/prompts/reference-rules.d.ts +15 -26
  37. package/dist/prompts/reference-rules.js +15 -93
  38. package/dist/prompts/shot-grammar.d.ts +154 -0
  39. package/dist/prompts/shot-grammar.js +184 -0
  40. package/dist/prompts/shot-spec.d.ts +265 -0
  41. package/dist/prompts/shot-spec.js +303 -0
  42. package/dist/skills/content.js +25 -22
  43. package/exports/slates-prompt-builder/generated/SKILL.md +3 -3
  44. package/exports/slates-prompt-builder/generated/reference-content-policy.md +6 -0
  45. package/exports/slates-prompt-builder/generated/reference-kling.md +22 -0
  46. package/exports/slates-prompt-builder/generated/reference-nano-banana.md +17 -0
  47. package/exports/slates-prompt-builder/generated/reference-seedance.md +19 -1
  48. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +17 -17
  49. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  50. package/package.json +83 -73
  51. package/skills/_partials/decision-log.md +5 -4
  52. package/skills/_partials/thresholds.md +19 -0
  53. package/skills/slates-content-policy.md +15 -1
  54. package/skills/slates-cost-discipline.md +26 -4
  55. package/skills/slates-model-selection.md +2 -2
  56. package/skills/slates-one-prompt-film.md +20 -12
  57. package/skills/slates-project-organization.md +1 -1
  58. package/skills/slates-prompting-elevenlabs.md +61 -2
  59. package/skills/slates-prompting-flux-2-max.md +39 -0
  60. package/skills/slates-prompting-gpt-image-2.md +109 -70
  61. package/skills/slates-prompting-inworld-tts.md +166 -0
  62. package/skills/slates-prompting-kling-v3.md +39 -0
  63. package/skills/slates-prompting-lip-sync.md +38 -0
  64. package/skills/slates-prompting-ltx-2-5.md +218 -0
  65. package/skills/slates-prompting-minimax-h3.md +39 -0
  66. package/skills/slates-prompting-motion-transfer.md +38 -0
  67. package/skills/slates-prompting-nano-banana-2.md +36 -0
  68. package/skills/slates-prompting-omni-flash.md +41 -0
  69. package/skills/slates-prompting-seed-audio.md +38 -0
  70. package/skills/slates-prompting-seedance-2-5.md +38 -0
  71. package/skills/slates-prompting-seedance.md +36 -1
  72. package/skills/slates-prompting-seedream-5-lite.md +38 -0
  73. package/skills/slates-prompting-veo-3.md +39 -0
  74. package/skills/slates-shot-variety.md +53 -0
  75. package/skills/slates-storyboard-from-script.md +31 -15
  76. package/skills/slates-style-prompting.md +1 -1
  77. package/skills/slates-vision-feedback-loop.md +1 -1
@@ -88,6 +88,18 @@ const SEEDANCE_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
88
88
  * is not an AspectRatio in this vocabulary.
89
89
  */
90
90
  const MINIMAX_H3_ASPECT_RATIOS = ['21:9', '16:9', '4:3', '1:1', '3:4', '9:16'];
91
+ /**
92
+ * LTX-2.5, all four integrated endpoints: TWO. Read off fal's live OpenAPI
93
+ * 2026-08-29 — `text-to-video/{fast,pro}` declare exactly `['16:9','9:16']`,
94
+ * the narrowest video set in the roster alongside Veo-on-fal.
95
+ *
96
+ * `image-to-video/{fast,pro}` additionally offer `auto` (follow the start
97
+ * frame). We never send it and it is not an `AspectRatio` in this vocabulary —
98
+ * identical treatment to H3's `adaptive`, and for the identical reason: the
99
+ * composer always holds an explicit ratio, so `auto` would only ever be a way
100
+ * to lose track of what was actually generated.
101
+ */
102
+ const LTX_2_5_ASPECT_RATIOS = ['16:9', '9:16'];
91
103
  /**
92
104
  * The provider every AGENT generation actually lands on for Kling and Veo.
93
105
  *
@@ -371,6 +383,87 @@ export const MODEL_CAPABILITIES = {
371
383
  // above what the handler sends is a SILENT DROP — the exact failure
372
384
  // `seedance-2.5-edit` shipped with. Absent means the composer refuses.
373
385
  },
386
+ // ── LTX-2.5 (both seats on fal — added 2026-08-29) ─────────────────────────
387
+ //
388
+ // Every value below is READ OFF fal's live OpenAPI, fetched 2026-08-29:
389
+ // lightricks/ltx-2.5/{text-to-video,image-to-video}/{fast,pro}
390
+ // Note the owner namespace — no `fal-ai/` prefix, exactly like `minimax/h3/`.
391
+ //
392
+ // 🚨 SAME PREFIX COLLISION AS THE MINIMAX PAIR, AND IT IS WORSE HERE.
393
+ // `ltx-2-5-pro` starts with `ltx-2-5`, so ANY `startsWith('ltx-2-5')` swallows
394
+ // the Pro row into the Fast row's branch — a shorter ladder, a shorter
395
+ // duration list AND a 31-33% higher price at both tiers they share. Every
396
+ // lookup downstream is an exact-id map, never a prefix test.
397
+ //
398
+ // 🚨 THE CHEAP ROW IS THE ONE WITH THE LONGER REACH. Counter to how every
399
+ // other Fast/Pro pair in this file behaves, `ltx-2-5` (the distilled 8-step
400
+ // build) reaches 1440p, 4K and 20s while `ltx-2-5-pro` (full diffusion) stops
401
+ // at 1080p and 10s. Pro buys fidelity on a NARROWER ladder. Do not "fix" this
402
+ // by assuming Pro is a superset — it is not, on either axis.
403
+ //
404
+ // ⛔ `audio-to-video/{fast,pro}` EXIST AND ARE DELIBERATELY ABSENT from this
405
+ // file. They carry no `resolution` and no `duration` parameter at all: output
406
+ // length is dictated by the uploaded audio, so they bill per second of INPUT
407
+ // while every credit key we own bills OUTPUT. That is a billing-SHAPE change,
408
+ // not a missing row. See slates-api/PRICING.md.
409
+ 'ltx-2-5': {
410
+ aspectRatios: LTX_2_5_ASPECT_RATIOS,
411
+ // Full ladder, all four tiers NATIVELY generated (no upscale pass anywhere
412
+ // — the contrast with H3's 2K/4K is the whole reason `1440p` is its own
413
+ // token). DEFAULT 1080p, which is also fal's own schema default: unusually
414
+ // for this file the default tier is not the floor. It is the cheapest
415
+ // native 1080p second in the roster at $0.130/s.
416
+ videoResolution: { options: ['720p', '1080p', '1440p', '4k'], default: '1080p' },
417
+ // 🚨 DISCRETE AND EVEN-ONLY. There is NO 5s LTX clip and no odd duration of
418
+ // any length — fal's enum is literally [6,8,10,12,14,16,18,20]. A
419
+ // `{ min: 6, max: 20, mode: 'continuous' }` here would offer 7s in the
420
+ // composer, quote a `ltx-2-5-1080p-7s` key that exists on no server, and
421
+ // fail at the proxy after the user had already chosen it.
422
+ //
423
+ // The evenness is also what makes all 28 cost keys round exactly (every
424
+ // basis is a multiple of CENTS_PER_CREDIT=3) — see PRICING.md. If fal ever
425
+ // admits odd seconds, the rounding proof must be re-verified.
426
+ duration: {
427
+ min: 6,
428
+ max: 20,
429
+ mode: 'discrete',
430
+ values: [6, 8, 10, 12, 14, 16, 18, 20],
431
+ // fal: "At 720p and 1080p, 24 or 25 FPS supports up to 20 seconds […] At
432
+ // 1440p and 2160p, all frame rates support up to 10 seconds."
433
+ //
434
+ // ⚠️ THE REAL CEILING IS A FUNCTION OF RESOLUTION *AND* FPS, and this
435
+ // field can only express the resolution half. It is correct ONLY because
436
+ // we pin fps to fal's default of 25 and never expose the control. If fps
437
+ // is ever exposed, 48/50 drops the 720p/1080p ceiling to 10s too, and
438
+ // that needs a second override axis — not a wider window here.
439
+ resolutionOverrides: {
440
+ '1440p': { min: 6, max: 10, mode: 'discrete', values: [6, 8, 10] },
441
+ '4k': { min: 6, max: 10, mode: 'discrete', values: [6, 8, 10] },
442
+ },
443
+ },
444
+ // NO reference caps of any kind, deliberately. fal publishes text-to-video
445
+ // and image-to-video for LTX and NOTHING else — there is no
446
+ // `reference-to-video` endpoint, so there is no transport for an ingredient
447
+ // or a multimodal reference, and a cap declared above what the handler
448
+ // sends is a SILENT DROP (the failure `seedance-2.5-edit` shipped with).
449
+ // The start/end frames i2v does carry are FRAME SLOTS, not references, and
450
+ // live in MODEL_REGISTRY.features.lastFrame — not here.
451
+ },
452
+ 'ltx-2-5-pro': {
453
+ aspectRatios: LTX_2_5_ASPECT_RATIOS,
454
+ // 720p/1080p ONLY. Declaring the shorter ladder here IS the whole Pro-seat
455
+ // mechanism: `assertVideoCapabilities` refuses 1440p/4K on this id, the
456
+ // desktop picker renders only what this entry declares, and the agent's Zod
457
+ // enum stays the union while the per-model guard narrows. Anything shaped
458
+ // like "hide the top tiers when Pro is selected" re-implements a guard that
459
+ // already exists.
460
+ videoResolution: { options: ['720p', '1080p'], default: '1080p' },
461
+ // Three values, full stop — fal's enum is [6,8,10] on both Pro endpoints,
462
+ // with no resolution override needed because the ceiling is already 10s at
463
+ // both tiers this row reaches.
464
+ duration: { min: 6, max: 10, mode: 'discrete', values: [6, 8, 10] },
465
+ // No reference caps — same reasoning as the Fast row above.
466
+ },
374
467
  // ── Audio ──────────────────────────────────────────────────────────────────
375
468
  //
376
469
  // `aspectRatios: []` is deliberate, not an oversight: audio has no frame, and
@@ -388,11 +481,53 @@ export const MODEL_CAPABILITIES = {
388
481
  'eleven-sfx': {
389
482
  aspectRatios: [],
390
483
  },
484
+ // ── Text-to-speech ─────────────────────────────────────────────────────────
485
+ //
486
+ // The TTS seat. `maxCharacters` is the one number the billing bucket is sized
487
+ // against, and it is MEASURED: the API rejects 2,001 characters by name
488
+ // ("text length should not exceed 2000 characters"). Do not raise it from a
489
+ // docs page — raise it from a request that succeeds.
490
+ //
491
+ // ⚠️ NO `durationSeconds` HERE, and that is the shape of the surface rather
492
+ // than an omission: speech length falls out of the text, so this row bills on
493
+ // characters and has no duration dimension at all. Every derivation that
494
+ // switches on an audio surface must read the BILLING UNIT, never assume one.
495
+ 'inworld-tts-2': {
496
+ aspectRatios: [],
497
+ maxCharacters: 2000,
498
+ voiceClone: {
499
+ // 5-15s of reference audio, ≤4 MB per sample — the vendor's documented
500
+ // spec, and a 12.6s / 555 KB sample cloned successfully against it.
501
+ minSeconds: 5,
502
+ maxSeconds: 15,
503
+ maxBytes: 4 * 1024 * 1024,
504
+ formats: ['wav', 'mp3', 'webm'],
505
+ // 🚨 MEASURED, and it appears in no documentation: the third clone inside
506
+ // one minute returned 429 "limit: 2, time window: m". This is workspace-
507
+ // wide, so it is shared across every Slates user.
508
+ clonesPerMinute: 2,
509
+ maxStoredVoices: 100,
510
+ // Measured: the design endpoint rejects a prompt outside these bounds by
511
+ // name ("design_prompt (Voice Description) must be between 7 and 1000").
512
+ designPromptChars: { min: 7, max: 1000 },
513
+ },
514
+ },
391
515
  };
392
516
  // ── Queries ──────────────────────────────────────────────────────────────────
393
517
  export function getModelCapability(model) {
394
518
  return MODEL_CAPABILITIES[model];
395
519
  }
520
+ /**
521
+ * The voice-cloning spec for a TTS surface, or undefined for anything else.
522
+ *
523
+ * Exists so the desktop reads the reference-audio bounds, the design-prompt
524
+ * bounds and the clone rate limit from HERE rather than retyping them into a
525
+ * form control. A control whose limit disagrees with the vendor's is a limit
526
+ * the user first meets AFTER pressing the button.
527
+ */
528
+ export function voiceCloneFor(model) {
529
+ return MODEL_CAPABILITIES[model]?.voiceClone;
530
+ }
396
531
  /** Aspect ratios a model accepts, honouring the provider override. */
397
532
  export function aspectRatiosFor(model, provider) {
398
533
  const cap = MODEL_CAPABILITIES[model];
@@ -461,8 +596,10 @@ export function aspectRatioUnion(models, provider) {
461
596
  /** Union of every resolution the given models accept. */
462
597
  export function videoResolutionUnion(models) {
463
598
  // Ascending by output height, so an enum reads as a ladder. 768p sits between
464
- // 720p and 1080p; 2k (≈2560×1440) between 1080p and 4k.
465
- const order = ['480p', '720p', '768p', '1080p', '2k', '4k'];
599
+ // 720p and 1080p; 1440p and 2k (≈2560×1440) are CO-HEIGHT and sit together
600
+ // between 1080p and 4k — their relative order here is cosmetic, because no
601
+ // model declares both (1440p is LTX-only, 2k is H3-only).
602
+ const order = ['480p', '720p', '768p', '1080p', '1440p', '2k', '4k'];
466
603
  const seen = new Set();
467
604
  for (const m of models)
468
605
  for (const r of videoResolutionsFor(m))
@@ -2,6 +2,14 @@ export interface ModelFact {
2
2
  id: string;
3
3
  label: string;
4
4
  kind: 'image' | 'video' | 'audio';
5
+ /**
6
+ * Which op this seat is reachable through. `edit` rows live on
7
+ * slates_edit_video and must never appear in slates_generate_video's routing
8
+ * list — they were mixed into one VIDEO block with nothing marking them.
9
+ * Explicit rather than an id-suffix test: a structural test on "-edit" holds
10
+ * only while that substring is unique, which is how isOmniFlashModel broke.
11
+ */
12
+ route: 'generate' | 'edit';
5
13
  /** Max reference images (image models) — null if not applicable. */
6
14
  maxRefImages: number | null;
7
15
  /** Max ingredient images (video models) — null if not applicable. */
@@ -59,9 +67,17 @@ export declare function seedanceTaskIntentWords(prompt: string): string[];
59
67
  /** Every model that reads reference video and/or audio, for op descriptions. */
60
68
  export declare function multimodalRefModels(): string[];
61
69
  export declare const MODEL_FACTS: ModelFact[];
70
+ /**
71
+ * Routing prose for one lane, generated from the SSOT.
72
+ *
73
+ * THE ONE RENDERER. The Studio Agent's system prompt, the MCP server's
74
+ * instructions and the generate/edit ops' `model` descriptions all call this —
75
+ * so "never restate model routing in an op description" (slates-mcp/CLAUDE.md)
76
+ * is now enforced by there being nothing to restate. Before this, the video op
77
+ * carried 1,282 characters of hand-written routing that repeated MODEL_FACTS
78
+ * phrase for phrase ("SECOND SEAT", "AUTHORED-AUDIO", "never the default"),
79
+ * in the same file that forbids exactly that.
80
+ */
81
+ export declare function describeRouting(kind: ModelFact['kind'], route?: ModelFact['route']): string;
62
82
  export declare function getModelFact(id: string): ModelFact | undefined;
63
- /** The official NB2 / general image prompt formula (subject-first). */
64
- export declare const IMAGE_PROMPT_FORMULA = "[Subject] + [Action] + [Location/context] + [Composition] + [Style]";
65
- /** The expanded cinematic/photoreal formula for NB2 start frames. */
66
- export declare const CINEMATIC_IMAGE_FORMULA = "Film still from [DIRECTOR] [GENRE]. Shot on [CAMERA] with [LENS]. [SUBJECT and action]. [3-5 specific visual details]. [LIGHTING \u2014 direction + quality]. [COLOR PALETTE]. [FILM STOCK or sensor language]. [1-2 word emotional tone].";
67
83
  //# sourceMappingURL=model-facts.d.ts.map
@@ -9,11 +9,22 @@
9
9
  // defect the capability SSOT exists to delete, so `caps()` below does the lookup
10
10
  // and a wrong id throws at module load instead of shipping a stale number.
11
11
  //
12
- // Prose that ALSO appears in a skill or the tips card comes from
13
- // skills/_partials/*.md via PARTIALS — never restated here. A `notes` string is
14
- // a third rendering of a fact, and a third rendering is a third thing that can
15
- // survive a doctrine reversal the other two got.
16
- import { PARTIALS } from './partials.generated.js';
12
+ // 🚨 `notes` IS ROUTING ONLY — why you would pick THIS seat over its neighbour.
13
+ // Nothing else. Not capability numbers (MODEL_CAPABILITIES owns those and already
14
+ // GENERATES prose for them into every op's param descriptions), not prices (the
15
+ // rate functions own those, and the agent's own REAL NUMBERS ONLY rule forbids it
16
+ // repeating a figure it cannot point to in a tool result), not prompt craft (the
17
+ // matching slates-prompting-* skill owns that, loaded on demand).
18
+ //
19
+ // It was all four for a while: 16,799 characters of `notes` carried 80 resolution
20
+ // tokens, 42 duration claims, 23 hard-typed prices and two skills' worth of craft
21
+ // into a prompt prefix that is always in context. A `notes` string is a third
22
+ // rendering of a fact, and a third rendering is a third thing that can survive a
23
+ // doctrine reversal the other two got. `scripts/agent-surface-lockstep-check.mjs`
24
+ // check 5 now fails the build if a price or a capability number reappears here.
25
+ //
26
+ // Relative cost claims STAY ("dearer than 2.0 at every shared tier") — that is
27
+ // routing. The figures go, because those are data.
17
28
  import { MODEL_CAPABILITIES } from './model-capabilities.js';
18
29
  /**
19
30
  * Reference caps for a fact, read out of the capability SSOT.
@@ -108,6 +119,7 @@ export function multimodalRefModels() {
108
119
  export const MODEL_FACTS = [
109
120
  {
110
121
  id: 'nano-banana-2',
122
+ route: 'generate',
111
123
  // Gemini 3.1 FLASH Image — verified against the runtime slug map in
112
124
  // slate/src/main/api/google.ts. Nano Banana PRO is a different model
113
125
  // (gemini-3-pro-image-preview); do not conflate them.
@@ -115,110 +127,124 @@ export const MODEL_FACTS = [
115
127
  kind: 'image',
116
128
  // 14 = 10 object-fidelity + 4 character-consistency; the categories don't trade.
117
129
  ...caps('nano-banana-2'),
118
- notes: 'Default image model. 14 refs hard cap (10 object + 4 character). Brief it like a creative director, not tag soup. No negativePrompt field — use positive reframing. Best image start-frame for legible text. Knowledge cutoff Jan 2025.',
130
+ notes: 'DEFAULT image model and the all-rounder — route here unless another seat\'s speciality is the point. Best start-frame for legible in-scene text. Knowledge cutoff Jan 2025: anything later needs reference images.',
119
131
  },
120
132
  {
121
133
  id: 'nano-banana-2-lite',
134
+ route: 'generate',
122
135
  label: 'Nano Banana 2 Lite',
123
136
  kind: 'image',
124
137
  ...caps('nano-banana-2-lite'),
125
- notes: 'FAST/DRAFT image tier — ~half the price of NB2 full, ~2.7× faster, 1K output ONLY. Same Gemini content filter as NB2. Route here for iteration volume and drafts where 1K is fine; keep NB2 full for final 2K/4K. Character consistency + legible text hold up.',
138
+ notes: 'FAST/DRAFT image tier — markedly cheaper and faster than NB2 full, at draft quality. Route here for iteration volume, then re-run the winner on NB2 full. Same Gemini content filter as NB2.',
126
139
  },
127
140
  {
128
141
  id: 'nano-banana-pro',
142
+ route: 'generate',
129
143
  label: 'Nano Banana Pro',
130
144
  kind: 'image',
131
145
  ...caps('nano-banana-pro'),
132
- notes: 'HERO-FRAME / typography PREMIUM image tier (Gemini 3 Pro backbone; ~2× NB2 price). NB2 ≈ 95% of Pro — route here only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element reasoning must be perfect. Up to 14 reference images (character locking, multi-subject fusion). Native 16:9 + 4K.',
146
+ notes: 'HERO-FRAME / typography PREMIUM image tier. NB2 is about 95% of Pro — escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene or deep multi-element reasoning must be perfect, and say why.',
133
147
  },
134
148
  {
135
149
  id: 'gpt-image-2',
150
+ route: 'generate',
136
151
  label: 'GPT Image 2',
137
152
  kind: 'image',
138
153
  ...caps('gpt-image-2'),
139
- notes: 'TEXT/DIAGRAM/PANEL king — near-perfect character-level text, ordered panels, exact placement (~3s gens). Route here for character sheets, shot grids, and text-bearing panels. Quality tiers: medium (default, the value seat — half NB2 price at 1080p) / high (~4×, max text precision). Third filter regime (OpenAI moderate). 4K is API-only — even paid ChatGPT can\'t render it. ALSO THE PHOTOREAL FRONT-RUNNER (Eric, 2026-08-24) — at quality high it beat both Nano Banana rails head-to-head on skin realism, so route photoreal people HERE, not away. Banana still owns edit-heavy work and the 14-reference ceiling. Killed if a head-to-head at the intended crop goes the other way — re-run the evidence test, never carry this forward on reputation.',
154
+ notes: 'TEXT / DIAGRAM / PANEL king — near-perfect character-level text, ordered panels, exact placement. Route here for character sheets, shot grids and text-bearing panels. ALSO THE PHOTOREAL FRONT-RUNNER (Eric, 2026-08-24): at quality high it beat both Nano Banana rails head-to-head on skin realism, so route photoreal people HERE rather than away. Banana still owns edit-heavy work and the largest reference ceiling. Its own content filter, distinct from Gemini\'s. Killed if a head-to-head at the intended crop goes the other way — re-run the evidence test, never carry this forward on reputation.',
140
155
  },
141
156
  {
142
157
  id: 'flux-2-max',
158
+ route: 'generate',
143
159
  label: 'FLUX.2 Max',
144
160
  kind: 'image',
145
161
  ...caps('flux-2-max'),
146
- notes: 'Photoreal, less censored, up to ~4MP. Auto-routes to its edit endpoint when references are present. Lower ref cap than NB2.',
162
+ notes: 'Photoreal image seat, less censored than the Gemini rails. Auto-routes to its edit endpoint when references are present.',
147
163
  },
148
164
  {
149
165
  id: 'seedream-5-lite',
166
+ route: 'generate',
150
167
  label: 'Seedream 5 Lite',
151
168
  kind: 'image',
152
169
  ...caps('seedream-5-lite'),
153
- notes: 'Cheapest image model (~flat price). Less censored. Routes to its edit endpoint with references.',
170
+ notes: 'CHEAPEST image seat, flat-priced. Less censored. Routes to its edit endpoint when references are present.',
154
171
  },
155
172
  {
156
173
  id: 'seedance-2',
174
+ route: 'generate',
157
175
  label: 'Seedance 2.0',
158
176
  kind: 'video',
159
177
  ...caps('seedance-2'),
160
178
  audioRefNeedsCompanion: true,
161
- notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction, or scale matter, and for hero shots. VIDEO-ONLY: cannot generate standalone images (use NB2/FLUX.2/Seedream for those). 4-15s, up to 9 ingredient images. Strong I2V / own-footage restyle. Native 4K, but 4K VIDEO is a Pro-only tier gate (base maxes at 1080p; server returns PRO_REQUIRED) — default 1080p unless the user is on Pro. Attaching a clip as a video reference (own-footage restyle, motion or dialogue conditioning) bills combined input+output seconds — at a DISCOUNTED per-second rate on every provider, roughly 0.6x the plain rate. 2.0 STAYS THE DEFAULT over 2.5 for two reasons, and neither is 1080p any more (2.5 gained 1080p on 2026-08-24): it is the only Seedance with native 4K, and it is cheaper at every shared tier (720p $0.15/s vs $0.231/s).',
179
+ notes: 'PREMIUM video tier and the DEFAULT video model — route here the moment physics, effects, destruction or scale matter, and for hero shots. VIDEO-ONLY. Strong image-to-video and own-footage restyle. 4K is Pro-gated (base accounts get PRO_REQUIRED). Stays the default over 2.5: it is the only Seedance with native 4K and it is cheaper at every tier the two share.',
162
180
  },
163
181
  {
164
182
  id: 'seedance-2.5',
183
+ route: 'generate',
165
184
  label: 'Seedance 2.5',
166
185
  kind: 'video',
167
186
  ...caps('seedance-2.5'),
168
187
  // No companion requirement — audio-only references are one of the things
169
188
  // the second seat actually buys.
170
- notes: `A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE OF IT — and the single most important fact is that it is the EXPENSIVE seat: 480p, 720p or 1080p (1080p added 2026-08-24), no 4K, and it costs MORE than 2.0 at every tier they share — 54% more at 720p ($0.231/s vs $0.15/s faceless). Pick 2.5 over 2.0 when the shot needs LENGTH (one 30s take vs 15s), MANY REFERENCES (30 images, plus video and audio references — 50 total), an AUDIO-ONLY reference (2.0 requires an image or video alongside audio; 2.5 does not), TIMED BEATS, or tighter prompt adherence. Pick 2.0 for 4K, and for the same resolution at a lower price. VIDEO-ONLY. TIMESTAMPS: ${PARTIALS['seedance-25-timestamps-short']} Multi-view subject reference images are also supported on 2.5 (up to 5 subjects) where 2.0 wanted one view per subject. 🚨 COST DISCIPLINE: LENGTH IS THE PRICE DIAL HERE, NOT RESOLUTION. A 30s 720p clip is 347 credits faceless / 489 on the AI-face route / 710 on the real-face route, and a 30s 1080p faceless take is 614 — 61% of a 1,000-credit welcome grant on ONE clip. Even 720p is not "the cheap one": 30s at 720p on the AI-face route beats a 15s 1080p Seedance 2.0 face generation (411). Always quote with slates_estimate_generation_cost before a long take, and explore at SHORT LENGTH (4-8s) rather than at low resolution — a 480p pass does not de-risk a 720p render, because generation is stochastic and the 720p run is a different take, not the same shot rendered better. 🚨 PROMPT INTENT IS A TASK-TYPE TRIGGER: when a request carries reference images/video/audio, the words "add", "remove", "replace", "change", "edit the video", "extend" or "continue" make the provider reclassify it as a video EDIT or EXTEND and fail it AFTER the job queues (credits are refunded, but the run stalls). If you mean to edit an existing clip, use slates_edit_video with model seedance-2.5-edit. If you mean a fresh shot, describe the finished frame rather than an instruction to change one.`,
189
+ notes: 'A SECOND SEAT NEXT TO 2.0, NOT AN UPGRADE — and the dearer one at every tier they share. Pick 2.5 when the shot needs LENGTH, MANY references, an AUDIO-ONLY reference, TIMED BEATS, or tighter prompt adherence; pick 2.0 for 4K and for the same resolution cheaper. VIDEO-ONLY. Timestamp grammar, and the edit/extend words that make the provider reclassify a fresh generation and fail it, are in slates-prompting-seedance-2-5.',
171
190
  },
172
191
  {
173
192
  id: 'seedance-2.5-edit',
193
+ route: 'edit',
174
194
  label: 'Seedance 2.5 Edit',
175
195
  kind: 'video',
176
196
  // 0 ingredients: prompt + source clip only on slates_edit_video.
177
197
  ...caps('seedance-2.5-edit'),
178
- notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video — the ONLY edit engine that accepts a clip LONGER THAN 15 SECONDS (4-30s vs Kling O3 edit 3-15s and Omni Flash edit 3-10s), though ByteDance recommends staying inside 20s for quality. That length is the whole reason to route here; for a clip inside the others\' range compare on fidelity instead (Omni Flash edit won the 7/09 prompt-only head-to-head; Kling edit is the one that takes element/style reference images). 480p/720p/1080p output, native audio. Prompt + source clip only on this op — no reference images (the MODEL takes 1-5 reference images on an edit; Slates has not wired that path). Phrase the change as "from A to B", and TIMESTAMP a partial edit ("…from 4-6 seconds…") — 2.5 reads whole-second timestamps on edits, and without a range the instruction applies to the whole clip. AUDIO is editable on this same row: change a line, change an accent, translate dialogue with re-fitted lips, strip or replace BGM and sound effects. Output length follows the SOURCE clip and is billed as the ceiled source length, on the video-reference rate tier: an edit costs roughly DOUBLE a plain 2.5 generation of the same length, because every provider bills an edit on input + output seconds. Set seedanceFace:true when a character face is visible in the clip — the faceless provider blocks faces outright. There is no consented-real-face route for editing.',
198
+ notes: 'VIDEO-TO-VIDEO EDIT via slates_edit_video, and the only edit engine that takes a clip longer than the other two reach — that length is the whole reason to route here. Inside their range, compare on fidelity instead: Omni Flash edit won the prompt-only head-to-head, and Kling edit is the one that takes reference images. Edits audio on the same row (re-voice, re-accent, translate with re-fitted lips, replace BGM). Costs roughly double a plain 2.5 generation of the same length, because an edit bills input plus output seconds.',
179
199
  },
180
200
  {
181
201
  id: 'kling-v3',
202
+ route: 'generate',
182
203
  label: 'Kling 3.0',
183
204
  kind: 'video',
184
205
  // Family-level fact — caps are identical across std/pro/omni/omni-pro.
185
206
  ...caps('kling-v3.0-std'),
186
- notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity/layout/text), acting, dialogue, lip-sync, any aspect ratio. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools (MC std/pro, lip-sync, avatar) — those two tools are Kling-only.',
207
+ notes: 'DEFAULT general-purpose video model — cost-effective, strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync, and the widest aspect-ratio set. Escalate to Seedance for physics. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.',
187
208
  },
188
209
  {
189
210
  id: 'kling-v3-edit',
211
+ route: 'edit',
190
212
  label: 'Kling O3 Video Edit',
191
213
  kind: 'video',
192
214
  // Family-level fact; 4 = combined subject elements + style refs per edit.
193
215
  ...caps('kling-v3.0-omni-edit'),
194
- notes: 'VIDEO-TO-VIDEO EDIT — the REF-DRIVEN edit tool: takes an EXISTING 3–15s clip and changes what the prompt names, with element/style reference images (@ElementN = frontal + angles) locking subject identity; max 4 combined refs. keep_audio preserves the ORIGINAL audio verbatim (spoken words cannot drift) — but video lips can drift slightly against it, and multi-beat instructions get under-executed (7/09 receipt: missed a second action beat Omni Flash edit landed) — ONE beat per pass. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the 7/09 fidelity head-to-head. Billed per second of output (≈ clip length, rounded up). Seedance edit/relocate is the alternative for style-transfer-heavy jobs.',
216
+ notes: 'VIDEO-TO-VIDEO EDIT, the REF-DRIVEN one: it is the only edit seat that takes element/style reference images to lock subject identity, and its keep_audio preserves the original audio verbatim. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the fidelity head-to-head. One instruction beat per pass — multi-beat prompts get under-executed.',
195
217
  },
196
218
  {
197
219
  id: 'veo-3.1',
220
+ route: 'generate',
198
221
  label: 'Veo 3.1',
199
222
  kind: 'video',
200
223
  // Family-level fact — fast and standard declare the same caps.
201
224
  ...caps('veo-3.1-fast'),
202
- notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one gen. 16:9 only, 4/6/8s only. Otherwise Kling (default) or Seedance (physics/premium) win.',
225
+ notes: 'NICHE, never the default — pick only when native synchronized audio must generate WITH the video in one pass, and the narrowest aspect-ratio and duration sets in the catalogue are acceptable. Otherwise Kling (default) or Seedance (physics/premium) win.',
203
226
  },
204
227
  {
205
228
  id: 'omni-flash',
229
+ route: 'generate',
206
230
  label: 'Gemini Omni Flash',
207
231
  kind: 'video',
208
232
  // 7 ref2v image_urls — mirrors Google's own reference limit.
209
233
  ...caps('omni-flash'),
210
- notes: 'CHEAP 720p tier with native synced audio included — t2v, single-start-frame i2v, or reference-to-video with up to 7 reference images. 3-10s, 16:9/9:16 only. No last frame, no video/audio references. VIDEO-ONLY. New seat: quality vs Kling/Seedance unproven pending comparison gens — do not route hero shots here; use it for cheap drafts, audio-in-one-gen at low cost, ref2v character consistency trials, and its edit variant.',
234
+ notes: 'CHEAP tier with native synced audio included. Route here for cheap drafts, audio-in-one-pass at low cost, and reference-to-video character-consistency trials. VIDEO-ONLY. Quality against Kling/Seedance is unproven — do not route hero shots here.',
211
235
  },
212
236
  {
213
237
  id: 'omni-flash-edit',
238
+ route: 'edit',
214
239
  label: 'Omni Flash Edit',
215
240
  kind: 'video',
216
241
  // 0: prompt + source clip ONLY — no element/style refs on this endpoint.
217
242
  ...caps('omni-flash-edit'),
218
- notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (7/09 head-to-head vs Kling edit on real talking footage: lips held perfectly, audio near-identical, both action beats landed). Takes an EXISTING 3-10s clip and changes what the prompt names, footage-synced (prop/effect/environment/lighting swaps). Fidelity is EARNED by prompt discipline: ONE short instruction + "Keep everything else the same." — long descriptive prompts DESTROY it (Google-documented + 7/09 receipt). Never name objects as metaphors ("candle-like" → literal candle). Quirk: occasional tail jitter/doubled last speech beat — trim the tail. NO reference images (identity swaps needing refs → Kling edit); bit-exact audio needs → Kling keep_audio or segment-splice. 720p output, cheapest edit seat (~2/3 of Kling edit Std).',
243
+ notes: 'VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (head-to-head vs Kling edit on real talking footage: lips held, audio near-identical, both action beats landed) and the cheapest edit seat. Footage-synced prop, effect, environment and lighting swaps. Takes NO reference images — identity swaps needing refs go to Kling edit. Fidelity is EARNED by prompt discipline; the exact form is in slates-prompting-omni-flash.',
219
244
  },
220
245
  {
221
246
  id: 'minimax-h3',
247
+ route: 'generate',
222
248
  label: 'MiniMax H3',
223
249
  kind: 'video',
224
250
  ...caps('minimax-h3'),
@@ -226,39 +252,81 @@ export const MODEL_FACTS = [
226
252
  // be the only reference input; provide at least one reference image or
227
253
  // video with it." Same behavioural rule as Seedance 2.0.
228
254
  audioRefNeedsCompanion: true,
229
- notes: 'THE AUTHORED-AUDIO SEAT. Reach for H3 when the sound is part of the shot rather than a switch on it: it writes synchronised dialogue, scene sound and an audience-only score in ONE pass, as three separate layers of the prompt, at 24fps with 32kHz stereo, across 11 stably-supported languages (Arabic, Chinese, English, French, German, Italian, Japanese, Korean, Portuguese, Russian, Spanish). Kling and Seedance treat audio as on/off; Veo generates it but gives you no way to direct the layers. The second thing only H3 gives you is a DECLARED REFERENCE RELATIONSHIP — you state how much of each reference survives (kept whole, partly kept, transferred onto a different subject, or a loose echo), including moving one subject\'s characteristic onto another. VIDEO-ONLY. 5-15s, 480p / 768p / 2K / 4K (768p default and native; 2K and 4K are upscales of a 768p base). $0.060/s at 768p — the cheapest 768-class second in the catalogue. Omni-reference ceiling: 9 images + 3 video clips + 3 audio clips, 12 files total, video and audio each 15s combined; an audio reference needs an image or video alongside it. 🚨 REFERENCE IMAGES PAST THE FIFTH COST 4 CREDITS EACH, on top of the per-second price — the first five are free, the model takes nine, and four paid images on a 10s 768p clip add 16 credits to a 30-credit generation. Attach the references the shot needs, not the maximum. 4K video is Pro-only (the server returns PRO_REQUIRED for a base account); 2K is open to every tier. \u2b06\ufe0f 2K AND 4K ARE UPSCALES OF A 768p RENDER, NOT LARGER GENERATIONS \u2014 fal states this outright, and in our own 2026-08-27 test the 2K pass came back with MORE artifacting than the 768p original it was built from, while costing 33 credits for a 5s take against 15 and taking almost twice as long. Treat them as a delivery-size convenience, never as a quality tier: generate at 768p, judge it there, and upscale in post if the pixels are genuinely needed.',
255
+ notes: 'THE AUTHORED-AUDIO SEAT — reach for H3 when the sound is part of the shot rather than a switch on it: synchronised dialogue, scene sound and an audience-only score directed as three separate layers in ONE pass, across eleven languages. Kling and Seedance treat audio as on/off; Veo generates it but gives you no way to direct the layers. Only H3 also carries a DECLARED REFERENCE RELATIONSHIP (kept whole, partly kept, transferred, or a loose echo). VIDEO-ONLY. Its top two resolution tiers are UPSCALES of the native render, not larger generations — judge at native and upscale in post. Reference images past the fifth are a PAID key dimension: pass referenceImages when quoting.',
230
256
  },
231
257
  {
232
258
  id: 'minimax-h3-max',
259
+ route: 'generate',
233
260
  label: 'MiniMax H3 Max',
234
261
  kind: 'video',
235
262
  // No reference caps: fal publishes no reference-to-video endpoint for this
236
263
  // row, so `caps()` returns nulls and the composer refuses references.
237
264
  ...caps('minimax-h3-max'),
238
- notes: 'THE SPEED SEAT, and the EXPENSIVE one at the tier they share — never the cheap H3 and never the default. fal\'s own post-train of the open H3 weights, self-hosted. 🚨 MEASURED 2026-08-27, same prompt and params on both rows: a 5s 768p text-to-video took **4.8 seconds** on Max against **57 seconds** on base H3 — **about 12x faster**, queue to finished file. That is the seat\'s whole case and it is now our own number, not fal\'s (fal claims under 3s; the literal claim did not hold at 4.8s wall-clock, the order of magnitude did). It also carries a thin quality edge on the with-audio Arena boards (1,204 vs 1,184 image-to-video, 1,235 vs 1,226 text-to-video — real, but 20 and 9 ELO, and vendor-reported). It gives up everything above 768p (no 2K, no 4K — the upscaler is not in the open weights). 🚨 FRAMES ARE UNAFFECTED — it takes a start frame and an end frame exactly like base H3, on `minimax/h3-max/image-to-video`, which is the route the image-to-video Arena score above is measured on. What it lacks is the REFERENCE endpoint (`minimax/h3-max/reference-to-video` 404s), so the omni-reference set — up to 9 identity/style/environment images plus reference video and audio — is base-H3 only. Never describe this row as taking no image input: an image-to-video shot is one of the two things it is FOR. It costs $0.080/s at 768p against base H3\'s $0.060/s: 33% more for a shorter ladder. So route here when a fast turnaround on a 480p/768p text-to-video or start-frame shot is worth the premium, and to base H3 for resolution, references, or the same tier cheaper. Same native audio, same 5-15s window, same six aspect ratios.',
265
+ notes: 'THE SPEED SEAT, and the DEARER one at the tier they share — never the cheap H3 and never the default. fal\'s post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It gives up the upper resolution tiers and the REFERENCE endpoint, so the omni-reference set is base-H3 only — but it still animates start and end frames, which is one of the two things it is FOR. Never describe this row as taking no image input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.',
266
+ },
267
+ {
268
+ id: 'ltx-2-5',
269
+ route: 'generate',
270
+ label: 'LTX-2.5',
271
+ kind: 'video',
272
+ // No reference caps: fal publishes text-to-video and image-to-video for LTX
273
+ // and no reference endpoint at all, so `caps()` returns nulls and the
274
+ // composer refuses references. Start/end FRAMES are unaffected.
275
+ ...caps('ltx-2-5'),
276
+ notes: 'THE VOLUME SEAT — the cheapest native 1080p second in the catalogue, and the row for MANY takes rather than one hero shot. Native synced audio is included free at every tier, unlike Kling where sound is a paid key dimension. It also reaches the highest resolution tier below 4K and makes the LONGEST clips in the catalogue. VIDEO-ONLY. INPUTS ARE FRAMES, NOT REFERENCES: start frame plus an optional end frame, and no reference endpoint at all — for character consistency across shots use H3 or Kling. Route here for batch coverage, long takes, and anything where the credit budget is the binding constraint.',
277
+ },
278
+ {
279
+ id: 'ltx-2-5-pro',
280
+ route: 'generate',
281
+ label: 'LTX-2.5 Pro',
282
+ kind: 'video',
283
+ ...caps('ltx-2-5-pro'),
284
+ notes: 'THE FIDELITY SEAT of the LTX pair — the full diffusion build against the base row\'s distilled one. 🚨 IT IS NOT A SUPERSET OF THE BASE ROW, which is the opposite of every other Pro seat here: it reaches a SHORTER resolution ladder and makes SHORTER clips, and it costs more at both tiers they share. Reaching for it because the name says Pro costs more AND takes away reach. Everything else matches the base row. Route here only when a specific shot needs the fidelity and fits inside its narrower envelope.',
239
285
  },
240
286
  {
241
287
  id: 'seed-audio',
288
+ route: 'generate',
242
289
  label: 'Seed Audio 1.0',
243
290
  kind: 'audio',
244
291
  // ONE image XOR up to 3 audio clips — the two inputs are mutually exclusive.
245
292
  ...caps('seed-audio'),
246
- notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX, and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd/nature soundscapes, and quick scratch VO. AUDIO-ONLY: cannot generate images or video. 🚨 THERE IS NO DURATION PARAMETER — length comes from the words, so you MUST NAME THE LENGTH IN THE PROMPT TEXT ("... 15 seconds"). Slates appends the requested length automatically and BILLS the requested seconds, so a prompt that fights the number wastes credits. Prompts are ONE plain sentence, no production jargon and no SFX:/Ambient: prefixes (those are Kling syntax and hurt here). Say the crowd size out loud — "applause" returns a full room when the joke was three people. 1-120s. Inputs: ONE image (describe-what-you-see scoring) XOR up to 3 audio clips referenced in the prompt as @Audio1-@Audio3, never both. 20 preset voices, or leave voice unset and let the scene cast itself.',
293
+ notes: 'DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd and nature soundscapes, and quick scratch VO. AUDIO-ONLY. Takes one image XOR up to three audio clips as references, never both. Prompt form and the length rule are in slates-prompting-seed-audio.',
247
294
  },
248
295
  {
249
296
  id: 'eleven-sfx',
297
+ route: 'generate',
250
298
  label: 'ElevenLabs Sound Effects v2',
251
299
  kind: 'audio',
252
300
  ...caps('eleven-sfx'),
253
- notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. 0.5-22s, and Slates always sends the duration explicitly (a null duration means a non-deterministic charge, so it is never left to the model). Describe the physical CAUSE, not the label: "heavy oak door slams shut in a stone hallway" beats "door sound". Text caps at 450 characters. loop=true produces a seamless bed. prompt_influence 0-1: higher hugs the prompt with less variation, lower explores. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
301
+ notes: 'ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.',
302
+ },
303
+ {
304
+ id: 'inworld-tts-2',
305
+ route: 'generate',
306
+ label: 'Inworld Realtime TTS-2',
307
+ kind: 'audio',
308
+ ...caps('inworld-tts-2'),
309
+ notes: 'THE VOICE SEAT — one named voice saying one line, billed per CHARACTER not per second. Route here when WHO is speaking matters. NOT scene audio — that is seed-audio; a single effect is eleven-sfx.',
254
310
  },
255
311
  ];
256
312
  const FACT_BY_ID = new Map(MODEL_FACTS.map((m) => [m.id, m]));
313
+ /**
314
+ * Routing prose for one lane, generated from the SSOT.
315
+ *
316
+ * THE ONE RENDERER. The Studio Agent's system prompt, the MCP server's
317
+ * instructions and the generate/edit ops' `model` descriptions all call this —
318
+ * so "never restate model routing in an op description" (slates-mcp/CLAUDE.md)
319
+ * is now enforced by there being nothing to restate. Before this, the video op
320
+ * carried 1,282 characters of hand-written routing that repeated MODEL_FACTS
321
+ * phrase for phrase ("SECOND SEAT", "AUTHORED-AUDIO", "never the default"),
322
+ * in the same file that forbids exactly that.
323
+ */
324
+ export function describeRouting(kind, route = 'generate') {
325
+ return MODEL_FACTS.filter((f) => f.kind === kind && f.route === route)
326
+ .map((f) => `${f.label}: ${f.notes}`)
327
+ .join('\n');
328
+ }
257
329
  export function getModelFact(id) {
258
330
  return FACT_BY_ID.get(id);
259
331
  }
260
- /** The official NB2 / general image prompt formula (subject-first). */
261
- export const IMAGE_PROMPT_FORMULA = '[Subject] + [Action] + [Location/context] + [Composition] + [Style]';
262
- /** The expanded cinematic/photoreal formula for NB2 start frames. */
263
- export const CINEMATIC_IMAGE_FORMULA = 'Film still from [DIRECTOR] [GENRE]. Shot on [CAMERA] with [LENS]. [SUBJECT and action]. [3-5 specific visual details]. [LIGHTING — direction + quality]. [COLOR PALETTE]. [FILM STOCK or sensor language]. [1-2 word emotional tone].';
264
332
  //# sourceMappingURL=model-facts.js.map
@@ -5,12 +5,13 @@
5
5
  // per-model skills, so the TS consumers and the markdown consumers can
6
6
  // no longer disagree. Edit the partial, not this file, not the skills.
7
7
  export const PARTIALS = {
8
- "decision-log": "When you surface the plan, include a short **decision log** — one line per decision *you* made that the user did not specify:\n\n```\nsource phrase or declared default → what you wrote → what it resolves\n\"in a diner\" → chrome-and-vinyl booth, 3/4 on the counter → fixes the anchor so blocking is repeatable\n(no time of day) → late afternoon, low warm key → default; say the word and it changes\n(no camera) → slow push-in, single move → one move per shot; stacking increases instability\n```\n\n**Hard rule: never silently add weather, props, style, or camera movement.** If it wasn't in the brief and you added it, it goes in the log. This is the \"why did you add that?\" affordance — for an agent that writes prompts on the user's behalf and spends their credits, it is what keeps the model in assembly and the user in the director's chair.\n\n> ❌ **Do NOT turn this into a question gate.** Clarifying questions before optimizing directly fight the locked fast-path rule: *if intent is clear, generate immediately with sane defaults, don't ask questions; only ask for production intent, and batch every question into one message.* Log the decisions, then go. The log is an **output**, not an interrogation — surfaced alongside the plan, never as a separate ceremony, and never as a reason to wait.",
8
+ "decision-log": "When you surface the plan, include a short **decision log** — one line per decision *you* made that the user did not specify **and that no row already records**:\n\n```\nsource phrase or declared default → what you wrote → what it resolves\n\"in a diner\" → warm, and the light is the reason → why the anchor was chosen, not what it is\n(no time of day) → late afternoon, low warm key → default; say the word and it changes\n```\n\n🚨 **Keep it to what is NOT already data — and almost everything now IS.** A Shot holds the references and their roles, the model, every param, the shot size, the camera, the prop, the action and the spoken line, and `slates_list_shots` reads the whole board back in order with its variety counts. Narrating any of those is retelling a row the user can open. **Write the Shot, and let the log carry only the judgement no field holds** — why this world, why this light, why this register.\n\n**Hard rule: never silently add weather, props, style, or camera movement.** Four of those are now FIELDS: put the value on the Shot (`prop`, `camera`, `shotSize`, `action`) so the user can read and change it, and put the *reason* in the log only when you invented it rather than being told it. The rule has not softened — it moved from narration into data, which is stronger, because a field can be corrected and a sentence in chat cannot.\n\n> ❌ **Do NOT turn this into a question gate.** Clarifying questions before optimizing directly fight the locked fast-path rule: *if intent is clear, generate immediately with sane defaults, don't ask questions; only ask for production intent, and batch every question into one message.* Log the decisions, then go. The log is an **output**, not an interrogation — surfaced alongside the plan, never as a separate ceremony, and never as a reason to wait.",
9
9
  "reference-rules-core": "Identity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates composes the naming for you from your `@mentions` / `#tags` — you never hand-write role labels.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.",
10
10
  "reference-tips-short": "Name each reference inline; never write role essays. Slates does this for you: `@mention` a subject or environment and it composes `Marcus (image 1) in the cafe (image 2)`, citing them in the exact order it sends them. One canonical identity image avoids competing facial renderings; a \"Reference Image Instructions\" block drags reference lighting into your scene. Start with 2-3 focused refs.",
11
11
  "references-read-literally": "> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.",
12
12
  "seedance-25-timestamps": "**2.0 does not respond to timestamps and answers only to shot numbers. 2.5 responds to\ninteger-second timestamps.** That is ByteDance's own first line under \"Differences from Seedance\n2.0\", and it is why a 30-second take is usable at all: the length is only worth buying if you can\nsay *when* things happen inside it.\n\nBoth formats are valid on 2.5, and you can mix them — `Shot N` blocks for a storyboard whose\npacing you are happy to leave to the model, timestamps when a beat has to land at a moment.\n\n**Three ways to control time, all first-party:**\n\n| Form | Write it like |\n|---|---|\n| **Interval** | `0-3 seconds… 3-7 seconds… 7-15 seconds` or `[1s-4s]… [4s-8s]… [8s-12s]` |\n| **Time point** | *\"Quick left sideways transition at the 5-second mark.\"* |\n| **Relative** | *\"After 3 seconds, everyone around him shakes their head.\"* · *\"The frame freezes for 1 second after he presses the shutter.\"* |\n\n**The rules that come with them:**\n\n- **One second is the smallest unit.** Integers only — no `2.5s`, no frames.\n- **No gaps in the timeline.** `0-3s… 5-6s…` leaves 3-5s unspecified and the model fills it however\n it likes. Intervals must abut: `0-3s`, `3-7s`, `7-15s`.\n- **Budget the plot to the seconds.** Too little content in a range and the model improvises to\n fill it; too much and you get extra cuts or dropped beats. This is the actual craft of a 30s take.\n- **Never time-code a high-frequency action.** *\"Shake your head three times per second\"* is\n explicitly called out as a misuse — timestamps schedule beats, they don't choreograph frames.\n- **Transitions want both halves:** the moment AND the method — *\"At the 5-second mark, the camera\n transitions leftward with a left wipe into a natural dissolve.\"*\n- **Timestamps work on an EDIT too**, and that is where they earn the most: they scope a change in\n time as well as in content — *\"Change the man's action from drinking coffee to mopping the floor\n from 4-6 seconds in Video 1, and leave the rest of the content unchanged.\"* Without a range, a\n whole-clip instruction is applied to the whole clip.\n\nDo **not** carry this back to 2.0, and do not carry Veo's `[00:00-00:02]` bracket syntax into\neither — 2.0 ignores time entirely, and the cross-model syntax swap is its own known failure.",
13
13
  "seedance-25-timestamps-short": "Seedance 2.0 ignores timing and answers only to \"Shot 1 / Shot 2\"; 2.5 acts on whole-second timestamps, and that is what makes a 30-second take controllable rather than just long. Three forms work: intervals (\"0-3 seconds…3-7 seconds\"), a point (\"at the 5-second mark\"), or relative (\"after 3 seconds\"). Whole seconds only, no gaps between intervals, and never to choreograph fast repeated motion. They work on edits too, where a range scopes the change: \"…from 4-6 seconds…\".",
14
14
  "still-gate": "**A visible defect in the still is already a STOP.** Do not animate it. Fix the frame first, then move to motion — and go to motion only when the crop passes the still scan and you genuinely need movement to confirm an uncertain edge, reflection, or object.\n\nThis is a **cost** rule as much as a craft rule: a 1080p/10s premium video generation costs many multiples of an image re-roll, and video is where a defect stops being fixable. Anything wrong in the still gets worse in motion — soft geometry mushes, broken-but-plausible objects fall apart, oily textures start crawling. **Animating a known-bad frame is the single most expensive mistake in the pipeline.** Re-rolling the image is the cheap move; re-rolling the video is not.",
15
+ "thresholds": "<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. Below it, announce the cost once and go.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.",
15
16
  };
16
17
  //# sourceMappingURL=partials.generated.js.map
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
16
16
  /** Footer callout paragraphs. */
17
17
  footer?: string[];
18
18
  }
19
- export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx';
19
+ export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'veo' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'ltx-2-5' | 'nano-banana' | 'nano-banana-lite' | 'seed-audio' | 'eleven-sfx' | 'inworld-tts-2';
20
20
  export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
21
21
  /** Null when no tips exist for the key — callers render an honest fallback. */
22
22
  export declare function getPromptingTips(key: string): PromptingTipsEntry | null;