@slatesvideo/shared 0.6.11 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/dist/auth.js +2 -2
  2. package/dist/clients/cloud.js +1 -1
  3. package/dist/index.d.ts +1 -1
  4. package/dist/index.js +1 -1
  5. package/dist/manual/content.d.ts +1 -1
  6. package/dist/manual/content.js +1 -1
  7. package/dist/operations/index.d.ts +817 -16
  8. package/dist/operations/index.js +1413 -360
  9. package/dist/operations/surface.d.ts +4 -1
  10. package/dist/operations/surface.js +41 -10
  11. package/dist/prompts/ad-presets.d.ts +77 -0
  12. package/dist/prompts/ad-presets.js +43 -0
  13. package/dist/prompts/agent-doctrine.js +27 -5
  14. package/dist/prompts/banned-tokens.d.ts +4 -29
  15. package/dist/prompts/banned-tokens.js +29 -204
  16. package/dist/prompts/craft-cards.js +2 -2
  17. package/dist/prompts/generation-policy.d.ts +41 -0
  18. package/dist/prompts/generation-policy.js +53 -0
  19. package/dist/prompts/guide-retrieval.d.ts +9 -0
  20. package/dist/prompts/guide-retrieval.js +53 -0
  21. package/dist/prompts/index.d.ts +1 -0
  22. package/dist/prompts/index.js +1 -0
  23. package/dist/prompts/model-capabilities.d.ts +18 -1
  24. package/dist/prompts/model-capabilities.js +72 -19
  25. package/dist/prompts/model-facts.d.ts +34 -2
  26. package/dist/prompts/model-facts.js +66 -5
  27. package/dist/prompts/partials.generated.js +8 -2
  28. package/dist/prompts/prompting-tips.d.ts +1 -1
  29. package/dist/prompts/prompting-tips.js +61 -16
  30. package/dist/prompts/reference-composer.d.ts +2 -0
  31. package/dist/prompts/reference-composer.js +51 -50
  32. package/dist/prompts/script-document.d.ts +165 -0
  33. package/dist/prompts/script-document.js +11 -0
  34. package/dist/prompts/shot-grammar.d.ts +4 -4
  35. package/dist/prompts/shot-grammar.js +3 -3
  36. package/dist/prompts/shot-spec.d.ts +13 -0
  37. package/dist/prompts/shot-spec.js +23 -5
  38. package/dist/skills/content.js +27 -24
  39. package/exports/slates-chatgpt-images/generated/SKILL.md +107 -0
  40. package/exports/slates-chatgpt-images/generated/slates-chatgpt-images.skill +0 -0
  41. package/exports/slates-prompt-builder/generated/SKILL.md +1 -1
  42. package/exports/slates-prompt-builder/generated/reference-character.md +9 -1
  43. package/exports/slates-prompt-builder/generated/reference-kling.md +3 -3
  44. package/exports/slates-prompt-builder/generated/reference-nano-banana.md +22 -10
  45. package/exports/slates-prompt-builder/generated/reference-seedance.md +4 -4
  46. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +17 -17
  47. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  48. package/package.json +9 -3
  49. package/skills/_partials/cinematic-card.md +8 -0
  50. package/skills/_partials/cinematic-routes-short.md +2 -0
  51. package/skills/_partials/cinematic-tips-short.md +2 -0
  52. package/skills/_partials/decision-log.md +1 -13
  53. package/skills/_partials/image-defaults.md +11 -0
  54. package/skills/_partials/lens-video-split.md +1 -0
  55. package/skills/_partials/reference-rules-core.md +1 -1
  56. package/skills/_partials/sheet-tool-defaults.md +6 -0
  57. package/skills/slates-character-identity.md +9 -1
  58. package/skills/slates-chatgpt-images.md +107 -0
  59. package/skills/slates-cinematic-look.md +237 -0
  60. package/skills/slates-cost-discipline.md +18 -12
  61. package/skills/slates-direct-response-ad.md +13 -53
  62. package/skills/slates-edit-and-iterate.md +1 -1
  63. package/skills/slates-model-selection.md +20 -14
  64. package/skills/slates-one-prompt-film.md +19 -77
  65. package/skills/slates-project-organization.md +7 -3
  66. package/skills/slates-prompting-flux-2-max.md +15 -4
  67. package/skills/slates-prompting-gpt-image-2-5.md +41 -28
  68. package/skills/slates-prompting-inworld-tts.md +174 -174
  69. package/skills/slates-prompting-kling-v3.md +3 -3
  70. package/skills/slates-prompting-lip-sync.md +1 -1
  71. package/skills/slates-prompting-minimax-h3.md +30 -17
  72. package/skills/slates-prompting-motion-transfer.md +1 -1
  73. package/skills/slates-prompting-nano-banana-2.md +24 -11
  74. package/skills/slates-prompting-seedance-2-5.md +7 -6
  75. package/skills/slates-prompting-seedance.md +5 -5
  76. package/skills/slates-prompting-seedream-5-lite.md +14 -3
  77. package/skills/slates-prompting-veo-3.md +1 -1
  78. package/skills/slates-script-craft.md +45 -0
  79. package/skills/slates-shot-variety.md +11 -40
  80. package/skills/slates-storyboard-from-script.md +14 -66
  81. package/skills/slates-style-prompting.md +4 -4
  82. package/skills/slates-ugc-influencer-ad.md +32 -309
  83. package/skills/slates-vision-feedback-loop.md +2 -1
@@ -137,7 +137,7 @@ const SEEDANCE_25 = {
137
137
  [
138
138
  {
139
139
  heading: '720p is not the cheap one here',
140
- example: '30s \u00b7 720p \u00b7 Face route = 484 credits\n15s \u00b7 1080p \u00b7 Seedance 2.0 Face = 411 credits',
140
+ example: '30s \u00b7 720p \u00b7 Face route = 489 credits\n15s \u00b7 1080p \u00b7 Seedance 2.0 Face = 411 credits',
141
141
  note: 'Length is what moves the price, and 2.5 doubles the length ceiling — so a 30-second 720p clip can cost more than a 15-second 1080p one, against a 1,000-credit starting balance. Explore at short LENGTH rather than low resolution: cut the seconds to 4-8 while you are finding the shot, and stay at the resolution you actually want. A 480p pass does not de-risk a 720p render — generation is stochastic, so the 720p run is a different take, not the same shot rendered better. The Generate button always shows the exact number first.',
142
142
  critical: true,
143
143
  },
@@ -151,7 +151,7 @@ const SEEDANCE_25 = {
151
151
  ],
152
152
  footer: [
153
153
  '30 image references is a budget, not a target — 2-4 strong references still beat both extremes, one per role. ByteDance\'s own ceilings for 2.5: 1-8 subjects bound by image reference stay stable (9-12 works but needs re-rolls), 1-5 subjects bound by video or audio reference, and 5-10 seconds is the sweet spot for a reference clip. Unlike 2.0, a multi-view turnaround sheet can be a single subject reference here — past 5 subjects, go back to one view per image. The larger budget is for long multi-shot takes and for video plus audio references alongside images.',
154
- 'A reference VIDEO bills input seconds PLUS output seconds, and 2.5 accepts references up to 30s combined — so a 20-second reference driving a 20-second output bills 40 seconds. The Generate button shows the total.',
154
+ 'A reference VIDEO bills input seconds PLUS output seconds, and 2.5 accepts references up to 30s combined — so a 20-second reference driving a 20-second output bills 40 seconds. On the AI-face route the reference counts as at least as long as the output: a 5-second reference on a 20-second output bills 40 seconds, not 25. The Generate button shows the total.',
155
155
  ...(SEEDANCE.footer ?? []).slice(0, 2),
156
156
  'Frames and reference images stay mutually exclusive, and on a first/last-frame generation Seedance 2.5 chooses the aspect ratio itself — the ratio control shows "Adaptive" because the start frame decides the shape.',
157
157
  ],
@@ -190,7 +190,7 @@ const SEEDANCE_25_EDIT = {
190
190
  [
191
191
  {
192
192
  heading: 'When to use it instead of the others',
193
- note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner and the cheapest seat, and Kling O3 Edit is the one that takes subject and style reference images.',
193
+ note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner and the cheapest option, and Kling O3 Edit is the one that takes subject and style reference images.',
194
194
  },
195
195
  {
196
196
  heading: 'It edits the audio too',
@@ -347,7 +347,7 @@ const OMNI_FLASH = {
347
347
  note: 'No negative-prompt field — write what to avoid as a direct instruction.',
348
348
  },
349
349
  {
350
- heading: 'Know its seat',
350
+ heading: 'Know its role',
351
351
  note: 'Cheap drafts, iteration volume, and audio-in-one-gen at low cost. For hero shots, Seedance 2.5 (the default) or Seedance 2.0 (4K, cheaper) still win.',
352
352
  },
353
353
  ],
@@ -398,6 +398,45 @@ const OMNI_FLASH_EDIT = {
398
398
  'Ship via segment-splice: edit only the seconds where the change happens (Trim / Split first), then splice back over the original on the timeline with the original audio underneath. Chain edits one change at a time — each edit saves as a new clip linked to its parent.',
399
399
  ],
400
400
  };
401
+ const GPT_IMAGE_25 = {
402
+ label: 'GPT Image 2.5',
403
+ intro: [
404
+ 'Two tiers at the same price: Flare is the faster one; Sunburst is the better-quality one, and slower. Explore on Flare, finish on Sunburst.',
405
+ 'The photoreal front-runner for people, and the most reliable model for readable text, ordered panels and exact placement.',
406
+ ],
407
+ columns: [
408
+ [
409
+ {
410
+ heading: 'Name each reference where it is used',
411
+ example: 'The woman from image 1 cooks on a rocky summit, lit and graded like image 2.',
412
+ note: PARTIALS['reference-tips-short'],
413
+ },
414
+ {
415
+ heading: 'Quote text that must render',
416
+ example: 'the word "SLATES" once in small plain letters on the left chest',
417
+ note: "Quoted strings render most reliably. Describe a font's feel, never its name, and keep on-image text under about 30 words.",
418
+ },
419
+ {
420
+ heading: 'Set the quality tier on purpose',
421
+ example: 'low · medium · high · xhigh · max',
422
+ note: 'High is the everyday tier and max costs about four times as much. Medium is cheap enough to draft on; go past high only when tiny type or a finished frame needs it.',
423
+ },
424
+ ],
425
+ [
426
+ {
427
+ heading: 'The look: describe what the camera sees',
428
+ example: 'She is close to a silhouette: her face falls into deep shadow.\nThe sky around the sun burns out to white.',
429
+ note: PARTIALS['cinematic-tips-short'],
430
+ critical: true,
431
+ },
432
+ {
433
+ heading: 'Describe the frame, or swap into one you own',
434
+ example: 'Dark hiking trousers. The only things on the rock are the stove and the pan.\nTake image 1 and change only the character to the character in image 2.',
435
+ note: PARTIALS['cinematic-routes-short'],
436
+ },
437
+ ],
438
+ ],
439
+ };
401
440
  const NANO_BANANA = {
402
441
  label: 'Nano Banana 2',
403
442
  intro: [
@@ -414,7 +453,7 @@ const NANO_BANANA = {
414
453
  {
415
454
  heading: 'Named lenses + apertures',
416
455
  example: '85mm f/1.4 · 135mm f/2.8 · 50mm f/1.2 · 35mm f/2 · Panavision anamorphic · 400mm telephoto',
417
- note: '135mm f/2.8 is the cheat code for skin texture and intimate compression. Anamorphic for cinematic width + horizontal flares.',
456
+ note: 'Describe the intended perspective, depth of field and texture alongside the lens. A model does not guarantee physical lens simulation.',
418
457
  },
419
458
  {
420
459
  heading: 'Named film stocks (one per prompt)',
@@ -424,7 +463,7 @@ const NANO_BANANA = {
424
463
  {
425
464
  heading: "Don't carry lens + stock into a video prompt",
426
465
  example: '85mm f/1.4, Portra 400\n→ close-up, shallow depth of field, warm natural colors, cinematic texture',
427
- note: 'Lenses, apertures, film stocks and camera bodies are an image-model lever and a video-model anti-pattern — ByteDance\'s Seedance guide never mentions f-stops, lens millimetres, fps or shutter angle. When you animate a frame you made here, translate the look into shot size, depth of field and colour tone instead of pasting the gear list across.',
466
+ note: PARTIALS['lens-video-split'],
428
467
  },
429
468
  {
430
469
  heading: 'Physics-based lighting',
@@ -434,14 +473,19 @@ const NANO_BANANA = {
434
473
  {
435
474
  heading: 'Imperfection vocabulary',
436
475
  example: 'visible pores · peach fuzz · ISO noise · sweat beading · slight hyperpigmentation · unretouched raw photography',
437
- note: 'Forces the model away from AI-clean skin. The default is too smooth — you have to ask for the imperfections that real photos have.',
476
+ note: 'Forces the model away from AI-clean skin. The default is too smooth — you have to ask for the imperfections that real photos have. Lead with the kind of photograph and the conditions on the skin; a bare list of flaw words reads as tokens.',
477
+ },
478
+ {
479
+ heading: 'The look: describe what the camera sees',
480
+ example: 'She is close to a silhouette: her face falls into deep shadow.\nThe sky around the sun burns out to white.',
481
+ note: PARTIALS['cinematic-tips-short'],
438
482
  },
439
483
  ],
440
484
  [
441
485
  {
442
486
  heading: '❌ The anti-list — avoid these',
443
487
  example: '8k · masterpiece · hyperrealistic · ultra-detailed · trending on ArtStation · perfect skin · flawless · airbrushed · cinematic (alone)',
444
- note: 'Tag-soup phrases from the Stable-Diffusion era. Measured success ~60-70% with these vs ~95%+ with positive description. Always specify which cinema — director, lens, era, stock.',
488
+ note: 'Generic quality tags do not specify an observable result. Describe the medium, light, exposure and texture the brief calls for.',
445
489
  critical: true,
446
490
  },
447
491
  {
@@ -537,7 +581,7 @@ const SEED_AUDIO = {
537
581
  note: 'Up to 3 audio clips (max 30s each), referenced as @Audio1–@Audio3 — OR one image to score what is in frame. Never both in the same generation.',
538
582
  },
539
583
  {
540
- heading: 'Know its seat',
584
+ heading: 'Know its role',
541
585
  note: 'Scenes, beds, room tone and dialogue in one pass. For a single effect that has to land on a specific frame, use Sound Effects.',
542
586
  },
543
587
  ],
@@ -579,7 +623,7 @@ const ELEVEN_SFX = {
579
623
  note: 'Higher hugs your wording with less variation between takes; lower explores. Raise it when a re-roll keeps wandering off the brief.',
580
624
  },
581
625
  {
582
- heading: 'Know its seat',
626
+ heading: 'Know its role',
583
627
  note: 'One precise effect on a known frame. Full rooms and layered scenes are cheaper and better in one Seed Audio pass.',
584
628
  },
585
629
  ],
@@ -589,7 +633,7 @@ const MINIMAX_H3 = {
589
633
  label: 'MiniMax H3',
590
634
  intro: [
591
635
  'MiniMax H3 generates picture and sound in one pass — 24fps, 32kHz stereo, 5-15 seconds, 11 stably-supported languages. It is the only video model in Slates where audio is AUTHORED rather than switched on: synchronised dialogue and action sounds go in the body of the prompt, ambience goes in a soundscape section, and audience-only music goes in a score section. Put a sound in the wrong section and it is dropped, doubled, or attributed to the wrong source.',
592
- 'Two seats that differ in LADDER and PRICE, not in what they accept. Base H3 runs 480p / 768p / 2K / 4K; H3 Max is fal\'s faster post-train and runs 480p / 768p / 1080p, dearer than base H3 at the tier they share - a deliberate speed pick, never the cheap one. BOTH read up to 9 reference images plus 3 video and 3 audio clips (12 files total, and audio never travels alone), and both animate a start frame and an end frame. 768p is the default on both because it is the tier the model natively generates; base H3\'s 2K and 4K are upscales of a 768p base. Reference images past the free allowance are billed and the allowances DIFFER: 5 free on base H3, 4 on Max.',
636
+ 'Three models. Base H3 runs 480p / 768p / 2K / 4K; H3 Max is fal\'s faster post-train and runs 480p / 768p / 1080p, dearer than base H3 at the tier they share - a deliberate speed pick, never the cheap one; H3 Max Turbo has Max\'s ladder at half Max\'s rate and takes NO references. Base H3 and Max read up to 9 reference images plus 3 video and 3 audio clips (12 files total, and audio never travels alone); all three animate a start frame and an end frame. 768p is the default on all three because it is the tier the model natively generates; base H3\'s 2K and 4K are upscales of a 768p base, and 1080p on Max and Turbo is a refinement of it. Reference images past the free allowance are billed and the allowances DIFFER: 5 free on base H3, 4 on Max.',
593
637
  ],
594
638
  columns: [
595
639
  [
@@ -624,7 +668,7 @@ const MINIMAX_H3 = {
624
668
  {
625
669
  heading: 'Say how much of a reference survives',
626
670
  example: 'Give the man in image 3 the weathered leather texture of the jacket in image 4.',
627
- note: 'H3 is the only seat that understands transferring a characteristic onto a DIFFERENT subject. State each reference\'s job and how much of it should carry through — kept whole, kept in part, transferred, or a loose echo.',
671
+ note: 'H3 is the only model that understands transferring a characteristic onto a DIFFERENT subject. State each reference\'s job and how much of it should carry through — kept whole, kept in part, transferred, or a loose echo.',
628
672
  },
629
673
  {
630
674
  heading: 'Reference images past the fifth cost extra',
@@ -650,7 +694,7 @@ const LTX_2_5 = {
650
694
  label: 'LTX-2.5',
651
695
  intro: [
652
696
  'LTX-2.5 scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE, not the last. Lightricks ranks the six parts of a prompt in this order: sound, camera, character detail, shot type and scene, then scene dressing — and scene dressing is the first thing to cut when a prompt sprawls. Everything goes in ONE flowing paragraph, not a list of labelled sections.',
653
- 'Two seats. Base LTX-2.5 is the distilled build: 720p / 1080p / 1440p / 4K and clips from 6 to 20 seconds, and it is the cheapest native 1080p second in Slates. LTX-2.5 Pro is the full diffusion build ("Diffusion Fidelity Rendering" spends extra compute on busy frames) but reaches a SHORTER ladder — 1080p and 10 seconds maximum — while costing about a third more. Pro is for a dense final render; base is for iteration, long takes and 4K.',
697
+ 'Two models. Base LTX-2.5 is the distilled build: 720p / 1080p / 1440p / 4K and clips from 6 to 20 seconds, and it is the cheapest native 1080p second in Slates. LTX-2.5 Pro is the full diffusion build ("Diffusion Fidelity Rendering" spends extra compute on busy frames) but reaches a SHORTER ladder — 1080p and 10 seconds maximum — while costing about a third more. Pro is for a dense final render; base is for iteration, long takes and 4K.',
654
698
  ],
655
699
  columns: [
656
700
  [
@@ -706,14 +750,14 @@ const LTX_2_5 = {
706
750
  ],
707
751
  footer: [
708
752
  'Frames, not references. LTX takes a start frame and an optional end frame (which generates a transition between the two) — it has no reference endpoint at all, so identity, style and environment reference images are not available on this model. For character consistency across separate shots, use MiniMax H3 or Kling.',
709
- 'Aspect ratios are 16:9 and 9:16 only, and native audio is included free at every resolution — there is no sound surcharge on either seat.',
753
+ 'Aspect ratios are 16:9 and 9:16 only, and native audio is included free at every resolution — there is no sound surcharge on either model.',
710
754
  'In image-to-video, do not cut away from the opening frame too early: you have paid for that frame, so let it play before the first move.',
711
755
  ],
712
756
  };
713
757
  const INWORLD_TTS = {
714
758
  label: 'Inworld TTS-2',
715
759
  intro: [
716
- 'The voice seat: one named voice saying one line. Unlike every other surface in Slates, the prompt is not a description of what you want — it IS the words that get spoken, verbatim, and its length is what you are billed for.',
760
+ 'The voice model: one named voice saying one line. Unlike every other surface in Slates, the prompt is not a description of what you want — it IS the words that get spoken, verbatim, and its length is what you are billed for.',
717
761
  'A voice belongs to a character, the same way a face does. Build it once from a clip or a description, then send it lines.',
718
762
  ],
719
763
  columns: [
@@ -761,7 +805,7 @@ const INWORLD_TTS = {
761
805
  note: 'Numbers, dates and abbreviations are read literally. Write them as they should sound.',
762
806
  },
763
807
  {
764
- heading: 'Know its seat',
808
+ heading: 'Know its role',
765
809
  note: 'One voice, cleanly. Dialogue mixed with effects and room tone in one pass is Seed Audio; a single non-speech sound is Sound Effects.',
766
810
  },
767
811
  ],
@@ -780,6 +824,7 @@ export const PROMPTING_TIPS = {
780
824
  'ltx-2-5': LTX_2_5,
781
825
  'nano-banana': NANO_BANANA,
782
826
  'nano-banana-lite': NANO_BANANA_LITE,
827
+ 'gpt-image-2-5': GPT_IMAGE_25,
783
828
  'seed-audio': SEED_AUDIO,
784
829
  'eleven-sfx': ELEVEN_SFX,
785
830
  'inworld-tts-2': INWORLD_TTS,
@@ -196,4 +196,6 @@ export declare const KLING_EDIT_MAX_REFS = 4;
196
196
  * trimmed first — subjects are the feature)
197
197
  */
198
198
  export declare function composeKlingEdit(rawPrompt: string, groups: ReferenceGroup[]): KlingEditComposition;
199
+ /** No-reference paths use the same composer and preserve authored prose. */
200
+ export declare function cleanPrompt(userPrompt: string): string;
199
201
  //# sourceMappingURL=reference-composer.d.ts.map
@@ -19,11 +19,12 @@
19
19
  // is each model's own official consistency lever (NB2 "assign a
20
20
  // distinct name", Seedance "Reference Subject_N in Image_N", Kling "reuse a fixed
21
21
  // label verbatim"); the heavy role-essay block was the off-doctrine part.
22
- // Normalize a name/token for matching: drop the sigil, lowercase, strip
23
- // spaces/underscores/hyphens. "@big_red" / "@Big Red" / "#Big-Red" all collapse
24
- // to the same key. Identical to the agent-side resolver's `norm`.
22
+ // A mention's matching key: lowercase, spaces/underscores/hyphens stripped, and
23
+ // the SIGIL KEPT. "@big_red" / "@Big Red" / "@Big-Red" are one key, and "@red"
24
+ // and "#red" are two: names are unique per sigil, so a subject and a look may
25
+ // share one, and a sigil-free key bound both mentions to whichever came last.
25
26
  function normToken(s) {
26
- return s.toLowerCase().replace(/[@#]/g, '').replace(/[\s_-]+/g, '');
27
+ return s.toLowerCase().replace(/[\s_-]+/g, '');
27
28
  }
28
29
  // Free-reference IMAGE kinds get an "image N" number. Frames are transported in
29
30
  // their own dedicated slots (start/last frame) by the per-model adapter and are
@@ -179,7 +180,7 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
179
180
  // ── 2. Inline-name token groups in the prompt body ──
180
181
  // For each character/environment group whose token appears in the prompt, the
181
182
  // FIRST occurrence becomes "Name (image N)"; later ones become just "Name".
182
- // Style tokens are removed (a single trailing clause carries the style). Token
183
+ // Style tokens become image citations at the user's chosen binding site. Token
183
184
  // groups NOT found in the prompt fall through to a key line in step 3.
184
185
  const tokenGroups = numbered.filter((g) => g.token && (g.kind === 'character' || g.kind === 'environment' || g.kind === 'style'));
185
186
  const byNorm = new Map();
@@ -204,20 +205,7 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
204
205
  unresolvedSeen.add(key);
205
206
  unresolvedTokens.push(`${sigil}${tok}`);
206
207
  };
207
- // First strip "in/with the style of #tag" phrases so the style reads as a
208
- // clean trailing clause, not a dangling preposition (legacy cleanPrompt
209
- // behaviour). ONLY when the tag resolves: an unresolved one leaves the whole
210
- // phrase exactly as authored and falls through to the token pass below, which
211
- // reports it and sends it as written.
212
- let body = rawPrompt.replace(/\s+(with|in)\s+the\s+style\s+of\s+([@#])([\w-]+(?:\.[\w-]+)*)/gi, (_full, _prep, sigil, tok) => {
213
- const g = byNorm.get(normToken(`${sigil}${tok}`));
214
- if (g && g.kind === 'style') {
215
- matchedInPrompt.add(normToken(`${sigil}${tok}`));
216
- return '';
217
- }
218
- return _full;
219
- });
220
- body = body.replace(TOKEN_RE, (_full, _sigil, tok) => {
208
+ const body = rawPrompt.replace(TOKEN_RE, (_full, _sigil, tok) => {
221
209
  const key = normToken(`${_sigil}${tok}`);
222
210
  const g = byNorm.get(key);
223
211
  if (!g) {
@@ -227,7 +215,7 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
227
215
  }
228
216
  matchedInPrompt.add(key);
229
217
  if (g.kind === 'style')
230
- return ''; // styles never inline — trailing clause only
218
+ return g.imageNums.length ? citeImages(g.imageNums) : g.name;
231
219
  if (!seenFirst.has(key)) {
232
220
  seenFirst.add(key);
233
221
  // 🚨 ONE BINDING SITE PER ENTITY, CARRYING EVERY MEDIUM SHE OWNS
@@ -257,8 +245,13 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
257
245
  }
258
246
  return g.name;
259
247
  });
260
- // Collapse the whitespace the token removals left behind.
261
- body = body.replace(/[ \t]{2,}/g, ' ').replace(/\s+([,.;:!?])/g, '$1').trim();
248
+ // Literal citations are authored bindings too. Do not add a competing role
249
+ // sentence when the user already describes what that image supplies.
250
+ const citedImages = new Set();
251
+ for (const match of body.matchAll(/\bimages?\s+(\d+(?:\s*(?:,\s*(?:(?:and|&)\s*)?|(?:and|&)\s*)\d+)*)\b/gi)) {
252
+ for (const n of match[1].match(/\d+/g) ?? [])
253
+ citedImages.add(Number(n));
254
+ }
262
255
  // ── 3. Build the key lines for token-less / unmatched-token groups ──
263
256
  // Video sources, pinned/base canvases, and picked subjects that have no token
264
257
  // in the prompt each get ONE short neutral key line (never an essay). The user's
@@ -314,10 +307,11 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
314
307
  for (const g of numbered) {
315
308
  if ((g.kind === 'character' || g.kind === 'environment') && g.imageNums.length > 0) {
316
309
  const tokenWasMatched = g.token && matchedInPrompt.has(normToken(g.token));
317
- if (!tokenWasMatched) {
318
- const noun = g.imageNums.length === 1 ? 'Image' : 'Images';
319
- const verb = g.imageNums.length === 1 ? 'is' : 'are';
320
- topKeys.push(`${noun} ${joinNums(g.imageNums)} ${verb} ${g.name}.`);
310
+ const unmentioned = g.imageNums.filter((n) => !citedImages.has(n));
311
+ if (!tokenWasMatched && unmentioned.length) {
312
+ const noun = unmentioned.length === 1 ? 'Image' : 'Images';
313
+ const verb = unmentioned.length === 1 ? 'is' : 'are';
314
+ topKeys.push(`${noun} ${joinNums(unmentioned)} ${verb} ${g.name}.`);
321
315
  }
322
316
  }
323
317
  }
@@ -419,11 +413,11 @@ export function composeReferences(rawPrompt, groups, opts = {}) {
419
413
  // `Audio 1` sitting mid-sentence among lowercase `image 1`s; moving the
420
414
  // binding inline removed the reason for the exception along with the
421
415
  // exception.
422
- // ── 4. Style trailing clause (one, at the end — style reads best last) ──
416
+ // ── 4. Fallback for style attachments the user has not cited ──
423
417
  const styleNums = [];
424
418
  for (const g of numbered) {
425
419
  if (g.kind === 'style')
426
- styleNums.push(...g.imageNums);
420
+ styleNums.push(...g.imageNums.filter((n) => !citedImages.has(n)));
427
421
  }
428
422
  const styleClauses = [];
429
423
  if (styleNums.length > 0) {
@@ -475,10 +469,12 @@ export function composeVoiceCitations(rawPrompt, voices) {
475
469
  if (v?.token)
476
470
  byNorm.set(normToken(v.token), { n: i + 1, name: v.name });
477
471
  });
472
+ // The one mention grammar (TOKEN_RE), so `joe@sarah.com` and `@sarah.extra`
473
+ // stay prose. A `#` token is a look, never a voice.
478
474
  const seen = new Set();
479
- const body = rawPrompt.replace(/@([\w-]+)/g, (full, tok) => {
480
- const key = normToken(`@${tok}`);
481
- const v = byNorm.get(key);
475
+ const body = rawPrompt.replace(TOKEN_RE, (full, sigil, tok) => {
476
+ const key = normToken(`${sigil}${tok}`);
477
+ const v = sigil === '@' ? byNorm.get(key) : undefined;
482
478
  if (!v)
483
479
  return full;
484
480
  if (seen.has(key))
@@ -525,7 +521,12 @@ export function composeKlingEdit(rawPrompt, groups) {
525
521
  const elements = [];
526
522
  const styleImages = [];
527
523
  const styleNums = [];
528
- let body = rawPrompt;
524
+ // Each mention's citation, by key. The prompt is walked ONCE with the one
525
+ // mention grammar (TOKEN_RE), so `joe@marcus.com` and `@marcus.extra` stay
526
+ // prose, as they do in `composeReferences`.
527
+ const citations = new Map();
528
+ const inPrompt = new Set([...rawPrompt.matchAll(TOKEN_RE)].map((t) => normToken(`${t[1]}${t[2]}`)));
529
+ const keyLines = [];
529
530
  // Subjects first — they own the @ElementN numbering.
530
531
  const subjectGroups = groups.filter((g) => (g.kind === 'character' || g.kind === 'environment') && g.media.some((m) => m.mediaKind === 'image'));
531
532
  for (const g of subjectGroups) {
@@ -536,20 +537,13 @@ export function composeKlingEdit(rawPrompt, groups) {
536
537
  continue;
537
538
  const n = elements.length + 1;
538
539
  elements.push({ frontal: imgs[0], angles: imgs.slice(1, 4), name: g.name });
539
- if (g.token) {
540
- // Replace every @token occurrence with the element citation.
541
- const escaped = g.token.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
542
- const re = new RegExp(`${escaped}\\b`, 'gi');
543
- if (re.test(body)) {
544
- body = body.replace(re, `@Element${n}`);
545
- }
546
- else {
547
- body = `${body}\n@Element${n} is ${g.name}.`;
548
- }
549
- }
550
- else {
551
- body = `${body}\n@Element${n} is ${g.name}.`;
552
- }
540
+ // Every mention of the subject becomes its element citation; a subject the
541
+ // prompt never names gets a key line.
542
+ const key = g.token ? normToken(g.token) : null;
543
+ if (key && inPrompt.has(key) && !citations.has(key))
544
+ citations.set(key, `@Element${n}`);
545
+ else
546
+ keyLines.push(`@Element${n} is ${g.name}.`);
553
547
  }
554
548
  // Style / pinned refs take the remaining slots as @ImageN.
555
549
  for (const g of groups) {
@@ -564,12 +558,15 @@ export function composeKlingEdit(rawPrompt, groups) {
564
558
  const n = styleImages.length;
565
559
  if (g.kind === 'style')
566
560
  styleNums.push(n);
567
- if (g.token) {
568
- const escaped = g.token.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
569
- body = body.replace(new RegExp(`${escaped}\\b`, 'gi'), `@Image${n}`);
570
- }
561
+ // A group's mention cites its FIRST image.
562
+ const key = g.token ? normToken(g.token) : null;
563
+ if (key && !citations.has(key))
564
+ citations.set(key, `@Image${n}`);
571
565
  }
572
566
  }
567
+ let body = rawPrompt.replace(TOKEN_RE, (full, sigil, tok) => citations.get(normToken(`${sigil}${tok}`)) ?? full);
568
+ for (const line of keyLines)
569
+ body = `${body}\n${line}`;
573
570
  if (styleNums.length > 0) {
574
571
  const cites = styleNums.map((n) => `@Image${n}`).join(' and ');
575
572
  body = `${body}\nApply the visual style of ${cites}.`;
@@ -584,4 +581,8 @@ export function composeKlingEdit(rawPrompt, groups) {
584
581
  styleImages,
585
582
  };
586
583
  }
584
+ /** No-reference paths use the same composer and preserve authored prose. */
585
+ export function cleanPrompt(userPrompt) {
586
+ return composeReferences(userPrompt, []).prompt;
587
+ }
587
588
  //# sourceMappingURL=reference-composer.js.map
@@ -0,0 +1,165 @@
1
+ /** Script document wire contract. Spoken text has one home: the scene string. */
2
+ export interface Block {
3
+ id: string;
4
+ kind: 'paragraph' | 'heading' | 'direction';
5
+ start: number;
6
+ end: number;
7
+ /** Only non-spoken blocks own text here. Paragraph text lives in Scene.script. */
8
+ label?: string;
9
+ level?: number;
10
+ marks: Array<{
11
+ start: number;
12
+ end: number;
13
+ type: 'strong' | 'em';
14
+ }>;
15
+ }
16
+ export interface Scene {
17
+ id: string;
18
+ name: string;
19
+ script: string;
20
+ blocks: Block[];
21
+ }
22
+ export interface Shot {
23
+ id: string;
24
+ sceneId: string;
25
+ range: [number, number] | null;
26
+ label: string;
27
+ prompt: string;
28
+ speaker: string;
29
+ }
30
+ export interface Document {
31
+ storyboardId: string;
32
+ recovered?: boolean;
33
+ version: 1;
34
+ revision: number;
35
+ title: string;
36
+ scenes: Scene[];
37
+ shots: Shot[];
38
+ }
39
+ export interface DocumentEdit {
40
+ sceneId: string;
41
+ at: number;
42
+ removed: number;
43
+ text: string;
44
+ }
45
+ export interface DocumentWrite {
46
+ expectedRevision: number;
47
+ edits: DocumentEdit[];
48
+ structures?: Array<{
49
+ sceneId: string;
50
+ blocks: Block[];
51
+ }>;
52
+ move?: {
53
+ sceneId: string;
54
+ blockId: string;
55
+ delta: -1 | 1;
56
+ };
57
+ makeShots?: Array<SectionFragment & {
58
+ waitingShotId?: string;
59
+ }>;
60
+ restoreRevision?: number;
61
+ sceneAction?: {
62
+ kind: 'split';
63
+ sceneId: string;
64
+ at: number;
65
+ } | {
66
+ kind: 'merge';
67
+ sceneId: string;
68
+ };
69
+ }
70
+ export interface SectionFragment {
71
+ sceneId: string;
72
+ start: number;
73
+ end: number;
74
+ }
75
+ export interface Section {
76
+ id: string;
77
+ storyboardId: string;
78
+ parentId: string | null;
79
+ label: string;
80
+ tags: string[];
81
+ fragments: SectionFragment[];
82
+ activeAlternativeId: string | null;
83
+ locallyChanged: boolean;
84
+ sourceSectionId?: string | null;
85
+ sourceRevisionId?: string | null;
86
+ alternatives: Array<{
87
+ id: string;
88
+ label: string;
89
+ revisionId: string;
90
+ }>;
91
+ }
92
+ /** `rename` and `archive` act on the section, or on one alternative when
93
+ * alternativeId is given. Archiving a section keeps its words on the page. */
94
+ export interface SectionInput {
95
+ expectedRevision: number;
96
+ action: 'create' | 'alternative' | 'choose' | 'save' | 'reuse' | 'updateUses' | 'rename' | 'archive' | 'tags';
97
+ sectionId?: string;
98
+ alternativeId?: string;
99
+ parentId?: string;
100
+ label?: string;
101
+ tags?: string[];
102
+ fragments?: SectionFragment[];
103
+ target?: {
104
+ sceneId: string;
105
+ at: number;
106
+ expectedRevision: number;
107
+ };
108
+ uses?: Array<{
109
+ sectionId: string;
110
+ expectedRevision: number;
111
+ }>;
112
+ }
113
+ /** A proposed replacement for exact words. It moves with edits around it and
114
+ * applies only while those words are unchanged (`stale` says they are not).
115
+ * Suggesting never edits the document; accepting is one undoable write. */
116
+ export interface ScriptSuggestion {
117
+ id: string;
118
+ storyboardId: string;
119
+ sceneId: string;
120
+ start: number;
121
+ end: number;
122
+ original: string;
123
+ replacement: string;
124
+ note: string;
125
+ status: 'pending' | 'accepted' | 'dismissed';
126
+ stale: boolean;
127
+ baseRevision: number;
128
+ }
129
+ export interface SuggestionInput {
130
+ expectedRevision: number;
131
+ action: 'create' | 'accept' | 'dismiss';
132
+ suggestions?: Array<{
133
+ sceneId: string;
134
+ start: number;
135
+ end: number;
136
+ original: string;
137
+ replacement: string;
138
+ note?: string;
139
+ }>;
140
+ suggestionId?: string;
141
+ }
142
+ export interface VariationChoice {
143
+ sectionId: string;
144
+ alternativeId: string;
145
+ }
146
+ /** Enumerate only on demand; even a large Cartesian set costs one row of memory. */
147
+ export declare function variationCombinations(axes: VariationChoice[][], prefix?: VariationChoice[]): Generator<VariationChoice[]>;
148
+ export interface VariationInput {
149
+ expectedRevision: number;
150
+ name: string;
151
+ choices: VariationChoice[];
152
+ /** When present, this sequence replaces the arrangement: omission and repetition are deliberate. */
153
+ arrangement?: VariationChoice[];
154
+ itemOverrides?: Array<{
155
+ from: string;
156
+ to: string;
157
+ voice: 'keep' | 'replace';
158
+ }>;
159
+ assetOverrides?: Array<{
160
+ from: string;
161
+ to: string;
162
+ }>;
163
+ idempotencyKey?: string;
164
+ }
165
+ //# sourceMappingURL=script-document.d.ts.map
@@ -0,0 +1,11 @@
1
+ /** Enumerate only on demand; even a large Cartesian set costs one row of memory. */
2
+ export function* variationCombinations(axes, prefix = []) {
3
+ if (!axes.length) {
4
+ yield prefix;
5
+ return;
6
+ }
7
+ const [head, ...tail] = axes;
8
+ for (const choice of head)
9
+ yield* variationCombinations(tail, [...prefix, choice]);
10
+ }
11
+ //# sourceMappingURL=script-document.js.map
@@ -87,7 +87,7 @@ export interface SpeechRate {
87
87
  */
88
88
  export declare const SPEECH_RATE_QUERY: string;
89
89
  /** When the numbers below were last derived from the query above. */
90
- export declare const SPEECH_RATE_MEASURED_ON = "2026-09-04";
90
+ export declare const SPEECH_RATE_MEASURED_ON = "2026-09-30";
91
91
  /**
92
92
  * 🔑 THE REGISTER SPLIT IS REAL AND MEASURED, which is why there is no single
93
93
  * number. the performed, genre-acted reads run a median of 133 wpm;
@@ -102,8 +102,8 @@ export declare const SPEECH_RATE: {
102
102
  readonly segment: "speech_rate = 'performed'";
103
103
  };
104
104
  readonly conversational: {
105
- readonly wpm: 159;
106
- readonly n: 41;
105
+ readonly wpm: 153;
106
+ readonly n: 43;
107
107
  readonly segment: "speech_rate = 'conversational'";
108
108
  };
109
109
  readonly direct_response: {
@@ -119,7 +119,7 @@ export declare const SPEECH_RATE: {
119
119
  */
120
120
  readonly ceiling: {
121
121
  readonly wpm: 283;
122
- readonly n: 65;
122
+ readonly n: 67;
123
123
  readonly segment: "max over every marked row";
124
124
  };
125
125
  };
@@ -124,7 +124,7 @@ SELECT id, speech_rate, duration_seconds, transcript
124
124
  -- statistic: median per register; ceiling = max over every marked row
125
125
  `.trim();
126
126
  /** When the numbers below were last derived from the query above. */
127
- export const SPEECH_RATE_MEASURED_ON = '2026-09-04';
127
+ export const SPEECH_RATE_MEASURED_ON = '2026-09-30';
128
128
  /**
129
129
  * 🔑 THE REGISTER SPLIT IS REAL AND MEASURED, which is why there is no single
130
130
  * number. the performed, genre-acted reads run a median of 133 wpm;
@@ -134,7 +134,7 @@ export const SPEECH_RATE_MEASURED_ON = '2026-09-04';
134
134
  */
135
135
  export const SPEECH_RATE = {
136
136
  performed: { wpm: 133, n: 4, segment: "speech_rate = 'performed'" },
137
- conversational: { wpm: 159, n: 41, segment: "speech_rate = 'conversational'" },
137
+ conversational: { wpm: 153, n: 43, segment: "speech_rate = 'conversational'" },
138
138
  direct_response: { wpm: 169, n: 20, segment: "speech_rate = 'direct_response'" },
139
139
  /**
140
140
  * The fastest read in the whole corpus. **The fit check flags only ABOVE
@@ -142,7 +142,7 @@ export const SPEECH_RATE = {
142
142
  * 250, not p90: those are rates real ads actually hit, and flagging an
143
143
  * achievable read is exactly how a check gets ignored.
144
144
  */
145
- ceiling: { wpm: 283, n: 65, segment: 'max over every marked row' },
145
+ ceiling: { wpm: 283, n: 67, segment: 'max over every marked row' },
146
146
  };
147
147
  /**
148
148
  * The default register, and there is deliberately no picker for it.