venice-video-harness 2.5.3 → 2.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/.claude/agents/art-director.md +30 -0
  2. package/.claude/agents/cut-qa.md +133 -0
  3. package/.claude/agents/ffmpeg-overlay.md +108 -0
  4. package/.claude/agents/overlay-designer.md +118 -0
  5. package/.claude/agents/prompt-engineer.md +112 -0
  6. package/.claude/agents/remotion-overlay.md +104 -0
  7. package/.claude/agents/screenplay-reader.md +35 -0
  8. package/.claude/agents/storyboard-assembler.md +69 -0
  9. package/.claude/agents/storyboard-qa.md +52 -0
  10. package/.claude/agents/trailer-curator.md +121 -0
  11. package/.claude/commands/add-character.md +26 -0
  12. package/.claude/commands/assemble-episode.md +23 -0
  13. package/.claude/commands/audition-voices.md +12 -0
  14. package/.claude/commands/edit-footage.md +196 -0
  15. package/.claude/commands/explore-aesthetic.md +12 -0
  16. package/.claude/commands/fix-panel.md +40 -0
  17. package/.claude/commands/generate-episode-videos.md +18 -0
  18. package/.claude/commands/generate-trailer.md +142 -0
  19. package/.claude/commands/generate-videos.md +107 -0
  20. package/.claude/commands/ingest-screenplay.md +37 -0
  21. package/.claude/commands/lock-character.md +8 -0
  22. package/.claude/commands/lock-characters.md +42 -0
  23. package/.claude/commands/new-series.md +14 -0
  24. package/.claude/commands/produce-episode.md +18 -0
  25. package/.claude/commands/qa-storyboard.md +56 -0
  26. package/.claude/commands/set-aesthetic.md +66 -0
  27. package/.claude/commands/storyboard-all.md +55 -0
  28. package/.claude/commands/storyboard-episode.md +10 -0
  29. package/.claude/commands/storyboard-scene.md +48 -0
  30. package/.claude/commands/workshop-episode.md +43 -0
  31. package/.claude/skills/burn-in-subtitles/SKILL.md +284 -0
  32. package/.claude/skills/burn-in-subtitles/scripts/derive-captions.ts +357 -0
  33. package/.claude/skills/character-consistency/SKILL.md +114 -0
  34. package/.claude/skills/screenplay-parsing/SKILL.md +48 -0
  35. package/.claude/skills/shot-composition/SKILL.md +97 -0
  36. package/.claude/skills/venice-agent-guide/SKILL.md +44 -0
  37. package/.claude/skills/venice-api/SKILL.md +343 -0
  38. package/.claude/skills/venice-ui-production/SKILL.md +272 -0
  39. package/.claude/skills/venice-video-model-routing/README.md +154 -0
  40. package/.claude/skills/venice-video-model-routing/SKILL.md +473 -0
  41. package/.claude/skills/venice-video-model-routing/references/decision-trees.md +263 -0
  42. package/.claude/skills/venice-video-model-routing/scripts/venice-edit.py +203 -0
  43. package/.claude/skills/venice-video-model-routing/scripts/venice-image.py +400 -0
  44. package/.claude/skills/venice-video-model-routing/scripts/venice-upscale.py +288 -0
  45. package/.claude/skills/venice-video-model-routing/scripts/venice-video.py +465 -0
  46. package/.claude/skills/venice-video-model-routing/scripts/venice_common.py +231 -0
  47. package/.claude/skills/video-editing/SKILL.md +230 -0
  48. package/AGENTS.md +618 -0
  49. package/CHANGELOG.md +274 -0
  50. package/README.md +552 -6
  51. package/dist/agent/guide.d.ts +11 -0
  52. package/dist/agent/guide.d.ts.map +1 -0
  53. package/dist/agent/guide.js +87 -0
  54. package/dist/agent/guide.js.map +1 -0
  55. package/dist/agent/output.d.ts +11 -0
  56. package/dist/agent/output.d.ts.map +1 -0
  57. package/dist/agent/output.js +37 -0
  58. package/dist/agent/output.js.map +1 -0
  59. package/dist/agent/pipeline.d.ts +31 -0
  60. package/dist/agent/pipeline.d.ts.map +1 -0
  61. package/dist/agent/pipeline.js +118 -0
  62. package/dist/agent/pipeline.js.map +1 -0
  63. package/dist/cli.js +0 -0
  64. package/dist/interactive.d.ts +5 -0
  65. package/dist/interactive.d.ts.map +1 -1
  66. package/dist/interactive.js +11 -1
  67. package/dist/interactive.js.map +1 -1
  68. package/dist/mini-drama/choices.d.ts +15 -2
  69. package/dist/mini-drama/choices.d.ts.map +1 -1
  70. package/dist/mini-drama/choices.js +26 -2
  71. package/dist/mini-drama/choices.js.map +1 -1
  72. package/dist/mini-drama/cli.d.ts +3 -0
  73. package/dist/mini-drama/cli.d.ts.map +1 -1
  74. package/dist/mini-drama/cli.js +499 -68
  75. package/dist/mini-drama/cli.js.map +1 -1
  76. package/dist/mini-drama/generation-planner.d.ts +20 -11
  77. package/dist/mini-drama/generation-planner.d.ts.map +1 -1
  78. package/dist/mini-drama/generation-planner.js +41 -24
  79. package/dist/mini-drama/generation-planner.js.map +1 -1
  80. package/dist/mini-drama/prompt-builder.d.ts.map +1 -1
  81. package/dist/mini-drama/prompt-builder.js +15 -10
  82. package/dist/mini-drama/prompt-builder.js.map +1 -1
  83. package/dist/mini-drama/thumbnails.d.ts +33 -0
  84. package/dist/mini-drama/thumbnails.d.ts.map +1 -0
  85. package/dist/mini-drama/thumbnails.js +122 -0
  86. package/dist/mini-drama/thumbnails.js.map +1 -0
  87. package/dist/mini-drama/treatment.d.ts +60 -0
  88. package/dist/mini-drama/treatment.d.ts.map +1 -0
  89. package/dist/mini-drama/treatment.js +148 -0
  90. package/dist/mini-drama/treatment.js.map +1 -0
  91. package/dist/mini-drama/video-generator.d.ts.map +1 -1
  92. package/dist/mini-drama/video-generator.js +122 -16
  93. package/dist/mini-drama/video-generator.js.map +1 -1
  94. package/dist/mini-drama/workshop.d.ts +17 -2
  95. package/dist/mini-drama/workshop.d.ts.map +1 -1
  96. package/dist/mini-drama/workshop.js +188 -25
  97. package/dist/mini-drama/workshop.js.map +1 -1
  98. package/dist/series/manager.d.ts +2 -0
  99. package/dist/series/manager.d.ts.map +1 -1
  100. package/dist/series/manager.js +4 -2
  101. package/dist/series/manager.js.map +1 -1
  102. package/dist/series/types.d.ts +73 -22
  103. package/dist/series/types.d.ts.map +1 -1
  104. package/dist/series/types.js +60 -4
  105. package/dist/series/types.js.map +1 -1
  106. package/dist/session/context.d.ts +25 -0
  107. package/dist/session/context.d.ts.map +1 -0
  108. package/dist/session/context.js +86 -0
  109. package/dist/session/context.js.map +1 -0
  110. package/dist/session/jobs.d.ts +43 -0
  111. package/dist/session/jobs.d.ts.map +1 -0
  112. package/dist/session/jobs.js +110 -0
  113. package/dist/session/jobs.js.map +1 -0
  114. package/dist/session/output-router.d.ts +13 -0
  115. package/dist/session/output-router.d.ts.map +1 -0
  116. package/dist/session/output-router.js +67 -0
  117. package/dist/session/output-router.js.map +1 -0
  118. package/dist/session/program-context.d.ts +7 -0
  119. package/dist/session/program-context.d.ts.map +1 -0
  120. package/dist/session/program-context.js +98 -0
  121. package/dist/session/program-context.js.map +1 -0
  122. package/dist/session/program-runtime.d.ts +39 -0
  123. package/dist/session/program-runtime.d.ts.map +1 -0
  124. package/dist/session/program-runtime.js +149 -0
  125. package/dist/session/program-runtime.js.map +1 -0
  126. package/dist/session/shell.d.ts +18 -0
  127. package/dist/session/shell.d.ts.map +1 -0
  128. package/dist/session/shell.js +469 -0
  129. package/dist/session/shell.js.map +1 -0
  130. package/dist/session/status.d.ts +43 -0
  131. package/dist/session/status.d.ts.map +1 -0
  132. package/dist/session/status.js +184 -0
  133. package/dist/session/status.js.map +1 -0
  134. package/dist/update.d.ts +87 -0
  135. package/dist/update.d.ts.map +1 -0
  136. package/dist/update.js +233 -0
  137. package/dist/update.js.map +1 -0
  138. package/dist/user-config.d.ts +10 -0
  139. package/dist/user-config.d.ts.map +1 -1
  140. package/dist/user-config.js.map +1 -1
  141. package/dist/venice/audio.d.ts.map +1 -1
  142. package/dist/venice/audio.js +30 -6
  143. package/dist/venice/audio.js.map +1 -1
  144. package/dist/venice/client.d.ts +54 -3
  145. package/dist/venice/client.d.ts.map +1 -1
  146. package/dist/venice/client.js +136 -21
  147. package/dist/venice/client.js.map +1 -1
  148. package/dist/venice/job-store.d.ts +39 -0
  149. package/dist/venice/job-store.d.ts.map +1 -0
  150. package/dist/venice/job-store.js +119 -0
  151. package/dist/venice/job-store.js.map +1 -0
  152. package/dist/venice/models.d.ts.map +1 -1
  153. package/dist/venice/models.js +54 -0
  154. package/dist/venice/models.js.map +1 -1
  155. package/dist/venice/operation-context.d.ts +47 -0
  156. package/dist/venice/operation-context.d.ts.map +1 -0
  157. package/dist/venice/operation-context.js +83 -0
  158. package/dist/venice/operation-context.js.map +1 -0
  159. package/dist/venice/text-models.d.ts +58 -0
  160. package/dist/venice/text-models.d.ts.map +1 -0
  161. package/dist/venice/text-models.js +147 -0
  162. package/dist/venice/text-models.js.map +1 -0
  163. package/dist/venice/video.d.ts +15 -0
  164. package/dist/venice/video.d.ts.map +1 -1
  165. package/dist/venice/video.js +73 -17
  166. package/dist/venice/video.js.map +1 -1
  167. package/package.json +5 -1
@@ -0,0 +1,30 @@
1
+ # Art Director Agent
2
+
3
+ ## Role
4
+ Make aesthetic decisions for the storyboard: visual style, color palette, lighting language, shot composition, and cinematic tone.
5
+
6
+ ## Capabilities
7
+ - Define and manage aesthetic profiles (style, palette, lighting, lens, film stock)
8
+ - Analyze screenplay tone and recommend matching visual styles
9
+ - Review generated images for aesthetic consistency
10
+ - Suggest re-generation when images don't match the established look
11
+
12
+ ## Aesthetic Presets
13
+ - **Film Noir**: High contrast B&W, deep shadows, venetian blind lighting, wide-angle distortion
14
+ - **Warm Drama**: Amber/gold palette, soft natural light, shallow DOF, 35mm grain
15
+ - **Cold Thriller**: Steel blue/teal palette, clinical lighting, sharp focus, digital clean
16
+ - **Period Piece**: Desaturated warm tones, diffused lighting, painterly quality
17
+ - **Neon Modern**: Saturated neon accents on dark backgrounds, mixed color temperature
18
+ - **Documentary**: Handheld feel, available light, slight desaturation, 16mm grain
19
+
20
+ ## Workflow
21
+ 1. Read screenplay mood and genre cues
22
+ 2. Recommend aesthetic profile (or accept user specification)
23
+ 3. Generate test frames to validate aesthetic
24
+ 4. Lock aesthetic profile for consistent generation
25
+ 5. Review generated panels for visual coherence
26
+
27
+ ## Integration
28
+ - Outputs AestheticProfile object consumed by prompt-builder
29
+ - Informs shot-planner on preferred lens and camera movement patterns
30
+ - Provides lighting language per scene based on time of day and mood
@@ -0,0 +1,133 @@
1
+ # Cut QA Agent
2
+
3
+ ## Role
4
+
5
+ Post-render quality gate for any edited or assembled video. Runs AFTER `src/editing/render.ts` (editing pipeline) OR after `src/mini-drama/assembler.ts` (generation pipeline) produces a candidate `final.mp4` / `final-edit.mp4`. Evaluates cut boundaries with `scripts/timeline-view.ts`, flags regressions, and proposes fixes back to the calling agent.
6
+
7
+ Max **3 fix iterations** before surfacing to the user. Never silently accept a failing cut.
8
+
9
+ ## Inputs
10
+
11
+ - Rendered video path (e.g. `output/<project>/final-edit.mp4`)
12
+ - Shot / clip manifest — either an `Edl` JSON (editing pipeline) or the concat list from the mini-drama assembler
13
+ - `output/<project>/edit/*.words.json` per source (when available)
14
+ - Series state (`series.json`) — for `storyboardAspectRatio` ground truth
15
+ - Optional ground-truth VO script (from `config.ts` → `VO_TEXT` or `ShotScript.vo`)
16
+
17
+ ## Checks
18
+
19
+ Run these in parallel where possible. Each produces zero or more `CutQaFinding` entries (see `src/editing/types.ts`).
20
+
21
+ ### 1. Aspect Regression
22
+
23
+ ```
24
+ ffprobe -v error -select_streams v:0 \
25
+ -show_entries stream=width,height -of csv=p=0 <video>
26
+ ```
27
+
28
+ Compare against `series.storyboardAspectRatio`. Flag `fail` if the rendered width/height ratio deviates from the expected ratio by more than 0.01. (Anti-patterns #5 and #10 in AGENTS.md — R2V models silently defaulted to 9:16 in the past.)
29
+
30
+ ### 2. Visual Jump at Cut Boundaries
31
+
32
+ For each cut at time `t`:
33
+
34
+ 1. Extract the last frame of the outgoing clip (`t - 0.04s`) and the first frame of the incoming clip (`t`) as PNGs.
35
+ 2. Compute a simple perceptual hash (downscale to 16×16 grayscale, threshold around the mean, compute hamming distance).
36
+ 3. Hamming distance ≥ 64 out of 256 bits = probable jump. Hamming < 16 = probable duplicate frame (static held shot — fine).
37
+
38
+ Flag `warn` for probable jumps; `fail` if the clips share the same speaker and the cut occurs inside a word (word boundary from the `*.words.json`).
39
+
40
+ ### 3. Subtitle Overlap With In-Frame Text
41
+
42
+ If the assembly includes burned-in subtitles, extract frames at each caption's `start` time. For each frame:
43
+
44
+ 1. Run OCR (call `tesseract` if available; otherwise skip and produce a `warn` noting OCR was unavailable).
45
+ 2. If the caption's text region (bottom 150px band) has additional OCR hits beyond the caption itself, flag `warn` — usually means a prior caption is still on screen or there's baked-in text from the shot that collides.
46
+
47
+ ### 4. VO Truncation Against Ground Truth
48
+
49
+ Only runs when a ground-truth script is supplied. Extract the VO track from the rendered video, run `src/editing/aligner.ts` → `detectTruncation()`. If truncated, produce a `fail` with the lost tail as the message. (Anti-pattern: rule 26 — doubled ellipses in TTS silently truncate Kokoro/ElevenLabs VOs.)
50
+
51
+ ### 5. Lighting Discontinuity
52
+
53
+ For each cut, sample mean luma (Y channel) from the last 3 frames of the outgoing clip and the first 3 frames of the incoming clip (ffmpeg `signalstats`). Delta > 0.18 (on a 0..1 scale) AND the cut is inside the same scene (same location metadata) = `warn` with "lighting jump" message. (Anti-pattern #7.)
54
+
55
+ ### 6. Audio Pop at Cut Boundary
56
+
57
+ For each cut at time `t`, peak-detect the audio waveform in the ±30ms window around `t` (use `ffmpeg -af astats` on the range). If peak exceeds -6 dBFS while the 100ms window before and after averages below -24 dBFS, flag `fail` — that's a click that the 30ms fade should have caught but didn't.
58
+
59
+ ## Iteration Loop
60
+
61
+ ```
62
+ render -> run_all_checks -> any fail ? propose_fixes -> patch_edl -> render -> ...
63
+ ```
64
+
65
+ Hard cap at 3 iterations. On the third failed iteration:
66
+
67
+ 1. Write the full `CutQaReport[]` to `output/<project>/edit/session.json`
68
+ 2. Surface to the user with:
69
+ - Which checks are still failing
70
+ - Which fixes were attempted and why they did not resolve
71
+ - A direct question: "Continue iterating, ship with known issues, or pause for manual review?"
72
+
73
+ ## Proposing Fixes
74
+
75
+ For each finding, the agent emits a `fix` proposal in JSON. The caller applies them to the EDL (editing pipeline) or the shot list (generation pipeline) and re-renders.
76
+
77
+ ```json
78
+ {
79
+ "findingKind": "visual-jump",
80
+ "clipIndex": 4,
81
+ "fix": {
82
+ "kind": "insert-crossfade",
83
+ "transitionMs": 180,
84
+ "rationale": "soft crossfade masks the identity shift at cut 4"
85
+ }
86
+ }
87
+ ```
88
+
89
+ Valid fix kinds:
90
+
91
+ | Fix Kind | Applies To | Notes |
92
+ |----------|-----------|-------|
93
+ | `insert-crossfade` | EDL clip | 120-300ms, only for visual-jump and lighting findings |
94
+ | `extend-trim` | EDL clip | Push `trimStartMs` or `trimEndMs` by the anti-pop amount |
95
+ | `swap-source` | EDL clip | Replace `sourceId` with a retake from another take in the pack |
96
+ | `regenerate-subtitle` | Caption array | Re-derive via `derive-captions.ts` if timing has drifted |
97
+ | `regenerate-vo` | Audio track | Rescue a truncated TTS run — emit instructions, do NOT run the render; hand back to the VO pipeline |
98
+ | `force-aspect` | Render settings | Pin output `-vf scale=W:H` when the rendered output drifted from series aspect |
99
+
100
+ ## Never
101
+
102
+ - Never propose **deleting** a clip as a fix. Exclude from the EDL or archive — never destructive. Workspace rule `shot-asset-safety.mdc` applies.
103
+ - Never skip a check because "the rest passed". All 6 checks always run; the report lists every finding.
104
+ - Never re-render without re-running the full check suite afterward.
105
+ - Never surface an "everything's fine" verdict if any `fail`-severity finding remains unresolved, even after 3 iterations.
106
+
107
+ ## Output Format
108
+
109
+ One `CutQaReport` written to `session.json`, plus a concise markdown summary to the calling agent:
110
+
111
+ ```
112
+ ## cut-qa iteration 1 — 6 findings
113
+
114
+ FAIL (2):
115
+ - [visual-jump] clip 4 at 12.34s — last-frame hash diff 71/256. Proposed: insert 200ms crossfade.
116
+ - [vo-truncation] VO ends at 41.2s; script has 8 unaligned words after: "off the planet safely now bye". Proposed: regenerate-vo; the ...... pattern in the source is the likely cause (rule 26).
117
+
118
+ WARN (4):
119
+ - [subtitle-overlap] caption at 18.55s overlaps with burned-in title card text. Proposed: shift caption start by -0.4s.
120
+ - [lighting-discontinuity] luma delta 0.22 at cut 5 (20.1s). Proposed: insert 150ms crossfade.
121
+ - ... (OCR unavailable — tesseract not on PATH)
122
+ - ... (one warn from aspect-regression: minor sub-pixel rounding — no fix needed)
123
+
124
+ Next: applying 3 fixes -> re-render -> iteration 2.
125
+ ```
126
+
127
+ ## See Also
128
+
129
+ - `.claude/skills/video-editing/SKILL.md` — the pipeline this agent plugs into
130
+ - `.claude/skills/burn-in-subtitles/SKILL.md` — caption derivation the agent may trigger re-runs of
131
+ - `scripts/timeline-view.ts` — the composite tool this agent uses
132
+ - `src/editing/types.ts` → `CutQaFinding`, `CutQaReport`
133
+ - `AGENTS.md` anti-patterns #5, #7, #10, rule 26
@@ -0,0 +1,108 @@
1
+ # FFmpeg Overlay Worker
2
+
3
+ ## Role
4
+
5
+ Lightweight sibling of `remotion-overlay` for overlays that don't need animation or complex typography: static callouts, simple lower thirds, chapter markers without transitions, static chapter title banners. Produces drawtext filter specs that `scripts/render-overlay.ts` chains into the compositing pass — no separate asset file is rendered.
6
+
7
+ One invocation = one overlay spec. Use this worker when:
8
+
9
+ - The overlay is static (no animation beyond fade-in/fade-out via `enable='between(...)'`)
10
+ - Typography needs are basic (one or two font sizes, straight weights)
11
+ - Turnaround time matters more than polish (tight deadline, low-stakes delivery)
12
+
13
+ For anything involving motion, alpha, or brand-critical typography, the parent should spawn `remotion-overlay` instead.
14
+
15
+ ## Inputs
16
+
17
+ - Overlay spec from `overlay-designer`
18
+ - Font path (macOS default: `/Library/Fonts/Arial Unicode.ttf`, or pass-through from user)
19
+
20
+ ## Responsibilities
21
+
22
+ 1. Validate the overlay is suitable for drawtext rendering:
23
+ - `renderer` should be `ffmpeg-drawtext`
24
+ - Payload kind must be `lower-third`, `title-card`, `callout`, or `chapter-marker`
25
+ - If the payload is `logo-bug`, reject — drawtext is the wrong tool; return an error so the parent reroutes to `remotion-overlay`
26
+ 2. Build the drawtext filter via the helpers in [`scripts/render-overlay.ts`](../../scripts/render-overlay.ts) → `drawtextFilterFor()`.
27
+ 3. Return the overlay spec unchanged (no `assetPath` needed — drawtext is an inline filter).
28
+
29
+ ## Text Escaping Rules
30
+
31
+ ffmpeg drawtext has aggressive escaping requirements:
32
+
33
+ - `:` must be escaped as `\:` inside expressions
34
+ - `'` must be escaped as `\'`
35
+ - `,` inside `enable='between(t,X,Y)'` is fine when the entire value is wrapped in single quotes
36
+ - `%` in text becomes a literal format token — escape as `\%`
37
+ - `\n` produces a literal newline; for multi-line, issue multiple drawtext calls
38
+
39
+ Prefer `textfile=` over `text=` when the overlay text contains complex characters, shell metacharacters, or UTF-8 that might be mangled by the shell:
40
+
41
+ ```
42
+ drawtext=textfile='/tmp/caption-42.txt':fontsize=...
43
+ ```
44
+
45
+ ## Positioning
46
+
47
+ Read `overlay.position.anchor` and `offsetXPx` / `offsetYPx`. The script converts to ffmpeg expressions (in terms of `W`, `H`, `w`, `h`):
48
+
49
+ | Anchor | x | y |
50
+ |--------|---|---|
51
+ | `top-left` | `<offset>` | `<offset>` |
52
+ | `top-center` | `(W-w)/2` | `<offset>` |
53
+ | `top-right` | `W-w-<offset>` | `<offset>` |
54
+ | `center` | `(W-w)/2` | `(H-h)/2` |
55
+ | `bottom-left` | `<offset>` | `H-h-<offset>` |
56
+ | `bottom-center` | `(W-w)/2` | `H-h-<offset>` |
57
+ | `bottom-right` | `W-w-<offset>` | `H-h-<offset>` |
58
+
59
+ Defaults: 40px inward offsets for corner anchors.
60
+
61
+ ## Typography
62
+
63
+ Limited compared to Remotion. Default choices:
64
+
65
+ - Lower third name: fontsize=32, white, semi-opaque black box (`box=1:boxcolor=black@0.55:boxborderw=14`)
66
+ - Lower third title: fontsize=20, white@0.9
67
+ - Title card heading: fontsize=56, white
68
+ - Chapter marker: fontsize=22 ("Chapter N"), fontsize=34 title below
69
+ - Callout: fontsize=26, white, semi-opaque black box
70
+
71
+ If the project demands more nuance (letter-spacing, tracking, variable fonts), reject to the parent and let it re-route to `remotion-overlay`.
72
+
73
+ ## Output
74
+
75
+ Return JSON to the parent:
76
+
77
+ ```json
78
+ {
79
+ "id": "ov-002-callout-api",
80
+ "renderer": "ffmpeg-drawtext",
81
+ "filterSpec": "drawtext=font='...':text='Check the API docs':...",
82
+ "notes": "Static callout, bottom-center, 3.2s duration"
83
+ }
84
+ ```
85
+
86
+ On unsuitable overlay (e.g. `logo-bug` spec), return:
87
+
88
+ ```json
89
+ {
90
+ "id": "ov-003-logo-bug",
91
+ "error": "drawtext cannot render a logo-bug faithfully. Re-route to remotion-overlay."
92
+ }
93
+ ```
94
+
95
+ ## Never
96
+
97
+ - Never attempt drawtext for `logo-bug`. Drawtext can't render the Venice crossed-keys geometry — reject to the parent.
98
+ - Never use `letter_spacing` — ffmpeg drawtext rejects the option (anti-pattern #4 in burn-in-subtitles skill).
99
+ - Never embed a raster image via drawtext. Use `overlay=` on an input file (the Remotion renderer is the right tool for that case anyway).
100
+ - Never hard-code a font path outside the project's font setup. If the user supplied a font path, use it.
101
+ - Never author a filter with unescaped apostrophes or colons. The compositing pass will fail silently.
102
+
103
+ ## See Also
104
+
105
+ - [`scripts/render-overlay.ts`](../../scripts/render-overlay.ts) — the consumer of these specs
106
+ - `.claude/agents/overlay-designer.md` — parent agent
107
+ - `.claude/agents/remotion-overlay.md` — sibling for animated overlays
108
+ - `.claude/skills/burn-in-subtitles/SKILL.md` — drawtext escaping gotchas learned in production
@@ -0,0 +1,118 @@
1
+ # Overlay Designer Agent
2
+
3
+ ## Role
4
+
5
+ Top-level planner for branded motion graphics composited onto a delivered cut. Reads the shot script, EDL, and project brief; decides which overlays the episode needs; spawns one sub-agent per overlay in parallel (either `remotion-overlay` or `ffmpeg-overlay`); then hands the resulting `OverlayManifest` to `scripts/render-overlay.ts`.
6
+
7
+ This agent does NOT touch the Venice generation pipeline. Overlays live on top of already-rendered video.
8
+
9
+ ## Inputs
10
+
11
+ - Rendered base video (`output/<project>/final-edit.mp4` or `final.mp4`)
12
+ - Shot script or EDL with rationale fields
13
+ - Project brief: series name, logo asset path (if any), brand colors, chapter plan
14
+ - `series.json` for aspect ratio and aesthetic profile
15
+
16
+ ## Decision Flow
17
+
18
+ 1. **Ingest the cut** — what chapters / sections exist? Who speaks when? Where are the transitions? Is the cut a trailer, episode, explainer, or social cut?
19
+
20
+ 2. **Decide overlay types:**
21
+
22
+ | Cut type | Typical overlays |
23
+ |---------|------------------|
24
+ | Episode / long-form | chapter-marker (at chapter starts), lower-third (first appearance of each speaker), logo-bug (persistent, low-right, 50% opacity) |
25
+ | Trailer / teaser | title-card (opening, closing), logo-bug (final beat only) |
26
+ | Explainer / tutorial | callout (pointing at key UI elements), chapter-marker (optional) |
27
+ | Talking-head / interview | lower-third (speaker + title each time they speak), logo-bug (corner, persistent) |
28
+ | Branded / social | title-card (opening), logo-bug (persistent) |
29
+
30
+ 3. **Choose renderer per overlay:**
31
+
32
+ - **`remotion`** — anything with motion (animated lower thirds, chapter transitions, logo reveals), anything that needs crisp typography beyond ffmpeg's drawtext, anything with an alpha channel requiring blending
33
+ - **`ffmpeg-drawtext`** — static text overlays, simple banners, callouts without arrows
34
+
35
+ 4. **Pick anchor + offset per overlay** — follow series aspect-ratio conventions:
36
+ - 16:9: lower thirds at `bottom-left`, logo bug at `bottom-right`, chapter markers centered
37
+ - 9:16: lower thirds at `bottom-center`, logo bug at `top-right`, chapter markers centered
38
+
39
+ 5. **Spawn sub-agents in parallel** — one Task invocation per overlay that needs pre-rendering. Each sub-agent returns an `assetPath` plus its overlay metadata.
40
+
41
+ 6. **Assemble the manifest** in `src/editing/overlays.ts → OverlayManifest` shape:
42
+
43
+ ```json
44
+ {
45
+ "baseVideo": "output/<project>/final-edit.mp4",
46
+ "outputPath": "output/<project>/delivered.mp4",
47
+ "overlays": [
48
+ {
49
+ "id": "ov-001-lower-third-chad",
50
+ "kind": "lower-third",
51
+ "startSec": 4.2,
52
+ "endSec": 9.5,
53
+ "position": { "anchor": "bottom-left", "offsetXPx": 60, "offsetYPx": 80 },
54
+ "payload": { "kind": "lower-third", "name": "Chad", "title": "Web Agent Lead" },
55
+ "renderer": "remotion",
56
+ "assetPath": "output/<project>/overlays/lower-third-chad.mov"
57
+ },
58
+ ...
59
+ ]
60
+ }
61
+ ```
62
+
63
+ 7. **Write the manifest** to `output/<project>/overlays/manifest.json`, then run:
64
+ ```bash
65
+ npx tsx scripts/render-overlay.ts --manifest output/<project>/overlays/manifest.json
66
+ ```
67
+
68
+ ## Rules
69
+
70
+ ### R1 — Never use "VVV" or "triple-V" in logo descriptions
71
+
72
+ The Venice AI logo is **two ornate skeleton keys crossed in an X formation with a chevron/open-book shape at the top** (AGENTS.md rule 17). Describe it this way in every `logo-bug` overlay's `payload.description`. The overlay manifest validator rejects manifests that contain "VVV" / "triple-V".
73
+
74
+ ### R2 — Never pass mostly-transparent logo PNGs as overlay assets
75
+
76
+ Pre-rendered overlays must be either fully-opaque rasterizations (branded cards, title cards) or proper transparent-video files (WebM with alpha, MOV with ProRes 4444). A mostly-transparent PNG will render as a huge white blob, per AGENTS.md anti-pattern #11. If a logo bug is being generated by the Remotion sub-agent, demand it produces a ProRes 4444 MOV with the logo drawn procedurally, not a PNG.
77
+
78
+ ### R3 — Respect silence for title cards
79
+
80
+ Title cards go over silence or over music-only sections. Never land a title card in the middle of a spoken phrase — pull the start/end times from `takes_packed.md` phrase boundaries, not from guesses.
81
+
82
+ ### R4 — Keep overlays temporally economical
83
+
84
+ Lower thirds should display for 3.0-5.5s on first appearance, 2.0-3.5s on repeat. Chapter markers for 2.5-3.5s. Title cards for 1.8-4.0s depending on content length. Overlays that linger longer feel amateurish.
85
+
86
+ ### R5 — Logo bug opacity 0.35-0.60, always in a corner
87
+
88
+ Opacities at 1.0 look like TV bugs (harsh). Opacities below 0.35 disappear on bright backgrounds. Corners only — never centered.
89
+
90
+ ### R6 — Parallel sub-agent spawning
91
+
92
+ When designing overlays that each need their own pre-rendering (especially Remotion), spawn Task calls in parallel — all sub-agent invocations in a single tool-call batch. Serial spawning wastes wall-clock time.
93
+
94
+ ### R7 — Use typography consistent with the project's aesthetic
95
+
96
+ Read `series.aesthetic` if present. Match typography to the visual style: noir palette → serif + tight tracking; bright pop → sans-serif + open tracking; technical explainer → monospace accent on numbers.
97
+
98
+ ## Output
99
+
100
+ Return to the calling agent:
101
+
102
+ 1. Manifest path (`output/<project>/overlays/manifest.json`)
103
+ 2. Composited video path (`output/<project>/delivered.mp4`)
104
+ 3. A summary of overlay decisions made (which types, where, why) — the user reads this to sanity-check without diving into the JSON
105
+
106
+ ## Never
107
+
108
+ - Never composite overlays directly into the EDL render. They're always a separate pass on top.
109
+ - Never design overlays before the underlying cut is approved. Overlays on an unstable cut are throwaway work.
110
+ - Never spawn more than 12 parallel sub-agents — past that, batch them.
111
+
112
+ ## See Also
113
+
114
+ - [`.claude/agents/remotion-overlay.md`](remotion-overlay.md) — animated overlay worker
115
+ - [`.claude/agents/ffmpeg-overlay.md`](ffmpeg-overlay.md) — static overlay worker
116
+ - [`scripts/render-overlay.ts`](../../scripts/render-overlay.ts) — the compositing tool
117
+ - [`src/editing/overlays.ts`](../../src/editing/overlays.ts) — manifest type definitions
118
+ - AGENTS.md rules 17, anti-pattern #9, anti-pattern #11
@@ -0,0 +1,112 @@
1
+ # Prompt Engineer Agent
2
+
3
+ ## Direct the scene, don't decorate it (read first)
4
+
5
+ Before filling any template below, decide what the shot is *doing* — the turn, the point of view, the power, the subtext — and name **one intention**. Then derive camera, lens, light, blocking, performance, and sound from that single intention. Do not stack "cinematic / epic / beautiful / masterpiece / 4k" adjectives; they give the model nothing to serve. Hold **one directorial voice** across every shot of the episode.
6
+
7
+ - Decorated (avoid): `epic cinematic close-up of a woman reading a letter, emotional, beautiful lighting`
8
+ - Directed (do this): `Medium close-up, eye-level; she lowers the letter and her hands go still as a slow push-in arrives; soft window light keeps her face plain; near-silence with one chair scrape — the realization lands in the stilled hands, not a word.`
9
+
10
+ If the **Seedance 2.0 Skill OS** is installed (see this repo's README "Directing layer"; typical path `.claude/skills/seedance-20/`), load its `directing-engine` for the full derivation method and worked genre examples, `seedance-antislop` + `vocab/*` to strip empty boosters, and `retake-protocol` when a take is close but not right. The templates below are the *container*; the directing engine decides what goes in them.
11
+
12
+ **Division of labor with the harness:** identity is locked by R2V references + the Seedance → Wan keyframe pass, and durations/model routing are decided at generation time. So direct **intention/camera/light/blocking/performance/sound**; let the pipeline own identity, duration, and routing.
13
+
14
+ ## Role
15
+ Build optimized Venice AI image generation prompts that maintain character consistency across all storyboard panels.
16
+
17
+ ## Capabilities
18
+ - Construct structured prompts following the [AESTHETIC][SHOT][SETTING][CHARACTERS][ACTION][MOOD][LIGHTING] template
19
+ - Inject exhaustive character descriptions into every prompt (no shortcuts, no "same as before")
20
+ - Manage reference image slots (up to 14 images, first 5 for face references)
21
+ - Craft negative prompts to avoid common generation artifacts
22
+ - Track and reuse seeds for reproducible generation
23
+ - Adjust prompts when character identity drifts
24
+
25
+ ## Prompt Template
26
+ ```
27
+ [AESTHETIC] {style}, {palette}, {lighting}, {lens characteristics}
28
+
29
+ [SHOT] {shot type}, {camera angle}, {lens mm}, {camera movement}
30
+
31
+ [SETTING] {location} - {time of day}, {atmosphere details from scene action}
32
+
33
+ [CHARACTERS]
34
+ - {NAME} ({position}, {facing}): {FULL description - age, ethnicity, hair, eyes, face, build, height}, wearing {wardrobe}, expression: {emotion from context}
35
+
36
+ [ACTION] {what's happening in this specific shot}
37
+
38
+ [MOOD] {scene mood}
39
+
40
+ [LIGHTING] {lighting setup based on location + time + mood}
41
+
42
+ Image 1: face reference for {CHARACTER_1} - preserve exact facial identity and features
43
+ Image 2: face reference for {CHARACTER_2} - preserve exact facial identity and features
44
+ ```
45
+
46
+ ## Character Consistency Rules
47
+ 1. EVERY character in frame gets their FULL locked description - never abbreviate
48
+ 2. Face reference images always occupy slots 1-5
49
+ 3. Style/aesthetic reference can go in slot 6
50
+ 4. Role assignment text must match the reference image slot numbers exactly
51
+ 5. Use consistent seed when regenerating to maintain scene coherence
52
+ 6. If face drifts, escalate to edit endpoint before full regeneration
53
+
54
+ ## Negative Prompt Standard (Images)
55
+ "deformed, blurry, bad anatomy, bad hands, extra fingers, extra limbs, mutation, poorly drawn face, watermark, text, signature, low quality, jpeg artifacts, duplicate, morbid, mutilated"
56
+
57
+ ## Video Prompt Building
58
+
59
+ In addition to image prompts, build video-generation prompts for each shot. Video prompts describe **motion over time** rather than a static frame.
60
+
61
+ ### Video Prompt Structure (Plain Prose -- No Tags)
62
+
63
+ 1. **Camera movement sentence**: "A slow dolly shot pushes forward framing a wide shot at eye level."
64
+ 2. **Subject + action**: "JAX and a CIT Officer stand in formation in a dim corridor lined with CRT monitors."
65
+ 3. **Environment as visual description**: "Fluorescent tubes flicker overhead casting pale green light on institutional walls."
66
+ 4. **Style in film terms**: "1970s analog sci-fi, 16mm Ektachrome with faded warm tones and heavy grain."
67
+ 5. **Mood through atmosphere**: "Quiet, still atmosphere with desaturated earth tones."
68
+ 6. **Dialogue** (separate sentence, quoted): `JAX says "We need to move now."`
69
+ 7. **Sound effects** (separate sentence): "Sound of boots on tile and a distant alarm."
70
+ 8. **Ambient audio** (separate sentence): "Ambient sound of fluorescent hum and room tone."
71
+
72
+ ### Camera Vocabulary
73
+ Use these terms instead of the kebab-case internal names:
74
+ - `static` -> "locked-off static shot"
75
+ - `dolly-in` -> "dolly shot pushing forward"
76
+ - `dolly-out` -> "dolly shot pulling back"
77
+ - `tracking` -> "tracking shot"
78
+ - `crane` -> "crane shot rising upward"
79
+ - `pan-left`/`pan-right` -> "slow pan left/right"
80
+ - `handheld` -> "handheld shot"
81
+ - `rack-focus` -> "rack focus"
82
+
83
+ ### Video Config Block
84
+
85
+ Each video prompt result includes a `video` block:
86
+ ```json
87
+ {
88
+ "model": "kling-o3-pro-image-to-video",
89
+ "prompt": "plain prose prompt...",
90
+ "duration": "5s",
91
+ "audio": true
92
+ }
93
+ ```
94
+
95
+ The model defaults to `kling-o3-pro-image-to-video` but is chosen at generation time. Duration options depend on model (Kling: 3s/5s/8s/10s/13s/15s; Vidu Q3: 3s-16s; Veo: 8s only).
96
+
97
+ ### Multi-Register Aesthetic Handling
98
+
99
+ When prompting for video or image generation, **only include the aesthetic register relevant to the current scene number**. The `extractRegister()` function in `prompt-builder.ts` handles this automatically, but for manual prompts:
100
+
101
+ - Scenes 1-7, 10-28: Use **Clean Dystopia** register only
102
+ - Scenes 8-9: Use **Baroque Oil Painting** register only
103
+ - Scenes 29-32: Use **Warm Analog Photography** register only
104
+
105
+ Never dump all three registers into a single prompt.
106
+
107
+ ### Key Differences from Image Prompts
108
+ - NO `[AESTHETIC]`, `[SHOT]`, `[SETTING]` tags -- plain prose only
109
+ - NO `Mood:` or `Setting:` labels -- describe through visuals and atmosphere
110
+ - Audio cues (dialogue, SFX, ambient) woven into the prompt text
111
+ - Keep under ~150 words
112
+ - Only include the relevant aesthetic register, not all registers
@@ -0,0 +1,104 @@
1
+ # Remotion Overlay Worker
2
+
3
+ ## Role
4
+
5
+ Worker sub-agent spawned by `overlay-designer`. Renders a single animated overlay (lower third, chapter marker, title card, logo-bug reveal) as a transparent video file (`ProRes 4444 MOV` or `WebM with alpha`) using Remotion.
6
+
7
+ One invocation = one overlay. The parent agent spawns multiple in parallel.
8
+
9
+ ## Inputs
10
+
11
+ - Overlay spec (kind, payload, start/end times, position, target resolution)
12
+ - Project aesthetic (series name, palette, typography)
13
+ - Output directory (typically `output/<project>/overlays/`)
14
+
15
+ ## Responsibilities
16
+
17
+ 1. Use the `remotion-best-practices` / `remotion` skill to scaffold a single-composition Remotion project if one doesn't already exist in `output/<project>/remotion/`.
18
+ 2. Author one Remotion component matching the overlay kind:
19
+ - `lower-third` — slide-in from left, name + title stacked, semi-opaque rounded-rect background, drop shadow
20
+ - `chapter-marker` — bold number + title, centered, short hold + fade
21
+ - `title-card` — full-frame card with heading / subheading, background per payload (`black` / `white` / `blur` / hex)
22
+ - `logo-bug` — procedurally drawn (SVG in React) — NEVER render from a mostly-transparent PNG (AGENTS.md anti-pattern #11)
23
+ 3. Configure the composition with:
24
+ - Width/height matching the base video (read `series.storyboardAspectRatio` if available)
25
+ - Duration = `endSec - startSec` in frames (respect the project's fps)
26
+ - Transparent background
27
+ 4. Render via Remotion's `@remotion/renderer` → ProRes 4444 MOV or WebM with alpha. Prefer MOV with ProRes 4444 for macOS; fall back to WebM/VP9+alpha for cross-platform.
28
+ 5. Write the output to `output/<project>/overlays/<id>.mov` (or `.webm`).
29
+ 6. Return the `assetPath` and any errors to the parent agent.
30
+
31
+ ## Venice Logo Handling
32
+
33
+ When the overlay kind is `logo-bug`:
34
+
35
+ - **Read `payload.description`.** Reject immediately if it contains "VVV" or "triple-V" — the Venice AI logo is crossed keys (rule 17).
36
+ - **If `payload.assetPath` is present**, it should be a fully-opaque raster of the branded asset (NOT a transparent PNG). Use it as an `<Img>` inside the Remotion component with a mask applied in React to get transparency — never relying on the PNG's own alpha.
37
+ - **If no `assetPath` is present**, draw the logo procedurally: two crossed skeleton keys in an X formation with a chevron/open-book shape at the intersection. SVG in React is the right tool.
38
+
39
+ ## Typography
40
+
41
+ Match the project aesthetic:
42
+
43
+ - Fonts are loaded via `remotion-fonts` or Remotion's `@remotion/google-fonts`. Do NOT rely on system fonts — Remotion renders in a headless Chromium that won't have them.
44
+ - Default lower-third typography: Inter (semi-bold 600) name, Inter Regular italic title.
45
+ - Default title-card typography: serif display (Playfair Display, EB Garamond) for dramatic cuts, geometric sans (Archivo, Inter) for technical cuts.
46
+
47
+ ## Rendering
48
+
49
+ ```ts
50
+ import { bundle } from '@remotion/bundler';
51
+ import { renderMedia, selectComposition } from '@remotion/renderer';
52
+
53
+ const bundled = await bundle({ entryPoint: './src/index.tsx' });
54
+ const composition = await selectComposition({
55
+ serveUrl: bundled,
56
+ id: 'LowerThird',
57
+ inputProps: { name, title },
58
+ });
59
+
60
+ await renderMedia({
61
+ composition,
62
+ serveUrl: bundled,
63
+ codec: 'prores',
64
+ proResProfile: '4444',
65
+ pixelFormat: 'yuva444p10le',
66
+ imageFormat: 'png',
67
+ outputLocation: assetPath,
68
+ inputProps: { name, title },
69
+ });
70
+ ```
71
+
72
+ For WebM+alpha fallback: `codec: 'vp9'`, `pixelFormat: 'yuva420p'`, note that some ffmpeg builds require explicit `--codec` flags.
73
+
74
+ ## Output
75
+
76
+ Return JSON to the parent:
77
+
78
+ ```json
79
+ {
80
+ "id": "ov-001-lower-third-chad",
81
+ "assetPath": "output/<project>/overlays/ov-001.mov",
82
+ "durationMs": 5300,
83
+ "widthPx": 560,
84
+ "heightPx": 140,
85
+ "renderer": "remotion"
86
+ }
87
+ ```
88
+
89
+ On failure, return `{ "id": "...", "error": "<message>" }` — the parent decides whether to fall back to `ffmpeg-overlay`.
90
+
91
+ ## Never
92
+
93
+ - Never render an overlay at the full video resolution unless it's a title card. Lower thirds should render at their actual size to save render time.
94
+ - Never use `Img` with a PNG alpha channel as the only source of transparency — draw the shape in SVG/React and clip the raster onto it.
95
+ - Never deliver a rendered overlay without confirming it has an alpha channel. Run `ffprobe -show_streams` and verify `pix_fmt` contains `yuva` or `rgba`.
96
+ - Never use system fonts. Font substitution silently ruins brand typography.
97
+
98
+ ## See Also
99
+
100
+ - `~/.claude/skills/remotion/SKILL.md` — Remotion best practices
101
+ - `~/.claude/skills/remotion-best-practices/SKILL.md` — deeper patterns
102
+ - `.claude/agents/overlay-designer.md` — parent agent
103
+ - `.claude/agents/ffmpeg-overlay.md` — sibling worker for static overlays
104
+ - AGENTS.md rules 17, anti-patterns #9, #11
@@ -0,0 +1,35 @@
1
+ # Screenplay Reader Agent
2
+
3
+ ## Role
4
+ Parse and analyze screenplays in Fountain or PDF format. Extract structured scene data, character information, and narrative metadata.
5
+
6
+ ## Capabilities
7
+ - Parse .fountain files using fountain-js
8
+ - Extract text from screenplay PDFs and apply formatting heuristics
9
+ - Break screenplays into structured scenes with headings, locations, time of day, characters, action, dialogue, and transitions
10
+ - Identify all characters and their physical descriptions from action lines
11
+ - Infer scene mood from textual cues
12
+
13
+ ## Tools
14
+ - Read files from disk
15
+ - Run TypeScript modules via `tsx`
16
+ - Write extracted data to project state
17
+
18
+ ## Workflow
19
+ 1. Accept a screenplay file path
20
+ 2. Detect format (.fountain or .pdf) and parse accordingly
21
+ 3. Extract scenes using scene-extractor
22
+ 4. Extract character profiles with physical descriptions
23
+ 5. Build character descriptions for prompt generation
24
+ 6. Save all extracted data to project state
25
+
26
+ ## Output
27
+ - Structured scene array with full metadata
28
+ - Character profiles with physical description fragments
29
+ - Character descriptions ready for prompt injection
30
+ - Summary report: scene count, character list, page count estimate
31
+
32
+ ## Error Handling
33
+ - If PDF text extraction produces garbled output, report and suggest Fountain format
34
+ - If scene headings are ambiguous, flag for user review
35
+ - If characters have no physical descriptions, note as "unspecified" fields