dreamcontext 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +2 -1
  2. package/agents/sleep-product.md +19 -2
  3. package/agents/sleep-state.md +43 -22
  4. package/agents/sleep-tasks.md +18 -0
  5. package/dist/agents/sleep-product.md +19 -2
  6. package/dist/agents/sleep-state.md +43 -22
  7. package/dist/agents/sleep-tasks.md +18 -0
  8. package/dist/dashboard/assets/{BrainCanvas3D-LLqVeXtb.js → BrainCanvas3D-lFgJbbhZ.js} +1 -1
  9. package/dist/dashboard/assets/{_baseUniq-DW0uA0ty.js → _baseUniq-BpANgc_i.js} +1 -1
  10. package/dist/dashboard/assets/{arc-fVUGtlbL.js → arc-CX32Jm7E.js} +1 -1
  11. package/dist/dashboard/assets/{architectureDiagram-Q4EWVU46-Dz2PKBsS.js → architectureDiagram-Q4EWVU46-ARASlGxO.js} +1 -1
  12. package/dist/dashboard/assets/{blockDiagram-DXYQGD6D-ByWKxzhL.js → blockDiagram-DXYQGD6D-BBvsYm9E.js} +1 -1
  13. package/dist/dashboard/assets/{c4Diagram-AHTNJAMY-dpHsVM3D.js → c4Diagram-AHTNJAMY-ChSUfXR9.js} +1 -1
  14. package/dist/dashboard/assets/channel-w_Bp182N.js +1 -0
  15. package/dist/dashboard/assets/{chunk-4BX2VUAB-HEXb6Yg5.js → chunk-4BX2VUAB-COHoVEpt.js} +1 -1
  16. package/dist/dashboard/assets/{chunk-4TB4RGXK-DeVy5g6H.js → chunk-4TB4RGXK-DKJFLaTC.js} +1 -1
  17. package/dist/dashboard/assets/{chunk-55IACEB6-DGy3ZgDZ.js → chunk-55IACEB6-CKZPWZRu.js} +1 -1
  18. package/dist/dashboard/assets/{chunk-EDXVE4YY-CYJehjz4.js → chunk-EDXVE4YY-k1y4mGUJ.js} +1 -1
  19. package/dist/dashboard/assets/{chunk-FMBD7UC4-DKTOmvgZ.js → chunk-FMBD7UC4-ixXe5R10.js} +1 -1
  20. package/dist/dashboard/assets/{chunk-OYMX7WX6-mmtWyiDA.js → chunk-OYMX7WX6-NDFap7xg.js} +1 -1
  21. package/dist/dashboard/assets/{chunk-QZHKN3VN-DSdV0iwY.js → chunk-QZHKN3VN-vY0UpHjB.js} +1 -1
  22. package/dist/dashboard/assets/{chunk-YZCP3GAM-DbY-KoDn.js → chunk-YZCP3GAM-yztvsKR-.js} +1 -1
  23. package/dist/dashboard/assets/classDiagram-6PBFFD2Q-CcQxcqy9.js +1 -0
  24. package/dist/dashboard/assets/classDiagram-v2-HSJHXN6E-CcQxcqy9.js +1 -0
  25. package/dist/dashboard/assets/clone-_znoR_ci.js +1 -0
  26. package/dist/dashboard/assets/{cose-bilkent-S5V4N54A-p0t1DY88.js → cose-bilkent-S5V4N54A-DKXM5Fh5.js} +1 -1
  27. package/dist/dashboard/assets/{dagre-KV5264BT-DYE0WzHM.js → dagre-KV5264BT-DN1Nlsmy.js} +1 -1
  28. package/dist/dashboard/assets/{diagram-5BDNPKRD-D9EiQCOP.js → diagram-5BDNPKRD-DnoJFRqR.js} +1 -1
  29. package/dist/dashboard/assets/{diagram-G4DWMVQ6-poDXSAfu.js → diagram-G4DWMVQ6-CF1_jTNI.js} +1 -1
  30. package/dist/dashboard/assets/{diagram-MMDJMWI5-CHZ7wgN1.js → diagram-MMDJMWI5-D4-bZZEt.js} +1 -1
  31. package/dist/dashboard/assets/{diagram-TYMM5635-DVk6fYG7.js → diagram-TYMM5635-al8RqVgf.js} +1 -1
  32. package/dist/dashboard/assets/{erDiagram-SMLLAGMA-D2sqkGin.js → erDiagram-SMLLAGMA-MEcC0rwO.js} +1 -1
  33. package/dist/dashboard/assets/{flowDiagram-DWJPFMVM-DrMlKBYA.js → flowDiagram-DWJPFMVM-DCLyNpW8.js} +1 -1
  34. package/dist/dashboard/assets/{ganttDiagram-T4ZO3ILL-Cf4FbEFp.js → ganttDiagram-T4ZO3ILL-DfEvbbJK.js} +1 -1
  35. package/dist/dashboard/assets/{gitGraphDiagram-UUTBAWPF-BsbOq1C9.js → gitGraphDiagram-UUTBAWPF-C_YozdVL.js} +1 -1
  36. package/dist/dashboard/assets/{graph-CUxxgCtS.js → graph-DpIXS1G1.js} +1 -1
  37. package/dist/dashboard/assets/index-Bo5CUa_M.js +480 -0
  38. package/dist/dashboard/assets/index-DrqurW1c.css +1 -0
  39. package/dist/dashboard/assets/{infoDiagram-42DDH7IO-DTkMnZiD.js → infoDiagram-42DDH7IO-AdT6kjzj.js} +1 -1
  40. package/dist/dashboard/assets/{ishikawaDiagram-UXIWVN3A-CahQ348K.js → ishikawaDiagram-UXIWVN3A-B0_9IZVO.js} +1 -1
  41. package/dist/dashboard/assets/{journeyDiagram-VCZTEJTY-BDY2Nslc.js → journeyDiagram-VCZTEJTY-BveNBswQ.js} +1 -1
  42. package/dist/dashboard/assets/{kanban-definition-6JOO6SKY-CgrwoTjI.js → kanban-definition-6JOO6SKY-CFI8j4jR.js} +1 -1
  43. package/dist/dashboard/assets/{layout-IaUxkFkm.js → layout-DI7XjZy1.js} +1 -1
  44. package/dist/dashboard/assets/{linear-shc0iNFn.js → linear-CSjp56iw.js} +1 -1
  45. package/dist/dashboard/assets/{min-BIL7YgTN.js → min-pAGUJmEC.js} +1 -1
  46. package/dist/dashboard/assets/{mindmap-definition-QFDTVHPH-GmO0JRtA.js → mindmap-definition-QFDTVHPH-C2mBnknr.js} +1 -1
  47. package/dist/dashboard/assets/{pieDiagram-DEJITSTG-Bfm_toYA.js → pieDiagram-DEJITSTG-i_phWDqD.js} +1 -1
  48. package/dist/dashboard/assets/{quadrantDiagram-34T5L4WZ-DIjTM3lm.js → quadrantDiagram-34T5L4WZ-Dv7TGJjw.js} +1 -1
  49. package/dist/dashboard/assets/{requirementDiagram-MS252O5E-B9AEoGYh.js → requirementDiagram-MS252O5E-CB2Jl-O5.js} +1 -1
  50. package/dist/dashboard/assets/{sankeyDiagram-XADWPNL6-C_O5XLXm.js → sankeyDiagram-XADWPNL6-DxCoN-EI.js} +1 -1
  51. package/dist/dashboard/assets/{sequenceDiagram-FGHM5R23-CHSxSvmJ.js → sequenceDiagram-FGHM5R23-BeZyaehJ.js} +1 -1
  52. package/dist/dashboard/assets/{stateDiagram-FHFEXIEX-B0URrhHw.js → stateDiagram-FHFEXIEX-D3AjQzD1.js} +1 -1
  53. package/dist/dashboard/assets/stateDiagram-v2-QKLJ7IA2-Ci9v--xK.js +1 -0
  54. package/dist/dashboard/assets/{timeline-definition-GMOUNBTQ-CycU2gVC.js → timeline-definition-GMOUNBTQ-3l_8RFUg.js} +1 -1
  55. package/dist/dashboard/assets/{vennDiagram-DHZGUBPP-DcU2536G.js → vennDiagram-DHZGUBPP-2QiY25JD.js} +1 -1
  56. package/dist/dashboard/assets/{wardley-RL74JXVD-BWqqaX3S.js → wardley-RL74JXVD-DNLmFofz.js} +1 -1
  57. package/dist/dashboard/assets/{wardleyDiagram-NUSXRM2D-CUnIHJgd.js → wardleyDiagram-NUSXRM2D-BI0tupiS.js} +1 -1
  58. package/dist/dashboard/assets/{xychartDiagram-5P7HB3ND-b9fcOYTj.js → xychartDiagram-5P7HB3ND-CghXPE7_.js} +1 -1
  59. package/dist/dashboard/index.html +2 -2
  60. package/dist/index.js +1669 -1064
  61. package/dist/skill-packs/excalidraw/SKILL.md +82 -3
  62. package/dist/skill-packs/video-watching/SKILL.md +54 -13
  63. package/dist/skill-packs/video-watching/scripts/build_frame_index.py +61 -10
  64. package/dist/skill-packs/video-watching/scripts/transcribe.sh +147 -54
  65. package/dist/templates/init/data-structures/default.md +26 -24
  66. package/package.json +1 -1
  67. package/skill/SKILL.md +26 -12
  68. package/skill-packs/excalidraw/SKILL.md +82 -3
  69. package/skill-packs/video-watching/SKILL.md +54 -13
  70. package/skill-packs/video-watching/scripts/build_frame_index.py +61 -10
  71. package/skill-packs/video-watching/scripts/transcribe.sh +147 -54
  72. package/dist/dashboard/assets/channel-zrLwggBX.js +0 -1
  73. package/dist/dashboard/assets/classDiagram-6PBFFD2Q-BvejNiwH.js +0 -1
  74. package/dist/dashboard/assets/classDiagram-v2-HSJHXN6E-BvejNiwH.js +0 -1
  75. package/dist/dashboard/assets/clone-BFVdml6g.js +0 -1
  76. package/dist/dashboard/assets/index-CqSkXBSu.css +0 -1
  77. package/dist/dashboard/assets/index-flRpQtDj.js +0 -476
  78. package/dist/dashboard/assets/stateDiagram-v2-QKLJ7IA2-FdAb9MBo.js +0 -1
  79. package/dist/skill-packs/video-watching/scripts/gap_fill.py +0 -45
  80. package/skill-packs/video-watching/scripts/gap_fill.py +0 -45
@@ -32,14 +32,90 @@ Write a spec JSON, then run it. The script prints `elements/images/texts` counts
32
32
 
33
33
  ### JS API (for pipelines that generate many boards)
34
34
  ```js
35
- const { buildExcalidraw, lane, grid } = require('.../scripts/build_excalidraw.js');
36
- buildExcalidraw({ out, elements: [ ...lane({ title, images, x, y, thumbW }) ] });
35
+ const path = require('path');
36
+ // skill lives at <project>/.claude/skills/excalidraw/ adjust leading ../ count to match your script's depth from project root
37
+ const { buildExcalidraw, lane, grid } = require(path.resolve(__dirname, '../.claude/skills/excalidraw/scripts/build_excalidraw.js'));
38
+ buildExcalidraw({ out: path.resolve(__dirname, '../boards/Board.excalidraw.md'), elements: [ ...lane({ title, images, x, y, thumbW }) ] });
37
39
  ```
38
40
 
41
+ ## File layout
42
+
43
+ ### Single board (default)
44
+ Keep the spec next to the generated board but clearly separated:
45
+ ```
46
+ boards/
47
+ ├── MyBoard.excalidraw.md ← generated deliverable; do not hand-edit
48
+ └── _spec/
49
+ └── MyBoard.json ← source of truth; edit this, then regenerate
50
+ ```
51
+ The `.excalidraw.md` is **disposable** — it is fully derived from the spec. If the two ever
52
+ disagree, the spec wins. Commit both (the board for Obsidian/GitHub preview, the spec for
53
+ reproducibility), but only edit the spec.
54
+
55
+ ### Multi-board pipeline
56
+ When a single generator produces several boards, isolate it in a `pipeline/` folder so the
57
+ deliverable boards stay at the top of the project and are easy to open in Obsidian:
58
+ ```
59
+ boards/
60
+ ├── Overview.excalidraw.md ← generated
61
+ ├── Funnel.excalidraw.md ← generated
62
+ ├── Pricing.excalidraw.md ← generated
63
+ └── pipeline/
64
+ ├── generate.js ← single regen entrypoint: `node pipeline/generate.js`
65
+ ├── shared-style.js ← shared palette / helpers
66
+ └── spec/
67
+ ├── Overview.json ← source spec for Overview board
68
+ ├── Funnel.json ← source spec for Funnel board
69
+ └── Pricing.json ← source spec for Pricing board
70
+ ```
71
+ - **Generated files** (`*.excalidraw.md`) live one level above `pipeline/` — open them in Obsidian without navigating into a sub-folder.
72
+ - **Source specs** live in `pipeline/spec/` — one JSON per board.
73
+ - **Single entrypoint**: `node pipeline/generate.js` rebuilds every board. No per-board manual commands.
74
+
75
+ ### Many boards from shared data (recipe)
76
+ Use this pattern when multiple boards pull from the same data set (e.g. one board per product, per region, or per funnel step):
77
+
78
+ ```js
79
+ // pipeline/generate.js (lives at boards/pipeline/generate.js)
80
+ const path = require('path');
81
+ const ROOT = path.resolve(__dirname, '..'); // boards/ directory
82
+ // ../../ = project root (boards/ → project/); adjust if boards/ is nested deeper
83
+ const { buildExcalidraw, lane } = require(path.resolve(__dirname, '../../.claude/skills/excalidraw/scripts/build_excalidraw.js'));
84
+ const style = require(path.resolve(__dirname, 'shared-style.js'));
85
+ const items = require(path.resolve(__dirname, 'spec/items.json')); // shared data
86
+
87
+ for (const item of items) {
88
+ const elements = style.buildItemBoard(item); // per-item spec logic
89
+ buildExcalidraw({
90
+ out: path.resolve(ROOT, `${item.slug}.excalidraw.md`),
91
+ elements,
92
+ });
93
+ console.log('wrote', item.slug);
94
+ }
95
+ ```
96
+
97
+ ```js
98
+ // pipeline/shared-style.js (lives at boards/pipeline/shared-style.js)
99
+ const path = require('path');
100
+ // ../../ = project root (boards/ → project/); adjust if boards/ is nested deeper
101
+ const { card, connector, sectionTitle } = require(path.resolve(__dirname, '../../.claude/skills/excalidraw/scripts/lib/style.js'));
102
+
103
+ exports.buildItemBoard = (item) => [
104
+ sectionTitle({ x: 0, y: 0, text: item.name, fontSize: 40 }),
105
+ // … common layout using item fields
106
+ ];
107
+ ```
108
+
109
+ Key conventions:
110
+ - `ROOT = path.resolve(__dirname, '..')` pins paths relative to the generator file, not the working directory. The generator works correctly wherever it is invoked from.
111
+ - Each item produces exactly one board; the mapping is `items.json → <slug>.excalidraw.md`.
112
+ - `shared-style.js` owns the layout logic — boards stay visually consistent; change the style once, regenerate all.
113
+ - Add a `package.json` script or `Makefile` alias so the command is always `npm run boards` (or similar) and never has to be rediscovered.
114
+
39
115
  ## Spec schema
40
116
  ```jsonc
41
117
  {
42
- "out": "/abs/path/Board.excalidraw.md", // required (or pass --out)
118
+ "out": "./boards/Board.excalidraw.md", // prefer __dirname-relative in JS generators; relative to cwd for CLI
43
119
  "vaultRoot": "/abs/vault", // optional; auto-detected by walking up to `.obsidian`
44
120
  "attachDir": "Attachments", // external images get copied here (relative to board dir)
45
121
  "wikilinkMode": "basename", // "basename" (default) or "path" (vault-relative)
@@ -48,6 +124,9 @@ buildExcalidraw({ out, elements: [ ...lane({ title, images, x, y, thumbW }) ] })
48
124
  }
49
125
  ```
50
126
 
127
+ **Paths in generators**: always use `path.resolve(__dirname, ...)` for `out` and image `path` fields — never
128
+ hardcode absolute paths. This keeps the generator portable: move the folder and it still runs.
129
+
51
130
  ### Element types
52
131
  - `text` — `{ x, y, text, fontSize?, color?, width?, align?, fontFamily? }` (fontFamily 1=hand, 2=normal, 3=code). **Set `width` for any caption/label that must stay inside a column or card** → the text WRAPS to that width (autoResize off) and its height is computed from the wrapped line count. Omit `width` only for short single-line text you want sized to content (it renders on one line and will overlap neighbours if long).
53
132
  - `image` — `{ x, y, path, width? , height? }` — give ONE of width/height; the other is derived from aspect. `path` is an absolute file path.
@@ -39,15 +39,36 @@ if it's installed.
39
39
  # language auto-detects — no need to pass it. Override only if auto mislabels a
40
40
  # short/ambiguous clip: ./transcribe.sh "/abs/path/clip.mp4" tr
41
41
  ```
42
+
43
+ **Pick the mode for the kind of video — this matters most for app/UI recordings:**
44
+ - **Talking-head / lecture / ad creative** → the default is right. Scene-detect + a
45
+ 10s gap-fill catches the visuals.
46
+ - **App screen-recording / onboarding funnel / UI walkthrough** → add `--mode ui`.
47
+ App screens linger 3–5s and change by **text only** (a questionnaire step, a
48
+ paywall) — they don't move enough to trip scene-detect, so the 10s default
49
+ silently drops most of them. `--mode ui` samples every ~2.5s so each screen lands.
50
+ If you only need the screens (no narration), add `--frames-only` to skip whisper:
51
+ ```bash
52
+ ./scripts/transcribe.sh "/abs/path/onboarding.mp4" --mode ui --frames-only --contact-sheet
53
+ ```
54
+
42
55
  This writes everything into `<video_dir>/<slug>.media/`:
43
- - `transcript.srt` / `.json` / `.txt` — timestamped transcript (large-v3-turbo)
56
+ - `transcript.srt` / `.json` / `.txt` — timestamped transcript (large-v3-turbo) — *skipped with `--frames-only`*
44
57
  - `frames/anchor_first.jpg` / `anchor_last.jpg` — first + last frame, always captured
45
58
  (short cut-heavy creatives carry the hook and CTA here; scene-detect misses both)
46
- - `frames/scene_*.jpg` — frames at each scene change (slides/UI transitions)
47
- - `frames/gap_*.jpg` fill frames so no stretch > 10s goes unsampled (catches
48
- static-but-important sections scene-detect misses: app demos, slides, CTAs)
59
+ - `frames/frame_*.jpg` — frames selected in **one time-based pass**: a scene change
60
+ fired (slide/UI transition) **or** `MAX_GAP` seconds elapsed since the last frame,
61
+ whichever comes first. The gap rule is wall-clock based (`prev_selected_t`), so it
62
+ works on variable-frame-rate screen recordings where frame-number sampling breaks,
63
+ and it guarantees every static stretch (app demos, slides, CTAs) gets a frame.
64
+ - `frames/contact_sheet.jpg` — tiled montage of all frames, *only with `--contact-sheet`*
49
65
  - `frames.json` — **the index you read**: `[{file, t, at, type}]`, sorted by time,
50
- near-duplicate timestamps collapsed (anchors always kept)
66
+ near-duplicate timestamps collapsed (anchors always kept). `type` is `scene` (a
67
+ picture change fired), `gap` (a periodic fill at the `MAX_GAP` cadence), or `anchor`.
68
+
69
+ The engine prints a **coverage check** at the end: `longest unsampled gap = Xs`. If it
70
+ warns the gap is >2× `MAX_GAP`, frames are likely missing — re-run denser (`--max-gap`
71
+ lower, or `--mode ui`) **before** any expensive deep-analysis pass.
51
72
 
52
73
  `audio.wav` is auto-deleted after transcription (it's a ~1.9MB/min whisper-only
53
74
  intermediate). The whole `*.media/` dir is gitignored — it stays next to the video
@@ -69,7 +90,9 @@ nothing on screen need no frame.
69
90
  > Heuristic: the two `anchor` frames (first/last) almost always matter — the hook
70
91
  > and the CTA. Every `scene` frame is a candidate (the picture changed for a
71
92
  > reason). `gap` frames cover static stretches scene-detect skipped — often the
72
- > most informative part (an app demo or slide that doesn't "move"), so check them.
93
+ > most informative part (an app demo or onboarding screen that doesn't "move"), so
94
+ > check them. In `--mode ui` runs most frames are `gap` — that's expected and you
95
+ > generally want to look at all of them, one per screen.
73
96
 
74
97
  **Need a frame the index doesn't have? Grab it on demand.** Scene-detect fires on
75
98
  motion, not on meaning — on fast-cut video the most informative moment often sits
@@ -116,6 +139,14 @@ relevant dreamcontext skill (per the skill-triage rule) and load it:
116
139
  - **Knowledge / training** → if it should persist for the project, hand the transcript to `dreamcontext knowledge` so it becomes durable context
117
140
  Don't guess the use-case — let the user direct it.
118
141
 
142
+ > **UI teardown (no transcript needed).** When the goal is purely the app flow —
143
+ > e.g. tearing down a competitor's onboarding funnel — run `--frames-only --mode ui`
144
+ > and skip the `.transcript.md` entirely. The deliverable becomes a **curated screen
145
+ > list** (one entry per onboarding step, in order, from `frames.json`), which feeds a
146
+ > board (`excalidraw`) or an `onboarding-design` analysis directly. Use `--contact-sheet`
147
+ > to eyeball coverage first, and trust the coverage warning — under-sampling here is
148
+ > exactly what makes a teardown wrongly report "screen X wasn't shown".
149
+
119
150
  ## What this skill is NOT
120
151
  - Not a knowledge-base writer or ingestion pipeline. This skill produces the
121
152
  transcript artifact; persisting it (chunking, embedding, ingest into a project's
@@ -140,12 +171,22 @@ brew install whisper-cpp ffmpeg # yt-dlp too, only for remote links
140
171
  ├── SKILL.md ← you are here
141
172
  └── scripts/
142
173
  ├── transcribe.sh ← video → transcript + frames + frames.json (the engine)
143
- ├── gap_fill.py timestamps to fill scene-detect gaps (> MAX_GAP)
144
- └── build_frame_index.py ← frames/*.jpg → frames.json (pts_time index)
174
+ └── build_frame_index.py frames/*.jpg frames.json (pts_time index + coverage check)
145
175
  ```
146
176
 
147
- ## Tuning
148
- - Too few frames on a slide-heavy video? `SCENE_THRESHOLD=0.1 ./transcribe.sh …`
149
- - Long static lecture over-sampled? Raise `MAX_GAP=20 ./transcribe.sh …` (default 10s).
150
- - Capturing too many near-identical frames? `DEDUPE_SEC=1.0 ./transcribe.sh …`
151
- - Force a model: `WHISPER_MODEL=/abs/ggml-large-v3.bin ./transcribe.sh …`
177
+ ## Flags & tuning
178
+ Flags (each also settable as an env var, e.g. `MODE=ui`):
179
+ - `--mode ui|lecture` sampling preset. `ui` = `MAX_GAP` 2.5s (app screens); `lecture`
180
+ (default) = 10s (talking-head). Explicit `--max-gap`/`--scene-threshold` always win.
181
+ - `--frames-only` skip whisper; extract frames only (UI/UX teardowns).
182
+ - `--max-gap N` — guarantee a frame at least every N seconds.
183
+ - `--scene-threshold N` — scene-change sensitivity, lower = more frames.
184
+ - `--contact-sheet` — also emit `frames/contact_sheet.jpg` (coverage at a glance).
185
+ - `--lang CODE` — force the transcript language (default: auto-detect).
186
+
187
+ Common adjustments:
188
+ - Onboarding/UI recording under-sampled? `--mode ui` (or push further: `--max-gap 1.5`).
189
+ - Too few frames on a slide-heavy talk? `--scene-threshold 0.1`.
190
+ - Long static lecture over-sampled? `--max-gap 20`.
191
+ - Capturing too many near-identical frames? `DEDUPE_SEC=1.0 ./transcribe.sh …`.
192
+ - Force a model: `WHISPER_MODEL=/abs/ggml-large-v3.bin ./transcribe.sh …`.
@@ -1,14 +1,22 @@
1
1
  #!/usr/bin/env python3
2
2
  """Build frames.json: map each extracted frame to its pts_time on the video timeline.
3
3
 
4
- Reads the per-frame pts_time dumps (scene_times.txt from ffmpeg metadata, gap_times.txt
5
- written by transcribe.sh) alongside the frames, pairs each timestamp with its
6
- zero-padded frame file in selection order, adds the first/last anchor frames, drops
7
- near-duplicate timestamps, and emits frames.json sorted by time.
4
+ transcribe.sh selects frames in ONE time-based pass (scene change OR every MAX_GAP),
5
+ dumping each kept frame's pts_time to frame_times.txt. This pairs every timestamp with
6
+ its zero-padded frame file in selection order, adds the first/last anchor frames, drops
7
+ near-duplicate timestamps, labels each frame (scene vs gap vs anchor), and emits
8
+ frames.json sorted by time. It also prints a COVERAGE check — the largest unsampled
9
+ stretch — so a human can catch under-sampling before an expensive deep-analysis pass.
10
+
11
+ Type labels are reconstructed from inter-frame spacing (no second decode): a frame that
12
+ landed sooner than MAX_GAP after the previous one was triggered by a scene change
13
+ ('scene'); one that landed at the MAX_GAP cadence is a periodic fill ('gap'). They are
14
+ hints for which frames to prioritize, not a hard contract.
8
15
 
9
16
  Stdlib only — no venv needed.
10
17
  Usage: build_frame_index.py <OUT_DIR> [DURATION_SEC]
11
18
  Env: DEDUPE_SEC=0.4 collapse non-anchor frames closer than this to the kept one.
19
+ MAX_GAP=10 the gap target the engine sampled at (drives type + coverage warn).
12
20
  """
13
21
  import json
14
22
  import os
@@ -19,6 +27,7 @@ from pathlib import Path
19
27
  # metadata=print header line, e.g.: "frame:0 pts:321024 pts_time:12.852500"
20
28
  FRAME_RE = re.compile(r"frame:(\d+)\b.*?pts_time:([\d.]+)", re.DOTALL)
21
29
  DEDUPE_SEC = float(os.environ.get("DEDUPE_SEC", "0.4"))
30
+ MAX_GAP = float(os.environ.get("MAX_GAP", "10"))
22
31
 
23
32
 
24
33
  def parse_times(meta_file: Path) -> dict[int, float]:
@@ -29,11 +38,11 @@ def parse_times(meta_file: Path) -> dict[int, float]:
29
38
  return {int(n): round(float(t), 2) for n, t in FRAME_RE.findall(text)}
30
39
 
31
40
 
32
- def collect(out_dir: Path, prefix: str, meta_name: str, kind: str) -> list[dict]:
41
+ def collect(out_dir: Path, prefix: str, meta_name: str) -> list[dict]:
33
42
  times = parse_times(out_dir / "frames" / meta_name)
34
43
  frames = sorted((out_dir / "frames").glob(f"{prefix}_*.jpg"))
35
44
  # transcribe.sh names files 1-based (%04d starts at 0001); metadata frame: is 0-based.
36
- return [{"file": f"frames/{f.name}", "t": times.get(i), "type": kind}
45
+ return [{"file": f"frames/{f.name}", "t": times.get(i), "type": "key"}
37
46
  for i, f in enumerate(frames)]
38
47
 
39
48
 
@@ -69,6 +78,40 @@ def dedupe(rows: list[dict]) -> list[dict]:
69
78
  return kept
70
79
 
71
80
 
81
+ def label_types(rows: list[dict]) -> None:
82
+ """Reconstruct scene/gap from spacing. Frames spaced >= ~MAX_GAP are periodic
83
+ fills ('gap'); closer ones were pulled in early by a scene change ('scene')."""
84
+ near = MAX_GAP * 0.9
85
+ prev_t = 0.0
86
+ for r in rows:
87
+ if r["type"] == "anchor":
88
+ if r["t"] is not None:
89
+ prev_t = r["t"]
90
+ continue
91
+ if r["t"] is None:
92
+ r["type"] = "scene"
93
+ continue
94
+ r["type"] = "gap" if (r["t"] - prev_t) >= near else "scene"
95
+ prev_t = r["t"]
96
+
97
+
98
+ def coverage(rows: list[dict], duration: float) -> tuple[float, float]:
99
+ """Largest unsampled stretch (s) and where it starts, over [0, duration]."""
100
+ ts = sorted(r["t"] for r in rows if r["t"] is not None)
101
+ if not ts:
102
+ return (duration, 0.0)
103
+ # Bound the right edge with the real duration; if ffprobe couldn't read it (0), fall
104
+ # back to the last sampled time so the tail gap (last frame -> end) still surfaces
105
+ # rather than being silently dropped.
106
+ right = round(duration, 2) if duration > 0 else ts[-1]
107
+ bounds = [0.0] + ts + [right]
108
+ worst, at = 0.0, 0.0
109
+ for a, b in zip(bounds, bounds[1:]):
110
+ if b - a > worst:
111
+ worst, at = b - a, a
112
+ return (worst, at)
113
+
114
+
72
115
  def main() -> int:
73
116
  if len(sys.argv) < 2:
74
117
  print("usage: build_frame_index.py <OUT_DIR> [DURATION_SEC]", file=sys.stderr)
@@ -76,18 +119,26 @@ def main() -> int:
76
119
  out_dir = Path(sys.argv[1])
77
120
  duration = float(sys.argv[2]) if len(sys.argv) > 2 else 0.0
78
121
 
79
- rows = collect(out_dir, "scene", "scene_times.txt", "scene")
80
- rows += collect(out_dir, "gap", "gap_times.txt", "gap")
122
+ rows = collect(out_dir, "frame", "frame_times.txt")
81
123
  rows += anchors(out_dir, duration)
82
- # Sort by time; unknown timestamps sink to the end.
83
- rows.sort(key=lambda r: (r["t"] is None, r["t"] or 0.0))
124
+ # Sort by time (unknown timestamps sink to the end); on a tie, anchors sort FIRST so
125
+ # the eq(n,0) seed frame at t=0 (and any selected frame coincident with anchor_last)
126
+ # dedupes away against the identical anchor instead of producing a duplicate entry.
127
+ rows.sort(key=lambda r: (r["t"] is None, r["t"] or 0.0, r["type"] != "anchor"))
84
128
  rows = dedupe(rows)
129
+ label_types(rows)
85
130
  for r in rows:
86
131
  r["at"] = fmt(r["t"])
87
132
 
88
133
  (out_dir / "frames.json").write_text(json.dumps(rows, ensure_ascii=False, indent=2))
134
+
89
135
  n_anchor = sum(r["type"] == "anchor" for r in rows)
136
+ worst, at = coverage(rows, duration)
90
137
  print(f" frames.json: {len(rows)} frames indexed ({n_anchor} anchors, dedupe<{DEDUPE_SEC}s)")
138
+ print(f" coverage: longest unsampled gap = {worst:.1f}s at {fmt(at)} (target MAX_GAP={MAX_GAP:g}s)")
139
+ if duration > 0 and worst > MAX_GAP * 2:
140
+ print(f" ⚠ coverage gap is >2x MAX_GAP — frames may be missing around {fmt(at)}. "
141
+ f"Re-run denser: --max-gap {MAX_GAP / 2:g} (or --mode ui for app recordings).")
91
142
  return 0
92
143
 
93
144
 
@@ -6,30 +6,92 @@
6
6
  # video. Claude reads the outputs (transcript + frames.json), decides which frames
7
7
  # matter, views them, and writes the curated <video>.transcript.md (see SKILL.md).
8
8
  #
9
- # Usage: ./transcribe.sh "/abs/path/to/video.mov" [lang]
9
+ # Usage: ./transcribe.sh "/abs/path/to/video.mov" [lang] [flags]
10
10
  # lang: whisper language code. DEFAULT 'auto' — whisper detects it, no need to pass.
11
11
  # Force only if auto mislabels a short/ambiguous clip (e.g. 'tr', 'en').
12
12
  #
13
+ # Flags (also settable as env vars):
14
+ # --frames-only skip whisper entirely; extract frames only. For pure
15
+ # UI/UX teardowns where no transcript is needed. (FRAMES_ONLY=1)
16
+ # --mode ui|lecture sampling preset. 'ui' = MAX_GAP 2.5s (app screens linger
17
+ # 3-5s and change by text only, below scene-detect). 'lecture'
18
+ # (default) = MAX_GAP 10s for talking-head video. (MODE=ui)
19
+ # --max-gap N guarantee a frame at least every N seconds. Overrides the
20
+ # mode preset. (MAX_GAP=N)
21
+ # --scene-threshold N ffmpeg scene-change sensitivity, lower = more frames. (SCENE_THRESHOLD=N)
22
+ # --contact-sheet also emit frames/contact_sheet.jpg — a tiled montage of every
23
+ # frame, so coverage is verifiable in one glance. (CONTACT_SHEET=1)
24
+ # --lang CODE same as the positional lang arg.
25
+ #
13
26
  # Env overrides:
14
27
  # WHISPER_MODEL=/abs/path/ggml-*.bin force a specific model
15
28
  # OUT_DIR=/abs/path where artifacts go (default: <video_dir>/<slug>.media)
16
- # SCENE_THRESHOLD=0.15 ffmpeg scene-change sensitivity (lower = more frames)
17
- # MAX_GAP=10 guarantee a frame at least every N seconds (fills static sections)
29
+ # NOTE: avoid ':' in OUT_DIR it is special inside the
30
+ # ffmpeg filter graph and would truncate the metadata path.
31
+ # DEDUPE_SEC=0.4 collapse non-anchor frames closer than this
18
32
  #
19
33
  # Outputs (in OUT_DIR):
20
- # transcript.srt timestamped segments (human-friendly)
21
- # transcript.json/.txt/.vtt
22
- # frames/scene_*.jpg frames at scene-change > SCENE_THRESHOLD
23
- # frames/gap_*.jpg fill frames where scene-detect left a stretch > MAX_GAP unsampled
34
+ # transcript.srt timestamped segments (human-friendly) [skipped with --frames-only]
35
+ # transcript.json/.txt/.vtt [skipped with --frames-only]
36
+ # frames/frame_*.jpg selected frames (scene change OR every MAX_GAP, whichever first)
24
37
  # frames/anchor_first.jpg / anchor_last.jpg always-captured first + last frame
38
+ # frames/contact_sheet.jpg tiled montage of all frames [--contact-sheet only]
25
39
  # frames.json [{file, t, at, type}] — each frame's pts_time, sorted (Claude reads this)
26
40
  # source-meta.txt ffprobe dump (audio.wav is created then deleted after transcription)
27
41
  set -euo pipefail
28
42
 
29
- VIDEO="${1:?Usage: transcribe.sh <video> [lang]}"
30
- LANG_CODE="${2:-auto}" # whisper auto-detects the spoken language unless overridden
43
+ # --- arg + flag parsing ---------------------------------------------------------
44
+ # Positional: first non-flag = VIDEO, second non-flag = LANG_CODE (back-compatible).
45
+ VIDEO=""
46
+ LANG_CODE=""
47
+ FRAMES_ONLY="${FRAMES_ONLY:-0}"
48
+ MODE="${MODE:-lecture}"
49
+ CONTACT_SHEET="${CONTACT_SHEET:-0}"
50
+ # MAX_GAP / SCENE_THRESHOLD: track whether explicitly set so a CLI/env value beats the
51
+ # mode preset. Empty here means "fall back to the mode preset" computed below.
52
+ MAX_GAP="${MAX_GAP:-}"
53
+ SCENE_THRESHOLD="${SCENE_THRESHOLD:-}"
54
+
55
+ while [ $# -gt 0 ]; do
56
+ case "$1" in
57
+ --frames-only) FRAMES_ONLY=1 ;;
58
+ --contact-sheet) CONTACT_SHEET=1 ;;
59
+ --mode) MODE="${2:?--mode needs a value}"; shift ;;
60
+ --mode=*) MODE="${1#*=}" ;;
61
+ --max-gap) MAX_GAP="${2:?--max-gap needs a value}"; shift ;;
62
+ --max-gap=*) MAX_GAP="${1#*=}" ;;
63
+ --scene-threshold) SCENE_THRESHOLD="${2:?--scene-threshold needs a value}"; shift ;;
64
+ --scene-threshold=*) SCENE_THRESHOLD="${1#*=}" ;;
65
+ --lang) LANG_CODE="${2:?--lang needs a value}"; shift ;;
66
+ --lang=*) LANG_CODE="${1#*=}" ;;
67
+ --*) echo "unknown flag: $1" >&2; exit 2 ;;
68
+ *) if [ -z "$VIDEO" ]; then VIDEO="$1"
69
+ elif [ -z "$LANG_CODE" ]; then LANG_CODE="$1"
70
+ else echo "unexpected arg: $1" >&2; exit 2; fi ;;
71
+ esac
72
+ shift
73
+ done
74
+
75
+ [ -n "$VIDEO" ] || { echo "Usage: transcribe.sh <video> [lang] [--frames-only] [--mode ui] [--max-gap N]" >&2; exit 1; }
76
+ LANG_CODE="${LANG_CODE:-auto}" # whisper auto-detects the spoken language unless overridden
77
+
78
+ # Mode presets fill in only what the caller left unset (explicit CLI/env always wins).
79
+ case "$MODE" in
80
+ ui) MODE_MAX_GAP=2.5 ;;
81
+ lecture) MODE_MAX_GAP=10 ;;
82
+ *) echo "unknown --mode '$MODE' (use ui|lecture)" >&2; exit 2 ;;
83
+ esac
84
+ MAX_GAP="${MAX_GAP:-$MODE_MAX_GAP}"
31
85
  SCENE_THRESHOLD="${SCENE_THRESHOLD:-0.15}"
32
- MAX_GAP="${MAX_GAP:-10}"
86
+
87
+ # Guard the two numeric knobs before they reach the ffmpeg filter / Python: a non-number
88
+ # would crash build_frame_index (float()), and MAX_GAP=0 makes gte(t-prev,0) select EVERY
89
+ # frame (disk-fill footgun). Fail loudly instead.
90
+ is_num='^[0-9]+([.][0-9]+)?$'
91
+ [[ "$MAX_GAP" =~ $is_num ]] || { echo "--max-gap must be a number (got '$MAX_GAP')" >&2; exit 2; }
92
+ awk "BEGIN{exit !($MAX_GAP > 0)}" || { echo "--max-gap must be > 0 (got '$MAX_GAP')" >&2; exit 2; }
93
+ [[ "$SCENE_THRESHOLD" =~ $is_num ]] || { echo "--scene-threshold must be a number (got '$SCENE_THRESHOLD')" >&2; exit 2; }
94
+
33
95
  SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
34
96
 
35
97
  # --- remote link? download with yt-dlp (local files skip this) ------------------
@@ -43,7 +105,10 @@ if [[ "$VIDEO" =~ ^https?:// ]]; then
43
105
  fi
44
106
  [ -f "$VIDEO" ] || { echo "video not found: $VIDEO" >&2; exit 1; }
45
107
 
108
+ command -v ffmpeg >/dev/null || { echo "ffmpeg not found (brew install ffmpeg)" >&2; exit 1; }
109
+
46
110
  # --- model: prefer turbo, then large-v3, then medium (override with WHISPER_MODEL)
111
+ # Only needed for transcription — skip the whole resolution when --frames-only.
47
112
  pick_model() {
48
113
  if [ -n "${WHISPER_MODEL:-}" ]; then printf '%s' "$WHISPER_MODEL"; return; fi
49
114
  local p
@@ -57,18 +122,19 @@ pick_model() {
57
122
  done
58
123
  printf ''
59
124
  }
60
- MODEL="$(pick_model)"
61
- [ -n "$MODEL" ] && [ -f "$MODEL" ] || {
62
- echo "no whisper model found. Install one, e.g.:" >&2
63
- echo " curl -L -o ~/.cache/whisper.cpp/models/ggml-large-v3-turbo.bin \\" >&2
64
- echo " https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin" >&2
65
- echo " or set WHISPER_MODEL=/abs/path/ggml-*.bin" >&2
66
- exit 1
67
- }
68
-
69
- WHISPER_BIN="$(command -v whisper-cli || command -v whisper-cpp || command -v main || true)"
70
- [ -n "$WHISPER_BIN" ] || { echo "whisper.cpp binary not found (brew install whisper-cpp)" >&2; exit 1; }
71
- command -v ffmpeg >/dev/null || { echo "ffmpeg not found (brew install ffmpeg)" >&2; exit 1; }
125
+ if [ "$FRAMES_ONLY" != "1" ]; then
126
+ MODEL="$(pick_model)"
127
+ [ -n "$MODEL" ] && [ -f "$MODEL" ] || {
128
+ echo "no whisper model found. Install one, e.g.:" >&2
129
+ echo " curl -L -o ~/.cache/whisper.cpp/models/ggml-large-v3-turbo.bin \\" >&2
130
+ echo " https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin" >&2
131
+ echo " or set WHISPER_MODEL=/abs/path/ggml-*.bin" >&2
132
+ echo " (or pass --frames-only to skip transcription entirely)" >&2
133
+ exit 1
134
+ }
135
+ WHISPER_BIN="$(command -v whisper-cli || command -v whisper-cpp || command -v main || true)"
136
+ [ -n "$WHISPER_BIN" ] || { echo "whisper.cpp binary not found (brew install whisper-cpp). Or pass --frames-only." >&2; exit 1; }
137
+ fi
72
138
 
73
139
  # --- where artifacts go: next to the video by default ---------------------------
74
140
  SRC_DIR="$(cd "$(dirname "$VIDEO")" && pwd)"
@@ -79,28 +145,51 @@ mkdir -p "$OUT/frames"
79
145
  # even-dim scaling keeps the mjpeg encoder happy on odd-width sources
80
146
  SCALE="scale=trunc(iw/2)*2:trunc(ih/2)*2"
81
147
 
82
- echo "==> [$SLUG] probing"
148
+ echo "==> [$SLUG] probing (mode=$MODE, max_gap=${MAX_GAP}s, scene>$SCENE_THRESHOLD, frames_only=$FRAMES_ONLY)"
83
149
  ffprobe -v error -show_entries format=duration,size:stream=codec_type,codec_name,width,height \
84
150
  -of default=noprint_wrappers=1 "$VIDEO" | tee "$OUT/source-meta.txt"
85
151
  DURATION="$(ffprobe -v error -show_entries format=duration -of csv=p=0 "$VIDEO" 2>/dev/null | head -1)"
86
152
  DURATION="${DURATION:-0}"
87
153
 
88
- echo "==> [$SLUG] extracting 16kHz mono audio"
89
- ffmpeg -y -loglevel error -i "$VIDEO" -ar 16000 -ac 1 -c:a pcm_s16le "$OUT/audio.wav"
154
+ if [ "$FRAMES_ONLY" = "1" ]; then
155
+ echo "==> [$SLUG] --frames-only: skipping audio + transcription"
156
+ else
157
+ echo "==> [$SLUG] extracting 16kHz mono audio"
158
+ ffmpeg -y -loglevel error -i "$VIDEO" -ar 16000 -ac 1 -c:a pcm_s16le "$OUT/audio.wav"
90
159
 
91
- echo "==> [$SLUG] transcribing with $(basename "$MODEL") (lang=$LANG_CODE)"
92
- "$WHISPER_BIN" -m "$MODEL" -f "$OUT/audio.wav" -l "$LANG_CODE" \
93
- --output-txt --output-srt --output-vtt --output-json -of "$OUT/transcript" -pp
94
- # audio.wav is a pure whisper intermediate (~1.9MB/min) — drop it once the transcript exists.
95
- rm -f "$OUT/audio.wav"
160
+ echo "==> [$SLUG] transcribing with $(basename "$MODEL") (lang=$LANG_CODE)"
161
+ "$WHISPER_BIN" -m "$MODEL" -f "$OUT/audio.wav" -l "$LANG_CODE" \
162
+ --output-txt --output-srt --output-vtt --output-json -of "$OUT/transcript" -pp
163
+ # audio.wav is a pure whisper intermediate (~1.9MB/min) — drop it once the transcript exists.
164
+ rm -f "$OUT/audio.wav"
165
+ fi
96
166
 
97
- echo "==> [$SLUG] extracting scene-change frames (scene > $SCENE_THRESHOLD)"
167
+ echo "==> [$SLUG] extracting frames (scene change OR every ${MAX_GAP}s, whichever fires first)"
168
+ # ONE decode pass, time-based and frame-rate-independent:
169
+ # eq(n,0) always seed the first frame (also primes prev_selected_t)
170
+ # gt(scene,$SCENE_THRESHOLD) a scene change (slide/UI transition) fired
171
+ # gte(t-prev_selected_t,MAX_GAP) MAX_GAP elapsed since the last KEPT frame — guarantees
172
+ # coverage of static stretches scene-detect misses (app
173
+ # screens that change by text only). prev_selected_t is
174
+ # wall-clock seconds, so this is robust on variable-frame-rate
175
+ # screen recordings where frame-number sampling (mod(n,N)) breaks.
98
176
  # pix_fmt yuvj420p avoids mjpeg "non full-range YUV" failures on screen recordings;
99
- # even-dim scaling keeps the encoder happy on odd-width sources.
100
177
  # metadata=print dumps each kept frame's pts_time so frames map back to the timeline.
101
178
  ffmpeg -y -loglevel error -i "$VIDEO" \
102
- -vf "select='gt(scene,$SCENE_THRESHOLD)',metadata=print:file=$OUT/frames/scene_times.txt,scale=trunc(iw/2)*2:trunc(ih/2)*2" \
103
- -fps_mode vfr -pix_fmt yuvj420p -q:v 3 "$OUT/frames/scene_%04d.jpg" || true
179
+ -vf "select='eq(n,0)+gt(scene,$SCENE_THRESHOLD)+gte(t-prev_selected_t,$MAX_GAP)',metadata=print:file=$OUT/frames/frame_times.txt,scale=trunc(iw/2)*2:trunc(ih/2)*2" \
180
+ -fps_mode vfr -pix_fmt yuvj420p -q:v 3 "$OUT/frames/frame_%04d.jpg" || true
181
+
182
+ # The || true above keeps a partial result usable, but zero frames means the decode
183
+ # produced nothing (no decodable video stream, unsupported codec, write failure). That
184
+ # must fail loudly — silently leaving an empty frames.json is the exact under-sampling
185
+ # trap issue #15 is about. Count safely (|| true so a no-match never trips set -e).
186
+ NSEL="$(find "$OUT/frames" -name 'frame_*.jpg' 2>/dev/null | wc -l | tr -d ' ')"
187
+ if [ "$NSEL" -eq 0 ]; then
188
+ echo "ERROR: ffmpeg extracted 0 frames from '$VIDEO'." >&2
189
+ echo " The file may have no decodable video stream (audio-only?), an unsupported codec," >&2
190
+ echo " or the output dir isn't writable. No frames.json written." >&2
191
+ exit 1
192
+ fi
104
193
 
105
194
  echo "==> [$SLUG] capturing first + last anchor frames"
106
195
  # Short, cut-heavy creatives carry the most information in the opening hook and the
@@ -110,27 +199,31 @@ ffmpeg -y -loglevel error -i "$VIDEO" -vf "$SCALE" -frames:v 1 -q:v 3 \
110
199
  ffmpeg -y -loglevel error -sseof -0.5 -i "$VIDEO" -vf "$SCALE" -update 1 -frames:v 1 -q:v 3 \
111
200
  "$OUT/frames/anchor_last.jpg" || true
112
201
 
113
- echo "==> [$SLUG] filling gaps > ${MAX_GAP}s (static sections scene-detect misses)"
114
- # scene-detect fires on motion, not meaning; a static-but-information-rich stretch
115
- # (app demo, slide, talking head) gets no frame. gap_fill.py returns timestamps that
116
- # keep every unsampled stretch <= MAX_GAP; we grab each and log its known pts_time.
117
- : > "$OUT/frames/gap_times.txt"
118
- gi=0
119
- while IFS= read -r t; do
120
- [ -z "$t" ] && continue
121
- printf -v fn "gap_%04d.jpg" "$((gi + 1))"
122
- if ffmpeg -y -loglevel error -ss "$t" -i "$VIDEO" -vf "$SCALE" -frames:v 1 -q:v 3 "$OUT/frames/$fn"; then
123
- echo "frame:$gi pts_time:$t" >> "$OUT/frames/gap_times.txt"
124
- gi=$((gi + 1))
125
- fi
126
- done < <(python3 "$SCRIPT_DIR/gap_fill.py" "$OUT" "$DURATION" "$MAX_GAP")
127
- echo " gap frames added: $gi"
202
+ echo "==> [$SLUG] building frames.json (timestamp index, dedup near-dups, coverage check)"
203
+ MAX_GAP="$MAX_GAP" python3 "$SCRIPT_DIR/build_frame_index.py" "$OUT" "$DURATION"
128
204
 
129
- echo "==> [$SLUG] building frames.json (timestamp index, dedup near-dups)"
130
- python3 "$SCRIPT_DIR/build_frame_index.py" "$OUT" "$DURATION"
205
+ # --- optional contact sheet: one glance to verify coverage before deep analysis --
206
+ if [ "$CONTACT_SHEET" = "1" ]; then
207
+ echo "==> [$SLUG] building contact sheet (coverage montage)"
208
+ NF="$(ls -1 "$OUT/frames"/frame_*.jpg 2>/dev/null | wc -l | tr -d ' ')"
209
+ if [ "$NF" -gt 0 ]; then
210
+ COLS="$(awk -v n="$NF" 'BEGIN{c=int(sqrt(n)); if(c*c<n)c++; print c}')"
211
+ ROWS="$(awk -v n="$NF" -v c="$COLS" 'BEGIN{r=int(n/c); if(r*c<n)r++; print r}')"
212
+ ffmpeg -y -loglevel error -pattern_type glob -i "$OUT/frames/frame_*.jpg" \
213
+ -vf "scale=240:-1,tile=${COLS}x${ROWS}:padding=4:margin=4" -frames:v 1 -q:v 4 \
214
+ "$OUT/frames/contact_sheet.jpg" \
215
+ && echo " contact sheet: $OUT/frames/contact_sheet.jpg (${COLS}x${ROWS})" \
216
+ || echo " contact sheet skipped (montage failed)"
217
+ fi
218
+ fi
131
219
 
132
- NFRAMES="$(ls -1 "$OUT/frames"/*.jpg 2>/dev/null | wc -l | tr -d ' ')"
220
+ # find -not -name keeps the count set -e-safe (grep -v exits 1 on no-match, tripping pipefail).
221
+ NFRAMES="$(find "$OUT/frames" -name '*.jpg' ! -name 'contact_sheet.jpg' 2>/dev/null | wc -l | tr -d ' ')"
133
222
  echo "==> [$SLUG] DONE"
134
- echo " transcript: $OUT/transcript.srt"
223
+ [ "$FRAMES_ONLY" = "1" ] || echo " transcript: $OUT/transcript.srt"
135
224
  echo " frames: $NFRAMES (index: $OUT/frames.json)"
136
- echo " next: Claude reads transcript + frames.json, views key frames, writes the curated transcript.md"
225
+ if [ "$FRAMES_ONLY" = "1" ]; then
226
+ echo " next: Claude reads frames.json, views key frames, writes the curated screen list / transcript.md"
227
+ else
228
+ echo " next: Claude reads transcript + frames.json, views key frames, writes the curated transcript.md"
229
+ fi