dreamcontext 0.5.4 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -21
- package/NOTICE +7 -0
- package/README.md +79 -13
- package/agents/sleep-product.md +48 -10
- package/agents/sleep-state.md +43 -22
- package/agents/sleep-tasks.md +62 -9
- package/dist/agents/sleep-product.md +48 -10
- package/dist/agents/sleep-state.md +43 -22
- package/dist/agents/sleep-tasks.md +62 -9
- package/dist/dashboard/assets/{BrainCanvas3D-1HMiA14F.js → BrainCanvas3D-lFgJbbhZ.js} +1 -1
- package/dist/dashboard/assets/{_baseUniq-C7pfRmXz.js → _baseUniq-BpANgc_i.js} +1 -1
- package/dist/dashboard/assets/{arc-B5x_GF9Q.js → arc-CX32Jm7E.js} +1 -1
- package/dist/dashboard/assets/{architectureDiagram-Q4EWVU46-CWmjaqno.js → architectureDiagram-Q4EWVU46-ARASlGxO.js} +1 -1
- package/dist/dashboard/assets/{blockDiagram-DXYQGD6D-B7SyUIdV.js → blockDiagram-DXYQGD6D-BBvsYm9E.js} +1 -1
- package/dist/dashboard/assets/{c4Diagram-AHTNJAMY-OeJBWXHU.js → c4Diagram-AHTNJAMY-ChSUfXR9.js} +1 -1
- package/dist/dashboard/assets/channel-w_Bp182N.js +1 -0
- package/dist/dashboard/assets/{chunk-4BX2VUAB-2Oop1B0M.js → chunk-4BX2VUAB-COHoVEpt.js} +1 -1
- package/dist/dashboard/assets/{chunk-4TB4RGXK-Bh7qSu_X.js → chunk-4TB4RGXK-DKJFLaTC.js} +1 -1
- package/dist/dashboard/assets/{chunk-55IACEB6-gNeHRO_i.js → chunk-55IACEB6-CKZPWZRu.js} +1 -1
- package/dist/dashboard/assets/{chunk-EDXVE4YY-BRniAH53.js → chunk-EDXVE4YY-k1y4mGUJ.js} +1 -1
- package/dist/dashboard/assets/{chunk-FMBD7UC4-B5r8U2gJ.js → chunk-FMBD7UC4-ixXe5R10.js} +1 -1
- package/dist/dashboard/assets/{chunk-OYMX7WX6-Dd1c17fm.js → chunk-OYMX7WX6-NDFap7xg.js} +1 -1
- package/dist/dashboard/assets/{chunk-QZHKN3VN-BIw-OdTn.js → chunk-QZHKN3VN-vY0UpHjB.js} +1 -1
- package/dist/dashboard/assets/{chunk-YZCP3GAM-Dl9MlwCW.js → chunk-YZCP3GAM-yztvsKR-.js} +1 -1
- package/dist/dashboard/assets/classDiagram-6PBFFD2Q-CcQxcqy9.js +1 -0
- package/dist/dashboard/assets/classDiagram-v2-HSJHXN6E-CcQxcqy9.js +1 -0
- package/dist/dashboard/assets/clone-_znoR_ci.js +1 -0
- package/dist/dashboard/assets/{cose-bilkent-S5V4N54A-DlrROH5K.js → cose-bilkent-S5V4N54A-DKXM5Fh5.js} +1 -1
- package/dist/dashboard/assets/{dagre-KV5264BT-B0GpPXKL.js → dagre-KV5264BT-DN1Nlsmy.js} +1 -1
- package/dist/dashboard/assets/{diagram-5BDNPKRD-BBYcDW2Z.js → diagram-5BDNPKRD-DnoJFRqR.js} +1 -1
- package/dist/dashboard/assets/{diagram-G4DWMVQ6-BxHxugdp.js → diagram-G4DWMVQ6-CF1_jTNI.js} +1 -1
- package/dist/dashboard/assets/{diagram-MMDJMWI5-C9Jc19Wo.js → diagram-MMDJMWI5-D4-bZZEt.js} +1 -1
- package/dist/dashboard/assets/{diagram-TYMM5635-Vmo5GAAU.js → diagram-TYMM5635-al8RqVgf.js} +1 -1
- package/dist/dashboard/assets/{erDiagram-SMLLAGMA-pskr9x-V.js → erDiagram-SMLLAGMA-MEcC0rwO.js} +1 -1
- package/dist/dashboard/assets/{flowDiagram-DWJPFMVM-B1D7Ky0W.js → flowDiagram-DWJPFMVM-DCLyNpW8.js} +1 -1
- package/dist/dashboard/assets/{ganttDiagram-T4ZO3ILL-BVJZZV1M.js → ganttDiagram-T4ZO3ILL-DfEvbbJK.js} +1 -1
- package/dist/dashboard/assets/{gitGraphDiagram-UUTBAWPF-BiOGN6We.js → gitGraphDiagram-UUTBAWPF-C_YozdVL.js} +1 -1
- package/dist/dashboard/assets/{graph-D58deEXr.js → graph-DpIXS1G1.js} +1 -1
- package/dist/dashboard/assets/index-Bo5CUa_M.js +480 -0
- package/dist/dashboard/assets/index-DrqurW1c.css +1 -0
- package/dist/dashboard/assets/{infoDiagram-42DDH7IO-DgJYg1X1.js → infoDiagram-42DDH7IO-AdT6kjzj.js} +1 -1
- package/dist/dashboard/assets/{ishikawaDiagram-UXIWVN3A-arJgr-8P.js → ishikawaDiagram-UXIWVN3A-B0_9IZVO.js} +1 -1
- package/dist/dashboard/assets/{journeyDiagram-VCZTEJTY-B3DCvb8p.js → journeyDiagram-VCZTEJTY-BveNBswQ.js} +1 -1
- package/dist/dashboard/assets/{kanban-definition-6JOO6SKY-B--dN6Ma.js → kanban-definition-6JOO6SKY-CFI8j4jR.js} +1 -1
- package/dist/dashboard/assets/{layout-z2qEbZWC.js → layout-DI7XjZy1.js} +1 -1
- package/dist/dashboard/assets/{linear-BQu1dwrM.js → linear-CSjp56iw.js} +1 -1
- package/dist/dashboard/assets/{min-DkF5b2sh.js → min-pAGUJmEC.js} +1 -1
- package/dist/dashboard/assets/{mindmap-definition-QFDTVHPH-Bw04koyq.js → mindmap-definition-QFDTVHPH-C2mBnknr.js} +1 -1
- package/dist/dashboard/assets/{pieDiagram-DEJITSTG-vNqz64o8.js → pieDiagram-DEJITSTG-i_phWDqD.js} +1 -1
- package/dist/dashboard/assets/{quadrantDiagram-34T5L4WZ-TuVP5Gi6.js → quadrantDiagram-34T5L4WZ-Dv7TGJjw.js} +1 -1
- package/dist/dashboard/assets/{requirementDiagram-MS252O5E-D3mS8_ke.js → requirementDiagram-MS252O5E-CB2Jl-O5.js} +1 -1
- package/dist/dashboard/assets/{sankeyDiagram-XADWPNL6-BmVRers6.js → sankeyDiagram-XADWPNL6-DxCoN-EI.js} +1 -1
- package/dist/dashboard/assets/{sequenceDiagram-FGHM5R23-CTYCgk5P.js → sequenceDiagram-FGHM5R23-BeZyaehJ.js} +1 -1
- package/dist/dashboard/assets/{stateDiagram-FHFEXIEX-DNbP2aCg.js → stateDiagram-FHFEXIEX-D3AjQzD1.js} +1 -1
- package/dist/dashboard/assets/stateDiagram-v2-QKLJ7IA2-Ci9v--xK.js +1 -0
- package/dist/dashboard/assets/{timeline-definition-GMOUNBTQ-DKHJd4z8.js → timeline-definition-GMOUNBTQ-3l_8RFUg.js} +1 -1
- package/dist/dashboard/assets/{vennDiagram-DHZGUBPP-ChotNMNq.js → vennDiagram-DHZGUBPP-2QiY25JD.js} +1 -1
- package/dist/dashboard/assets/{wardley-RL74JXVD-Dmo2_ksP.js → wardley-RL74JXVD-DNLmFofz.js} +1 -1
- package/dist/dashboard/assets/{wardleyDiagram-NUSXRM2D-Z6Wbiv2H.js → wardleyDiagram-NUSXRM2D-BI0tupiS.js} +1 -1
- package/dist/dashboard/assets/{xychartDiagram-5P7HB3ND-BJa0AHtN.js → xychartDiagram-5P7HB3ND-CghXPE7_.js} +1 -1
- package/dist/dashboard/index.html +2 -2
- package/dist/dashboard/media/README.md +29 -0
- package/dist/dashboard/media/brain-hero.png +0 -0
- package/dist/dashboard/media/brain.mp4 +0 -0
- package/dist/dashboard/media/brain.webm +0 -0
- package/dist/dashboard/media/shot-disabled.png +0 -0
- package/dist/dashboard/media/shot-enabled.png +0 -0
- package/dist/index.js +3352 -2035
- package/dist/skill-packs/catalog.json +44 -0
- package/dist/skill-packs/excalidraw/SKILL.md +202 -0
- package/dist/skill-packs/excalidraw/examples/hello.spec.json +14 -0
- package/dist/skill-packs/excalidraw/examples/sample.png +0 -0
- package/dist/skill-packs/excalidraw/examples/style_board.js +31 -0
- package/dist/skill-packs/excalidraw/package.json +5 -0
- package/dist/skill-packs/excalidraw/reference/format.md +93 -0
- package/dist/skill-packs/excalidraw/scripts/build_excalidraw.js +357 -0
- package/dist/skill-packs/excalidraw/scripts/lib/fractional-indexing.LICENSE +121 -0
- package/dist/skill-packs/excalidraw/scripts/lib/fractional-indexing.js +311 -0
- package/dist/skill-packs/excalidraw/scripts/lib/imagesize.js +65 -0
- package/dist/skill-packs/excalidraw/scripts/lib/style.js +130 -0
- package/dist/skill-packs/video-watching/SKILL.md +192 -0
- package/dist/skill-packs/video-watching/scripts/build_frame_index.py +146 -0
- package/dist/skill-packs/video-watching/scripts/transcribe.sh +229 -0
- package/dist/templates/init/data-structures/default.md +26 -24
- package/package.json +4 -2
- package/skill/SKILL.md +31 -12
- package/skill-packs/catalog.json +44 -0
- package/skill-packs/excalidraw/SKILL.md +202 -0
- package/skill-packs/excalidraw/examples/hello.spec.json +14 -0
- package/skill-packs/excalidraw/examples/sample.png +0 -0
- package/skill-packs/excalidraw/examples/style_board.js +31 -0
- package/skill-packs/excalidraw/package.json +5 -0
- package/skill-packs/excalidraw/reference/format.md +93 -0
- package/skill-packs/excalidraw/scripts/build_excalidraw.js +357 -0
- package/skill-packs/excalidraw/scripts/lib/fractional-indexing.LICENSE +121 -0
- package/skill-packs/excalidraw/scripts/lib/fractional-indexing.js +311 -0
- package/skill-packs/excalidraw/scripts/lib/imagesize.js +65 -0
- package/skill-packs/excalidraw/scripts/lib/style.js +130 -0
- package/skill-packs/video-watching/SKILL.md +192 -0
- package/skill-packs/video-watching/scripts/build_frame_index.py +146 -0
- package/skill-packs/video-watching/scripts/transcribe.sh +229 -0
- package/dist/dashboard/assets/channel-CbuTKKsM.js +0 -1
- package/dist/dashboard/assets/classDiagram-6PBFFD2Q-HXl3A_sM.js +0 -1
- package/dist/dashboard/assets/classDiagram-v2-HSJHXN6E-HXl3A_sM.js +0 -1
- package/dist/dashboard/assets/clone-NKHOtdi8.js +0 -1
- package/dist/dashboard/assets/index-DK_2-eWY.css +0 -1
- package/dist/dashboard/assets/index-Ds-JXWtr.js +0 -476
- package/dist/dashboard/assets/stateDiagram-v2-QKLJ7IA2-CfN6kjo8.js +0 -1
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: video-watching
|
|
3
|
+
description: >-
|
|
4
|
+
Watch a video the way a human would — turn it into a time-mapped transcript
|
|
5
|
+
with the on-screen visuals (slides, UI, diagrams, text) described inline, then
|
|
6
|
+
reason about it. Triggers: "watch this video", "şu videoyu izle", "videoyu
|
|
7
|
+
analiz et", "transkript çıkar", "video transcript", "bu videoda ne var", or any
|
|
8
|
+
time the user hands over a video file/link. Produces a curated
|
|
9
|
+
<video>.transcript.md NEXT TO the source video, then continues per use-case
|
|
10
|
+
(marketing teardown, FTE analysis, knowledge capture, …).
|
|
11
|
+
alwaysApply: false
|
|
12
|
+
ruleType: "Tool Workflow"
|
|
13
|
+
version: "1.0"
|
|
14
|
+
---
|
|
15
|
+
|
|
16
|
+
# Video Watching
|
|
17
|
+
|
|
18
|
+
Turn a video into something an AI can fully reason about: a **time-mapped
|
|
19
|
+
transcript** with **what's on screen described inline at the same timestamps**.
|
|
20
|
+
The output is one self-contained file dropped **next to the source video**, so any
|
|
21
|
+
model (our cloud, this session, anything) can read the whole video without the
|
|
22
|
+
binary.
|
|
23
|
+
|
|
24
|
+
The clever part: **transcript first, then YOU decide which frames to look at.**
|
|
25
|
+
You don't blindly OCR every frame. You read the transcript, see where the speaker
|
|
26
|
+
references something visual ("as you can see here", "this screen", "the chart"),
|
|
27
|
+
look up that moment in `frames.json`, and open only those frames.
|
|
28
|
+
|
|
29
|
+
## The loop (do these in order)
|
|
30
|
+
|
|
31
|
+
### 0. Get the video
|
|
32
|
+
Ask for the path or link if not given. Local files work today. Remote links
|
|
33
|
+
(YouTube etc.) need `yt-dlp` (`brew install yt-dlp`) — the engine auto-downloads
|
|
34
|
+
if it's installed.
|
|
35
|
+
|
|
36
|
+
### 1. Run the engine (mechanical — one command)
|
|
37
|
+
```bash
|
|
38
|
+
.claude/skills/video-watching/scripts/transcribe.sh "/abs/path/to/video.mov"
|
|
39
|
+
# language auto-detects — no need to pass it. Override only if auto mislabels a
|
|
40
|
+
# short/ambiguous clip: ./transcribe.sh "/abs/path/clip.mp4" tr
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
**Pick the mode for the kind of video — this matters most for app/UI recordings:**
|
|
44
|
+
- **Talking-head / lecture / ad creative** → the default is right. Scene-detect + a
|
|
45
|
+
10s gap-fill catches the visuals.
|
|
46
|
+
- **App screen-recording / onboarding funnel / UI walkthrough** → add `--mode ui`.
|
|
47
|
+
App screens linger 3–5s and change by **text only** (a questionnaire step, a
|
|
48
|
+
paywall) — they don't move enough to trip scene-detect, so the 10s default
|
|
49
|
+
silently drops most of them. `--mode ui` samples every ~2.5s so each screen lands.
|
|
50
|
+
If you only need the screens (no narration), add `--frames-only` to skip whisper:
|
|
51
|
+
```bash
|
|
52
|
+
./scripts/transcribe.sh "/abs/path/onboarding.mp4" --mode ui --frames-only --contact-sheet
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
This writes everything into `<video_dir>/<slug>.media/`:
|
|
56
|
+
- `transcript.srt` / `.json` / `.txt` — timestamped transcript (large-v3-turbo) — *skipped with `--frames-only`*
|
|
57
|
+
- `frames/anchor_first.jpg` / `anchor_last.jpg` — first + last frame, always captured
|
|
58
|
+
(short cut-heavy creatives carry the hook and CTA here; scene-detect misses both)
|
|
59
|
+
- `frames/frame_*.jpg` — frames selected in **one time-based pass**: a scene change
|
|
60
|
+
fired (slide/UI transition) **or** `MAX_GAP` seconds elapsed since the last frame,
|
|
61
|
+
whichever comes first. The gap rule is wall-clock based (`prev_selected_t`), so it
|
|
62
|
+
works on variable-frame-rate screen recordings where frame-number sampling breaks,
|
|
63
|
+
and it guarantees every static stretch (app demos, slides, CTAs) gets a frame.
|
|
64
|
+
- `frames/contact_sheet.jpg` — tiled montage of all frames, *only with `--contact-sheet`*
|
|
65
|
+
- `frames.json` — **the index you read**: `[{file, t, at, type}]`, sorted by time,
|
|
66
|
+
near-duplicate timestamps collapsed (anchors always kept). `type` is `scene` (a
|
|
67
|
+
picture change fired), `gap` (a periodic fill at the `MAX_GAP` cadence), or `anchor`.
|
|
68
|
+
|
|
69
|
+
The engine prints a **coverage check** at the end: `longest unsampled gap = Xs`. If it
|
|
70
|
+
warns the gap is >2× `MAX_GAP`, frames are likely missing — re-run denser (`--max-gap`
|
|
71
|
+
lower, or `--mode ui`) **before** any expensive deep-analysis pass.
|
|
72
|
+
|
|
73
|
+
`audio.wav` is auto-deleted after transcription (it's a ~1.9MB/min whisper-only
|
|
74
|
+
intermediate). The whole `*.media/` dir is gitignored — it stays next to the video
|
|
75
|
+
as scratch but never enters the repo. Only `<slug>.transcript.md` is the keeper;
|
|
76
|
+
delete a `.media/` folder anytime to reclaim disk (re-running regenerates it).
|
|
77
|
+
|
|
78
|
+
### 2. Read the transcript
|
|
79
|
+
Read `transcript.srt`. It carries the spoken content with `[mm:ss]` timing. Form a
|
|
80
|
+
first-pass understanding of structure and topic.
|
|
81
|
+
|
|
82
|
+
### 3. Decide which frames you NEED — then view them
|
|
83
|
+
Read `frames.json`. For each moment where understanding depends on the visual —
|
|
84
|
+
the speaker points at something, a slide/UI/diagram/number is on screen, or the
|
|
85
|
+
transcript is ambiguous without the picture — pick the frame whose `at` is closest
|
|
86
|
+
to that moment and **open it with the Read tool** (it renders images). Don't open
|
|
87
|
+
all of them; open the ones that carry information. Talking-head stretches with
|
|
88
|
+
nothing on screen need no frame.
|
|
89
|
+
|
|
90
|
+
> Heuristic: the two `anchor` frames (first/last) almost always matter — the hook
|
|
91
|
+
> and the CTA. Every `scene` frame is a candidate (the picture changed for a
|
|
92
|
+
> reason). `gap` frames cover static stretches scene-detect skipped — often the
|
|
93
|
+
> most informative part (an app demo or onboarding screen that doesn't "move"), so
|
|
94
|
+
> check them. In `--mode ui` runs most frames are `gap` — that's expected and you
|
|
95
|
+
> generally want to look at all of them, one per screen.
|
|
96
|
+
|
|
97
|
+
**Need a frame the index doesn't have? Grab it on demand.** Scene-detect fires on
|
|
98
|
+
motion, not on meaning — on fast-cut video the most informative moment often sits
|
|
99
|
+
between the indexed frames. When the VO points at something ("look at this
|
|
100
|
+
screen", a number, a result) and no indexed frame lands there, fetch that exact
|
|
101
|
+
second yourself:
|
|
102
|
+
```bash
|
|
103
|
+
ffmpeg -y -loglevel error -ss <seconds> -i "<video>" -frames:v 1 -q:v 3 \
|
|
104
|
+
"<slug>.media/frames/grab_<mmss>.jpg"
|
|
105
|
+
```
|
|
106
|
+
This is the whole point of transcript-first: the frame set is yours to extend, not
|
|
107
|
+
a fixed dump you're stuck with.
|
|
108
|
+
|
|
109
|
+
### 4. Write the curated transcript NEXT TO the video
|
|
110
|
+
Write `<video_dir>/<slug>.transcript.md` — the deliverable. One timeline, voice and
|
|
111
|
+
visuals interleaved, every line timestamped:
|
|
112
|
+
|
|
113
|
+
```markdown
|
|
114
|
+
# <Video Title>
|
|
115
|
+
source: <path or URL> · duration: <mm:ss> · transcribed: large-v3-turbo
|
|
116
|
+
|
|
117
|
+
## Summary
|
|
118
|
+
2–4 sentences: what this video is and what it's for.
|
|
119
|
+
|
|
120
|
+
## Timeline
|
|
121
|
+
- **[00:00]** <what's said>
|
|
122
|
+
- **[00:06]** 🖼 <what's on screen — slide title, UI state, chart, on-screen text>
|
|
123
|
+
- **[00:12]** <what's said> … 🖼 <visual if relevant>
|
|
124
|
+
...
|
|
125
|
+
|
|
126
|
+
## Visual notes
|
|
127
|
+
- **[00:06]** Slide: "Funnel → Billing → Gateway → PSP" (4 layers)
|
|
128
|
+
- ...
|
|
129
|
+
```
|
|
130
|
+
Rules: never invent content not in transcript or frames; mark anything uncertain;
|
|
131
|
+
keep the source path in the header so the file stands alone.
|
|
132
|
+
|
|
133
|
+
### 5. Hand off to the use-case
|
|
134
|
+
Once the `.transcript.md` exists you know the video cold. Tell the user it's ready
|
|
135
|
+
and ask what they want from it — the answer is domain-specific. Triage to the
|
|
136
|
+
relevant dreamcontext skill (per the skill-triage rule) and load it:
|
|
137
|
+
- **Marketing video** → teardown, hook/CTA analysis, "we could do X" (→ `growth` / `meta-marketing` skills)
|
|
138
|
+
- **First-time-experience / app demo** → reconstruct the FTE flow, friction points (→ `design` / `onboarding-design`)
|
|
139
|
+
- **Knowledge / training** → if it should persist for the project, hand the transcript to `dreamcontext knowledge` so it becomes durable context
|
|
140
|
+
Don't guess the use-case — let the user direct it.
|
|
141
|
+
|
|
142
|
+
> **UI teardown (no transcript needed).** When the goal is purely the app flow —
|
|
143
|
+
> e.g. tearing down a competitor's onboarding funnel — run `--frames-only --mode ui`
|
|
144
|
+
> and skip the `.transcript.md` entirely. The deliverable becomes a **curated screen
|
|
145
|
+
> list** (one entry per onboarding step, in order, from `frames.json`), which feeds a
|
|
146
|
+
> board (`excalidraw`) or an `onboarding-design` analysis directly. Use `--contact-sheet`
|
|
147
|
+
> to eyeball coverage first, and trust the coverage warning — under-sampling here is
|
|
148
|
+
> exactly what makes a teardown wrongly report "screen X wasn't shown".
|
|
149
|
+
|
|
150
|
+
## What this skill is NOT
|
|
151
|
+
- Not a knowledge-base writer or ingestion pipeline. This skill produces the
|
|
152
|
+
transcript artifact; persisting it (chunking, embedding, ingest into a project's
|
|
153
|
+
long-term memory) is a separate, downstream concern — keep that boundary clean.
|
|
154
|
+
- Not an auto-summarizer that skips the frames. The visual pass is the point —
|
|
155
|
+
that's what separates "watched the video" from "read the subtitles".
|
|
156
|
+
|
|
157
|
+
## Setup (once per machine)
|
|
158
|
+
```bash
|
|
159
|
+
brew install whisper-cpp ffmpeg # yt-dlp too, only for remote links
|
|
160
|
+
# model: the engine prefers large-v3-turbo, falls back to large-v3 then medium.
|
|
161
|
+
# pull turbo if missing:
|
|
162
|
+
# curl -L -o ~/.cache/whisper.cpp/models/ggml-large-v3-turbo.bin \
|
|
163
|
+
# https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin
|
|
164
|
+
```
|
|
165
|
+
`build_frame_index.py` is stdlib-only — no venv needed. Language **auto-detects**
|
|
166
|
+
(whisper); only pass a code (`tr`, `en`) to force it on a short/ambiguous clip.
|
|
167
|
+
|
|
168
|
+
## Files
|
|
169
|
+
```
|
|
170
|
+
.claude/skills/video-watching/
|
|
171
|
+
├── SKILL.md ← you are here
|
|
172
|
+
└── scripts/
|
|
173
|
+
├── transcribe.sh ← video → transcript + frames + frames.json (the engine)
|
|
174
|
+
└── build_frame_index.py ← frames/*.jpg → frames.json (pts_time index + coverage check)
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
## Flags & tuning
|
|
178
|
+
Flags (each also settable as an env var, e.g. `MODE=ui`):
|
|
179
|
+
- `--mode ui|lecture` — sampling preset. `ui` = `MAX_GAP` 2.5s (app screens); `lecture`
|
|
180
|
+
(default) = 10s (talking-head). Explicit `--max-gap`/`--scene-threshold` always win.
|
|
181
|
+
- `--frames-only` — skip whisper; extract frames only (UI/UX teardowns).
|
|
182
|
+
- `--max-gap N` — guarantee a frame at least every N seconds.
|
|
183
|
+
- `--scene-threshold N` — scene-change sensitivity, lower = more frames.
|
|
184
|
+
- `--contact-sheet` — also emit `frames/contact_sheet.jpg` (coverage at a glance).
|
|
185
|
+
- `--lang CODE` — force the transcript language (default: auto-detect).
|
|
186
|
+
|
|
187
|
+
Common adjustments:
|
|
188
|
+
- Onboarding/UI recording under-sampled? `--mode ui` (or push further: `--max-gap 1.5`).
|
|
189
|
+
- Too few frames on a slide-heavy talk? `--scene-threshold 0.1`.
|
|
190
|
+
- Long static lecture over-sampled? `--max-gap 20`.
|
|
191
|
+
- Capturing too many near-identical frames? `DEDUPE_SEC=1.0 ./transcribe.sh …`.
|
|
192
|
+
- Force a model: `WHISPER_MODEL=/abs/ggml-large-v3.bin ./transcribe.sh …`.
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build frames.json: map each extracted frame to its pts_time on the video timeline.
|
|
3
|
+
|
|
4
|
+
transcribe.sh selects frames in ONE time-based pass (scene change OR every MAX_GAP),
|
|
5
|
+
dumping each kept frame's pts_time to frame_times.txt. This pairs every timestamp with
|
|
6
|
+
its zero-padded frame file in selection order, adds the first/last anchor frames, drops
|
|
7
|
+
near-duplicate timestamps, labels each frame (scene vs gap vs anchor), and emits
|
|
8
|
+
frames.json sorted by time. It also prints a COVERAGE check — the largest unsampled
|
|
9
|
+
stretch — so a human can catch under-sampling before an expensive deep-analysis pass.
|
|
10
|
+
|
|
11
|
+
Type labels are reconstructed from inter-frame spacing (no second decode): a frame that
|
|
12
|
+
landed sooner than MAX_GAP after the previous one was triggered by a scene change
|
|
13
|
+
('scene'); one that landed at the MAX_GAP cadence is a periodic fill ('gap'). They are
|
|
14
|
+
hints for which frames to prioritize, not a hard contract.
|
|
15
|
+
|
|
16
|
+
Stdlib only — no venv needed.
|
|
17
|
+
Usage: build_frame_index.py <OUT_DIR> [DURATION_SEC]
|
|
18
|
+
Env: DEDUPE_SEC=0.4 collapse non-anchor frames closer than this to the kept one.
|
|
19
|
+
MAX_GAP=10 the gap target the engine sampled at (drives type + coverage warn).
|
|
20
|
+
"""
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import re
|
|
24
|
+
import sys
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
# metadata=print header line, e.g.: "frame:0 pts:321024 pts_time:12.852500"
|
|
28
|
+
FRAME_RE = re.compile(r"frame:(\d+)\b.*?pts_time:([\d.]+)", re.DOTALL)
|
|
29
|
+
DEDUPE_SEC = float(os.environ.get("DEDUPE_SEC", "0.4"))
|
|
30
|
+
MAX_GAP = float(os.environ.get("MAX_GAP", "10"))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def parse_times(meta_file: Path) -> dict[int, float]:
|
|
34
|
+
"""frame index (0-based, as ffmpeg numbers kept frames) -> pts_time seconds."""
|
|
35
|
+
if not meta_file.exists():
|
|
36
|
+
return {}
|
|
37
|
+
text = meta_file.read_text(errors="replace")
|
|
38
|
+
return {int(n): round(float(t), 2) for n, t in FRAME_RE.findall(text)}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def collect(out_dir: Path, prefix: str, meta_name: str) -> list[dict]:
|
|
42
|
+
times = parse_times(out_dir / "frames" / meta_name)
|
|
43
|
+
frames = sorted((out_dir / "frames").glob(f"{prefix}_*.jpg"))
|
|
44
|
+
# transcribe.sh names files 1-based (%04d starts at 0001); metadata frame: is 0-based.
|
|
45
|
+
return [{"file": f"frames/{f.name}", "t": times.get(i), "type": "key"}
|
|
46
|
+
for i, f in enumerate(frames)]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def anchors(out_dir: Path, duration: float) -> list[dict]:
|
|
50
|
+
rows = []
|
|
51
|
+
if (out_dir / "frames" / "anchor_first.jpg").exists():
|
|
52
|
+
rows.append({"file": "frames/anchor_first.jpg", "t": 0.0, "type": "anchor"})
|
|
53
|
+
if (out_dir / "frames" / "anchor_last.jpg").exists():
|
|
54
|
+
t = round(duration, 2) if duration > 0 else None
|
|
55
|
+
rows.append({"file": "frames/anchor_last.jpg", "t": t, "type": "anchor"})
|
|
56
|
+
return rows
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def fmt(t):
|
|
60
|
+
if t is None:
|
|
61
|
+
return "??:??"
|
|
62
|
+
m, s = divmod(int(t), 60)
|
|
63
|
+
return f"{m:02d}:{s:02d}"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def dedupe(rows: list[dict]) -> list[dict]:
|
|
67
|
+
"""Keep all anchors; drop a non-anchor frame within DEDUPE_SEC of the last kept."""
|
|
68
|
+
kept: list[dict] = []
|
|
69
|
+
last_t = None
|
|
70
|
+
for r in rows:
|
|
71
|
+
is_anchor = r["type"] == "anchor"
|
|
72
|
+
if (not is_anchor and last_t is not None and r["t"] is not None
|
|
73
|
+
and abs(r["t"] - last_t) < DEDUPE_SEC):
|
|
74
|
+
continue
|
|
75
|
+
kept.append(r)
|
|
76
|
+
if r["t"] is not None:
|
|
77
|
+
last_t = r["t"]
|
|
78
|
+
return kept
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def label_types(rows: list[dict]) -> None:
|
|
82
|
+
"""Reconstruct scene/gap from spacing. Frames spaced >= ~MAX_GAP are periodic
|
|
83
|
+
fills ('gap'); closer ones were pulled in early by a scene change ('scene')."""
|
|
84
|
+
near = MAX_GAP * 0.9
|
|
85
|
+
prev_t = 0.0
|
|
86
|
+
for r in rows:
|
|
87
|
+
if r["type"] == "anchor":
|
|
88
|
+
if r["t"] is not None:
|
|
89
|
+
prev_t = r["t"]
|
|
90
|
+
continue
|
|
91
|
+
if r["t"] is None:
|
|
92
|
+
r["type"] = "scene"
|
|
93
|
+
continue
|
|
94
|
+
r["type"] = "gap" if (r["t"] - prev_t) >= near else "scene"
|
|
95
|
+
prev_t = r["t"]
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def coverage(rows: list[dict], duration: float) -> tuple[float, float]:
|
|
99
|
+
"""Largest unsampled stretch (s) and where it starts, over [0, duration]."""
|
|
100
|
+
ts = sorted(r["t"] for r in rows if r["t"] is not None)
|
|
101
|
+
if not ts:
|
|
102
|
+
return (duration, 0.0)
|
|
103
|
+
# Bound the right edge with the real duration; if ffprobe couldn't read it (0), fall
|
|
104
|
+
# back to the last sampled time so the tail gap (last frame -> end) still surfaces
|
|
105
|
+
# rather than being silently dropped.
|
|
106
|
+
right = round(duration, 2) if duration > 0 else ts[-1]
|
|
107
|
+
bounds = [0.0] + ts + [right]
|
|
108
|
+
worst, at = 0.0, 0.0
|
|
109
|
+
for a, b in zip(bounds, bounds[1:]):
|
|
110
|
+
if b - a > worst:
|
|
111
|
+
worst, at = b - a, a
|
|
112
|
+
return (worst, at)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def main() -> int:
|
|
116
|
+
if len(sys.argv) < 2:
|
|
117
|
+
print("usage: build_frame_index.py <OUT_DIR> [DURATION_SEC]", file=sys.stderr)
|
|
118
|
+
return 2
|
|
119
|
+
out_dir = Path(sys.argv[1])
|
|
120
|
+
duration = float(sys.argv[2]) if len(sys.argv) > 2 else 0.0
|
|
121
|
+
|
|
122
|
+
rows = collect(out_dir, "frame", "frame_times.txt")
|
|
123
|
+
rows += anchors(out_dir, duration)
|
|
124
|
+
# Sort by time (unknown timestamps sink to the end); on a tie, anchors sort FIRST so
|
|
125
|
+
# the eq(n,0) seed frame at t=0 (and any selected frame coincident with anchor_last)
|
|
126
|
+
# dedupes away against the identical anchor instead of producing a duplicate entry.
|
|
127
|
+
rows.sort(key=lambda r: (r["t"] is None, r["t"] or 0.0, r["type"] != "anchor"))
|
|
128
|
+
rows = dedupe(rows)
|
|
129
|
+
label_types(rows)
|
|
130
|
+
for r in rows:
|
|
131
|
+
r["at"] = fmt(r["t"])
|
|
132
|
+
|
|
133
|
+
(out_dir / "frames.json").write_text(json.dumps(rows, ensure_ascii=False, indent=2))
|
|
134
|
+
|
|
135
|
+
n_anchor = sum(r["type"] == "anchor" for r in rows)
|
|
136
|
+
worst, at = coverage(rows, duration)
|
|
137
|
+
print(f" frames.json: {len(rows)} frames indexed ({n_anchor} anchors, dedupe<{DEDUPE_SEC}s)")
|
|
138
|
+
print(f" coverage: longest unsampled gap = {worst:.1f}s at {fmt(at)} (target MAX_GAP={MAX_GAP:g}s)")
|
|
139
|
+
if duration > 0 and worst > MAX_GAP * 2:
|
|
140
|
+
print(f" ⚠ coverage gap is >2x MAX_GAP — frames may be missing around {fmt(at)}. "
|
|
141
|
+
f"Re-run denser: --max-gap {MAX_GAP / 2:g} (or --mode ui for app recordings).")
|
|
142
|
+
return 0
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
if __name__ == "__main__":
|
|
146
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Video -> time-mapped transcript + significant frames (with per-frame timestamps).
|
|
3
|
+
# whisper.cpp (large-v3-turbo by default) for the transcript, ffmpeg for frames.
|
|
4
|
+
#
|
|
5
|
+
# This is the MECHANICAL step. It produces artifacts; it does NOT understand the
|
|
6
|
+
# video. Claude reads the outputs (transcript + frames.json), decides which frames
|
|
7
|
+
# matter, views them, and writes the curated <video>.transcript.md (see SKILL.md).
|
|
8
|
+
#
|
|
9
|
+
# Usage: ./transcribe.sh "/abs/path/to/video.mov" [lang] [flags]
|
|
10
|
+
# lang: whisper language code. DEFAULT 'auto' — whisper detects it, no need to pass.
|
|
11
|
+
# Force only if auto mislabels a short/ambiguous clip (e.g. 'tr', 'en').
|
|
12
|
+
#
|
|
13
|
+
# Flags (also settable as env vars):
|
|
14
|
+
# --frames-only skip whisper entirely; extract frames only. For pure
|
|
15
|
+
# UI/UX teardowns where no transcript is needed. (FRAMES_ONLY=1)
|
|
16
|
+
# --mode ui|lecture sampling preset. 'ui' = MAX_GAP 2.5s (app screens linger
|
|
17
|
+
# 3-5s and change by text only, below scene-detect). 'lecture'
|
|
18
|
+
# (default) = MAX_GAP 10s for talking-head video. (MODE=ui)
|
|
19
|
+
# --max-gap N guarantee a frame at least every N seconds. Overrides the
|
|
20
|
+
# mode preset. (MAX_GAP=N)
|
|
21
|
+
# --scene-threshold N ffmpeg scene-change sensitivity, lower = more frames. (SCENE_THRESHOLD=N)
|
|
22
|
+
# --contact-sheet also emit frames/contact_sheet.jpg — a tiled montage of every
|
|
23
|
+
# frame, so coverage is verifiable in one glance. (CONTACT_SHEET=1)
|
|
24
|
+
# --lang CODE same as the positional lang arg.
|
|
25
|
+
#
|
|
26
|
+
# Env overrides:
|
|
27
|
+
# WHISPER_MODEL=/abs/path/ggml-*.bin force a specific model
|
|
28
|
+
# OUT_DIR=/abs/path where artifacts go (default: <video_dir>/<slug>.media)
|
|
29
|
+
# NOTE: avoid ':' in OUT_DIR — it is special inside the
|
|
30
|
+
# ffmpeg filter graph and would truncate the metadata path.
|
|
31
|
+
# DEDUPE_SEC=0.4 collapse non-anchor frames closer than this
|
|
32
|
+
#
|
|
33
|
+
# Outputs (in OUT_DIR):
|
|
34
|
+
# transcript.srt timestamped segments (human-friendly) [skipped with --frames-only]
|
|
35
|
+
# transcript.json/.txt/.vtt [skipped with --frames-only]
|
|
36
|
+
# frames/frame_*.jpg selected frames (scene change OR every MAX_GAP, whichever first)
|
|
37
|
+
# frames/anchor_first.jpg / anchor_last.jpg always-captured first + last frame
|
|
38
|
+
# frames/contact_sheet.jpg tiled montage of all frames [--contact-sheet only]
|
|
39
|
+
# frames.json [{file, t, at, type}] — each frame's pts_time, sorted (Claude reads this)
|
|
40
|
+
# source-meta.txt ffprobe dump (audio.wav is created then deleted after transcription)
|
|
41
|
+
set -euo pipefail
|
|
42
|
+
|
|
43
|
+
# --- arg + flag parsing ---------------------------------------------------------
|
|
44
|
+
# Positional: first non-flag = VIDEO, second non-flag = LANG_CODE (back-compatible).
|
|
45
|
+
VIDEO=""
|
|
46
|
+
LANG_CODE=""
|
|
47
|
+
FRAMES_ONLY="${FRAMES_ONLY:-0}"
|
|
48
|
+
MODE="${MODE:-lecture}"
|
|
49
|
+
CONTACT_SHEET="${CONTACT_SHEET:-0}"
|
|
50
|
+
# MAX_GAP / SCENE_THRESHOLD: track whether explicitly set so a CLI/env value beats the
|
|
51
|
+
# mode preset. Empty here means "fall back to the mode preset" computed below.
|
|
52
|
+
MAX_GAP="${MAX_GAP:-}"
|
|
53
|
+
SCENE_THRESHOLD="${SCENE_THRESHOLD:-}"
|
|
54
|
+
|
|
55
|
+
while [ $# -gt 0 ]; do
|
|
56
|
+
case "$1" in
|
|
57
|
+
--frames-only) FRAMES_ONLY=1 ;;
|
|
58
|
+
--contact-sheet) CONTACT_SHEET=1 ;;
|
|
59
|
+
--mode) MODE="${2:?--mode needs a value}"; shift ;;
|
|
60
|
+
--mode=*) MODE="${1#*=}" ;;
|
|
61
|
+
--max-gap) MAX_GAP="${2:?--max-gap needs a value}"; shift ;;
|
|
62
|
+
--max-gap=*) MAX_GAP="${1#*=}" ;;
|
|
63
|
+
--scene-threshold) SCENE_THRESHOLD="${2:?--scene-threshold needs a value}"; shift ;;
|
|
64
|
+
--scene-threshold=*) SCENE_THRESHOLD="${1#*=}" ;;
|
|
65
|
+
--lang) LANG_CODE="${2:?--lang needs a value}"; shift ;;
|
|
66
|
+
--lang=*) LANG_CODE="${1#*=}" ;;
|
|
67
|
+
--*) echo "unknown flag: $1" >&2; exit 2 ;;
|
|
68
|
+
*) if [ -z "$VIDEO" ]; then VIDEO="$1"
|
|
69
|
+
elif [ -z "$LANG_CODE" ]; then LANG_CODE="$1"
|
|
70
|
+
else echo "unexpected arg: $1" >&2; exit 2; fi ;;
|
|
71
|
+
esac
|
|
72
|
+
shift
|
|
73
|
+
done
|
|
74
|
+
|
|
75
|
+
[ -n "$VIDEO" ] || { echo "Usage: transcribe.sh <video> [lang] [--frames-only] [--mode ui] [--max-gap N]" >&2; exit 1; }
|
|
76
|
+
LANG_CODE="${LANG_CODE:-auto}" # whisper auto-detects the spoken language unless overridden
|
|
77
|
+
|
|
78
|
+
# Mode presets fill in only what the caller left unset (explicit CLI/env always wins).
|
|
79
|
+
case "$MODE" in
|
|
80
|
+
ui) MODE_MAX_GAP=2.5 ;;
|
|
81
|
+
lecture) MODE_MAX_GAP=10 ;;
|
|
82
|
+
*) echo "unknown --mode '$MODE' (use ui|lecture)" >&2; exit 2 ;;
|
|
83
|
+
esac
|
|
84
|
+
MAX_GAP="${MAX_GAP:-$MODE_MAX_GAP}"
|
|
85
|
+
SCENE_THRESHOLD="${SCENE_THRESHOLD:-0.15}"
|
|
86
|
+
|
|
87
|
+
# Guard the two numeric knobs before they reach the ffmpeg filter / Python: a non-number
|
|
88
|
+
# would crash build_frame_index (float()), and MAX_GAP=0 makes gte(t-prev,0) select EVERY
|
|
89
|
+
# frame (disk-fill footgun). Fail loudly instead.
|
|
90
|
+
is_num='^[0-9]+([.][0-9]+)?$'
|
|
91
|
+
[[ "$MAX_GAP" =~ $is_num ]] || { echo "--max-gap must be a number (got '$MAX_GAP')" >&2; exit 2; }
|
|
92
|
+
awk "BEGIN{exit !($MAX_GAP > 0)}" || { echo "--max-gap must be > 0 (got '$MAX_GAP')" >&2; exit 2; }
|
|
93
|
+
[[ "$SCENE_THRESHOLD" =~ $is_num ]] || { echo "--scene-threshold must be a number (got '$SCENE_THRESHOLD')" >&2; exit 2; }
|
|
94
|
+
|
|
95
|
+
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
96
|
+
|
|
97
|
+
# --- remote link? download with yt-dlp (local files skip this) ------------------
|
|
98
|
+
if [[ "$VIDEO" =~ ^https?:// ]]; then
|
|
99
|
+
command -v yt-dlp >/dev/null || { echo "remote link needs yt-dlp: brew install yt-dlp" >&2; exit 1; }
|
|
100
|
+
DL_DIR="${OUT_DIR:-$PWD}/_download"; mkdir -p "$DL_DIR"
|
|
101
|
+
echo "==> downloading remote video with yt-dlp"
|
|
102
|
+
yt-dlp -o "$DL_DIR/%(title)s.%(ext)s" "$VIDEO"
|
|
103
|
+
VIDEO="$(ls -t "$DL_DIR"/* | head -1)"
|
|
104
|
+
echo "==> downloaded: $VIDEO"
|
|
105
|
+
fi
|
|
106
|
+
[ -f "$VIDEO" ] || { echo "video not found: $VIDEO" >&2; exit 1; }
|
|
107
|
+
|
|
108
|
+
command -v ffmpeg >/dev/null || { echo "ffmpeg not found (brew install ffmpeg)" >&2; exit 1; }
|
|
109
|
+
|
|
110
|
+
# --- model: prefer turbo, then large-v3, then medium (override with WHISPER_MODEL)
|
|
111
|
+
# Only needed for transcription — skip the whole resolution when --frames-only.
|
|
112
|
+
pick_model() {
|
|
113
|
+
if [ -n "${WHISPER_MODEL:-}" ]; then printf '%s' "$WHISPER_MODEL"; return; fi
|
|
114
|
+
local p
|
|
115
|
+
for p in \
|
|
116
|
+
"$HOME/.cache/whisper.cpp/models/ggml-large-v3-turbo.bin" \
|
|
117
|
+
"$HOME/.cache/openwhispr/whisper-models/ggml-large-v3-turbo.bin" \
|
|
118
|
+
"$HOME/.cache/whisper.cpp/models/ggml-large-v3.bin" \
|
|
119
|
+
"$HOME/.cache/whisper.cpp/models/ggml-medium.bin" \
|
|
120
|
+
"$HOME/.cache/openwhispr/whisper-models/ggml-medium.bin"; do
|
|
121
|
+
[ -f "$p" ] && { printf '%s' "$p"; return; }
|
|
122
|
+
done
|
|
123
|
+
printf ''
|
|
124
|
+
}
|
|
125
|
+
if [ "$FRAMES_ONLY" != "1" ]; then
|
|
126
|
+
MODEL="$(pick_model)"
|
|
127
|
+
[ -n "$MODEL" ] && [ -f "$MODEL" ] || {
|
|
128
|
+
echo "no whisper model found. Install one, e.g.:" >&2
|
|
129
|
+
echo " curl -L -o ~/.cache/whisper.cpp/models/ggml-large-v3-turbo.bin \\" >&2
|
|
130
|
+
echo " https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin" >&2
|
|
131
|
+
echo " or set WHISPER_MODEL=/abs/path/ggml-*.bin" >&2
|
|
132
|
+
echo " (or pass --frames-only to skip transcription entirely)" >&2
|
|
133
|
+
exit 1
|
|
134
|
+
}
|
|
135
|
+
WHISPER_BIN="$(command -v whisper-cli || command -v whisper-cpp || command -v main || true)"
|
|
136
|
+
[ -n "$WHISPER_BIN" ] || { echo "whisper.cpp binary not found (brew install whisper-cpp). Or pass --frames-only." >&2; exit 1; }
|
|
137
|
+
fi
|
|
138
|
+
|
|
139
|
+
# --- where artifacts go: next to the video by default ---------------------------
|
|
140
|
+
SRC_DIR="$(cd "$(dirname "$VIDEO")" && pwd)"
|
|
141
|
+
SLUG="$(basename "$VIDEO")"; SLUG="${SLUG%.*}"
|
|
142
|
+
SLUG="$(echo "$SLUG" | tr ' ' '-' | tr -cd '[:alnum:]._-' | sed 's/--*/-/g; s/^-//; s/-$//')"
|
|
143
|
+
OUT="${OUT_DIR:-$SRC_DIR/$SLUG.media}"
|
|
144
|
+
mkdir -p "$OUT/frames"
|
|
145
|
+
# even-dim scaling keeps the mjpeg encoder happy on odd-width sources
|
|
146
|
+
SCALE="scale=trunc(iw/2)*2:trunc(ih/2)*2"
|
|
147
|
+
|
|
148
|
+
echo "==> [$SLUG] probing (mode=$MODE, max_gap=${MAX_GAP}s, scene>$SCENE_THRESHOLD, frames_only=$FRAMES_ONLY)"
|
|
149
|
+
ffprobe -v error -show_entries format=duration,size:stream=codec_type,codec_name,width,height \
|
|
150
|
+
-of default=noprint_wrappers=1 "$VIDEO" | tee "$OUT/source-meta.txt"
|
|
151
|
+
DURATION="$(ffprobe -v error -show_entries format=duration -of csv=p=0 "$VIDEO" 2>/dev/null | head -1)"
|
|
152
|
+
DURATION="${DURATION:-0}"
|
|
153
|
+
|
|
154
|
+
if [ "$FRAMES_ONLY" = "1" ]; then
|
|
155
|
+
echo "==> [$SLUG] --frames-only: skipping audio + transcription"
|
|
156
|
+
else
|
|
157
|
+
echo "==> [$SLUG] extracting 16kHz mono audio"
|
|
158
|
+
ffmpeg -y -loglevel error -i "$VIDEO" -ar 16000 -ac 1 -c:a pcm_s16le "$OUT/audio.wav"
|
|
159
|
+
|
|
160
|
+
echo "==> [$SLUG] transcribing with $(basename "$MODEL") (lang=$LANG_CODE)"
|
|
161
|
+
"$WHISPER_BIN" -m "$MODEL" -f "$OUT/audio.wav" -l "$LANG_CODE" \
|
|
162
|
+
--output-txt --output-srt --output-vtt --output-json -of "$OUT/transcript" -pp
|
|
163
|
+
# audio.wav is a pure whisper intermediate (~1.9MB/min) — drop it once the transcript exists.
|
|
164
|
+
rm -f "$OUT/audio.wav"
|
|
165
|
+
fi
|
|
166
|
+
|
|
167
|
+
echo "==> [$SLUG] extracting frames (scene change OR every ${MAX_GAP}s, whichever fires first)"
|
|
168
|
+
# ONE decode pass, time-based and frame-rate-independent:
|
|
169
|
+
# eq(n,0) always seed the first frame (also primes prev_selected_t)
|
|
170
|
+
# gt(scene,$SCENE_THRESHOLD) a scene change (slide/UI transition) fired
|
|
171
|
+
# gte(t-prev_selected_t,MAX_GAP) MAX_GAP elapsed since the last KEPT frame — guarantees
|
|
172
|
+
# coverage of static stretches scene-detect misses (app
|
|
173
|
+
# screens that change by text only). prev_selected_t is
|
|
174
|
+
# wall-clock seconds, so this is robust on variable-frame-rate
|
|
175
|
+
# screen recordings where frame-number sampling (mod(n,N)) breaks.
|
|
176
|
+
# pix_fmt yuvj420p avoids mjpeg "non full-range YUV" failures on screen recordings;
|
|
177
|
+
# metadata=print dumps each kept frame's pts_time so frames map back to the timeline.
|
|
178
|
+
ffmpeg -y -loglevel error -i "$VIDEO" \
|
|
179
|
+
-vf "select='eq(n,0)+gt(scene,$SCENE_THRESHOLD)+gte(t-prev_selected_t,$MAX_GAP)',metadata=print:file=$OUT/frames/frame_times.txt,scale=trunc(iw/2)*2:trunc(ih/2)*2" \
|
|
180
|
+
-fps_mode vfr -pix_fmt yuvj420p -q:v 3 "$OUT/frames/frame_%04d.jpg" || true
|
|
181
|
+
|
|
182
|
+
# The || true above keeps a partial result usable, but zero frames means the decode
|
|
183
|
+
# produced nothing (no decodable video stream, unsupported codec, write failure). That
|
|
184
|
+
# must fail loudly — silently leaving an empty frames.json is the exact under-sampling
|
|
185
|
+
# trap issue #15 is about. Count safely (|| true so a no-match never trips set -e).
|
|
186
|
+
NSEL="$(find "$OUT/frames" -name 'frame_*.jpg' 2>/dev/null | wc -l | tr -d ' ')"
|
|
187
|
+
if [ "$NSEL" -eq 0 ]; then
|
|
188
|
+
echo "ERROR: ffmpeg extracted 0 frames from '$VIDEO'." >&2
|
|
189
|
+
echo " The file may have no decodable video stream (audio-only?), an unsupported codec," >&2
|
|
190
|
+
echo " or the output dir isn't writable. No frames.json written." >&2
|
|
191
|
+
exit 1
|
|
192
|
+
fi
|
|
193
|
+
|
|
194
|
+
echo "==> [$SLUG] capturing first + last anchor frames"
|
|
195
|
+
# Short, cut-heavy creatives carry the most information in the opening hook and the
|
|
196
|
+
# closing CTA — but scene-detect rarely fires at t=0 or t=end. Always grab both.
|
|
197
|
+
ffmpeg -y -loglevel error -i "$VIDEO" -vf "$SCALE" -frames:v 1 -q:v 3 \
|
|
198
|
+
"$OUT/frames/anchor_first.jpg" || true
|
|
199
|
+
ffmpeg -y -loglevel error -sseof -0.5 -i "$VIDEO" -vf "$SCALE" -update 1 -frames:v 1 -q:v 3 \
|
|
200
|
+
"$OUT/frames/anchor_last.jpg" || true
|
|
201
|
+
|
|
202
|
+
echo "==> [$SLUG] building frames.json (timestamp index, dedup near-dups, coverage check)"
|
|
203
|
+
MAX_GAP="$MAX_GAP" python3 "$SCRIPT_DIR/build_frame_index.py" "$OUT" "$DURATION"
|
|
204
|
+
|
|
205
|
+
# --- optional contact sheet: one glance to verify coverage before deep analysis --
|
|
206
|
+
if [ "$CONTACT_SHEET" = "1" ]; then
|
|
207
|
+
echo "==> [$SLUG] building contact sheet (coverage montage)"
|
|
208
|
+
NF="$(ls -1 "$OUT/frames"/frame_*.jpg 2>/dev/null | wc -l | tr -d ' ')"
|
|
209
|
+
if [ "$NF" -gt 0 ]; then
|
|
210
|
+
COLS="$(awk -v n="$NF" 'BEGIN{c=int(sqrt(n)); if(c*c<n)c++; print c}')"
|
|
211
|
+
ROWS="$(awk -v n="$NF" -v c="$COLS" 'BEGIN{r=int(n/c); if(r*c<n)r++; print r}')"
|
|
212
|
+
ffmpeg -y -loglevel error -pattern_type glob -i "$OUT/frames/frame_*.jpg" \
|
|
213
|
+
-vf "scale=240:-1,tile=${COLS}x${ROWS}:padding=4:margin=4" -frames:v 1 -q:v 4 \
|
|
214
|
+
"$OUT/frames/contact_sheet.jpg" \
|
|
215
|
+
&& echo " contact sheet: $OUT/frames/contact_sheet.jpg (${COLS}x${ROWS})" \
|
|
216
|
+
|| echo " contact sheet skipped (montage failed)"
|
|
217
|
+
fi
|
|
218
|
+
fi
|
|
219
|
+
|
|
220
|
+
# find -not -name keeps the count set -e-safe (grep -v exits 1 on no-match, tripping pipefail).
|
|
221
|
+
NFRAMES="$(find "$OUT/frames" -name '*.jpg' ! -name 'contact_sheet.jpg' 2>/dev/null | wc -l | tr -d ' ')"
|
|
222
|
+
echo "==> [$SLUG] DONE"
|
|
223
|
+
[ "$FRAMES_ONLY" = "1" ] || echo " transcript: $OUT/transcript.srt"
|
|
224
|
+
echo " frames: $NFRAMES (index: $OUT/frames.json)"
|
|
225
|
+
if [ "$FRAMES_ONLY" = "1" ]; then
|
|
226
|
+
echo " next: Claude reads frames.json, views key frames, writes the curated screen list / transcript.md"
|
|
227
|
+
else
|
|
228
|
+
echo " next: Claude reads transcript + frames.json, views key frames, writes the curated transcript.md"
|
|
229
|
+
fi
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{aq as o,ar as n}from"./index-Ds-JXWtr.js";const t=(r,a)=>o.lang.round(n.parse(r)[a]);export{t as c};
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{s as a,c as s,a as e,C as t}from"./chunk-4TB4RGXK-Bh7qSu_X.js";import{_ as i}from"./index-Ds-JXWtr.js";import"./chunk-FMBD7UC4-B5r8U2gJ.js";import"./chunk-YZCP3GAM-Dl9MlwCW.js";import"./chunk-55IACEB6-gNeHRO_i.js";import"./chunk-EDXVE4YY-BRniAH53.js";var u={parser:e,get db(){return new t},renderer:s,styles:a,init:i(r=>{r.class||(r.class={}),r.class.arrowMarkerAbsolute=r.arrowMarkerAbsolute},"init")};export{u as diagram};
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{s as a,c as s,a as e,C as t}from"./chunk-4TB4RGXK-Bh7qSu_X.js";import{_ as i}from"./index-Ds-JXWtr.js";import"./chunk-FMBD7UC4-B5r8U2gJ.js";import"./chunk-YZCP3GAM-Dl9MlwCW.js";import"./chunk-55IACEB6-gNeHRO_i.js";import"./chunk-EDXVE4YY-BRniAH53.js";var u={parser:e,get db(){return new t},renderer:s,styles:a,init:i(r=>{r.class||(r.class={}),r.class.arrowMarkerAbsolute=r.arrowMarkerAbsolute},"init")};export{u as diagram};
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{b as r}from"./graph-D58deEXr.js";var e=4;function a(o){return r(o,e)}export{a as c};
|