dreamcontext 0.6.0 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/agents/sleep-product.md +19 -2
- package/agents/sleep-state.md +43 -22
- package/agents/sleep-tasks.md +38 -0
- package/dist/agents/sleep-product.md +19 -2
- package/dist/agents/sleep-state.md +43 -22
- package/dist/agents/sleep-tasks.md +38 -0
- package/dist/dashboard/assets/{BrainCanvas3D-LLqVeXtb.js → BrainCanvas3D-Bb8WCKrn.js} +21 -21
- package/dist/dashboard/assets/{_baseUniq-DW0uA0ty.js → _baseUniq-BatmaIH2.js} +1 -1
- package/dist/dashboard/assets/ar-SA-G6X2FPQ2-D5ja_wjH.js +10 -0
- package/dist/dashboard/assets/{arc-fVUGtlbL.js → arc-DqjLlF-x.js} +1 -1
- package/dist/dashboard/assets/{architectureDiagram-Q4EWVU46-Dz2PKBsS.js → architectureDiagram-Q4EWVU46-u9hbL6uu.js} +1 -1
- package/dist/dashboard/assets/az-AZ-76LH7QW2-CCrmRwql.js +1 -0
- package/dist/dashboard/assets/bg-BG-XCXSNQG7-ByFwPOzm.js +5 -0
- package/dist/dashboard/assets/{blockDiagram-DXYQGD6D-ByWKxzhL.js → blockDiagram-DXYQGD6D-dX-e3JNh.js} +1 -1
- package/dist/dashboard/assets/bn-BD-2XOGV67Q-CqnEIuTG.js +5 -0
- package/dist/dashboard/assets/{c4Diagram-AHTNJAMY-dpHsVM3D.js → c4Diagram-AHTNJAMY-DAT2Y4Ei.js} +1 -1
- package/dist/dashboard/assets/ca-ES-6MX7JW3Y-DxUow-Oi.js +8 -0
- package/dist/dashboard/assets/channel-JlaJkLX7.js +1 -0
- package/dist/dashboard/assets/{chunk-4BX2VUAB-HEXb6Yg5.js → chunk-4BX2VUAB-ng6_uP1b.js} +1 -1
- package/dist/dashboard/assets/{chunk-4TB4RGXK-DeVy5g6H.js → chunk-4TB4RGXK-JyQRd6F4.js} +1 -1
- package/dist/dashboard/assets/{chunk-55IACEB6-DGy3ZgDZ.js → chunk-55IACEB6-WENWKOcW.js} +1 -1
- package/dist/dashboard/assets/{chunk-EDXVE4YY-CYJehjz4.js → chunk-EDXVE4YY-BupZpcZ0.js} +1 -1
- package/dist/dashboard/assets/{chunk-FMBD7UC4-DKTOmvgZ.js → chunk-FMBD7UC4-6-Dp_BQr.js} +1 -1
- package/dist/dashboard/assets/{chunk-OYMX7WX6-mmtWyiDA.js → chunk-OYMX7WX6-CRMXMXMR.js} +1 -1
- package/dist/dashboard/assets/{chunk-QZHKN3VN-DSdV0iwY.js → chunk-QZHKN3VN-CdV3chea.js} +1 -1
- package/dist/dashboard/assets/{chunk-YZCP3GAM-DbY-KoDn.js → chunk-YZCP3GAM-DGd7UTQr.js} +1 -1
- package/dist/dashboard/assets/classDiagram-6PBFFD2Q-BBVZgLT7.js +1 -0
- package/dist/dashboard/assets/classDiagram-v2-HSJHXN6E-BBVZgLT7.js +1 -0
- package/dist/dashboard/assets/clone-DSjY5D8g.js +1 -0
- package/dist/dashboard/assets/{cose-bilkent-S5V4N54A-p0t1DY88.js → cose-bilkent-S5V4N54A-86wk8_0k.js} +1 -1
- package/dist/dashboard/assets/cs-CZ-2BRQDIVT-Vsx90G3k.js +11 -0
- package/dist/dashboard/assets/da-DK-5WZEPLOC-BqlSmIrC.js +5 -0
- package/dist/dashboard/assets/{dagre-KV5264BT-DYE0WzHM.js → dagre-KV5264BT-BsnIu1de.js} +1 -1
- package/dist/dashboard/assets/de-DE-XR44H4JA-D1X2JmBd.js +8 -0
- package/dist/dashboard/assets/{diagram-5BDNPKRD-D9EiQCOP.js → diagram-5BDNPKRD-BfKZpkO_.js} +1 -1
- package/dist/dashboard/assets/{diagram-G4DWMVQ6-poDXSAfu.js → diagram-G4DWMVQ6-BGEdOZis.js} +1 -1
- package/dist/dashboard/assets/{diagram-MMDJMWI5-CHZ7wgN1.js → diagram-MMDJMWI5-4PK6U-9k.js} +1 -1
- package/dist/dashboard/assets/{diagram-TYMM5635-DVk6fYG7.js → diagram-TYMM5635-Co4XFxnB.js} +1 -1
- package/dist/dashboard/assets/directory-open-01563666-DWU9wJ6I.js +1 -0
- package/dist/dashboard/assets/directory-open-4ed118d0-CunoC1EB.js +1 -0
- package/dist/dashboard/assets/el-GR-BZB4AONW-JfJ7Iw6d.js +10 -0
- package/dist/dashboard/assets/{erDiagram-SMLLAGMA-D2sqkGin.js → erDiagram-SMLLAGMA-tggQbGeM.js} +1 -1
- package/dist/dashboard/assets/es-ES-U4NZUMDT-CrTG5SH1.js +9 -0
- package/dist/dashboard/assets/eu-ES-A7QVB2H4-CLi-f2S4.js +11 -0
- package/dist/dashboard/assets/extends-CF3RwP-h.js +1 -0
- package/dist/dashboard/assets/fa-IR-HGAKTJCU-B78t-ifl.js +8 -0
- package/dist/dashboard/assets/fi-FI-Z5N7JZ37-BiYGo0tM.js +6 -0
- package/dist/dashboard/assets/file-open-002ab408-DIuFHtCF.js +1 -0
- package/dist/dashboard/assets/file-open-7c801643-684qeFg4.js +1 -0
- package/dist/dashboard/assets/file-save-3189631c-C1wFhQhH.js +1 -0
- package/dist/dashboard/assets/file-save-745eba88-Bb9F9Kg7.js +1 -0
- package/dist/dashboard/assets/{flowDiagram-DWJPFMVM-DrMlKBYA.js → flowDiagram-DWJPFMVM-DfOOuTHa.js} +1 -1
- package/dist/dashboard/assets/fr-FR-RHASNOE6-CHuvxlm9.js +9 -0
- package/dist/dashboard/assets/{ganttDiagram-T4ZO3ILL-Cf4FbEFp.js → ganttDiagram-T4ZO3ILL-CqRBfUlO.js} +1 -1
- package/dist/dashboard/assets/{gitGraphDiagram-UUTBAWPF-BsbOq1C9.js → gitGraphDiagram-UUTBAWPF-DCooi0ou.js} +1 -1
- package/dist/dashboard/assets/gl-ES-HMX3MZ6V-DloFVgGH.js +10 -0
- package/dist/dashboard/assets/{graph-CUxxgCtS.js → graph-CbTgvSod.js} +1 -1
- package/dist/dashboard/assets/he-IL-6SHJWFNN-sKyHtCj5.js +10 -0
- package/dist/dashboard/assets/hi-IN-IWLTKZ5I-BDBktoGz.js +4 -0
- package/dist/dashboard/assets/hu-HU-A5ZG7DT2-CI0m6WdK.js +7 -0
- package/dist/dashboard/assets/id-ID-SAP4L64H-o3oCWYJ1.js +10 -0
- package/dist/dashboard/assets/image-blob-reduce.esm-D6s-rqMO.js +7 -0
- package/dist/dashboard/assets/index-BMwAG0PF.css +1 -0
- package/dist/dashboard/assets/index-BzUItYQF.js +19 -0
- package/dist/dashboard/assets/index-CIMJcxbn.js +480 -0
- package/dist/dashboard/assets/{infoDiagram-42DDH7IO-DTkMnZiD.js → infoDiagram-42DDH7IO-DvXAiIKM.js} +1 -1
- package/dist/dashboard/assets/{ishikawaDiagram-UXIWVN3A-CahQ348K.js → ishikawaDiagram-UXIWVN3A-B3k3F-SR.js} +1 -1
- package/dist/dashboard/assets/it-IT-JPQ66NNP-B69FvIBl.js +11 -0
- package/dist/dashboard/assets/ja-JP-DBVTYXUO-09Dz5R3w.js +8 -0
- package/dist/dashboard/assets/{journeyDiagram-VCZTEJTY-BDY2Nslc.js → journeyDiagram-VCZTEJTY-3sYlQlzX.js} +1 -1
- package/dist/dashboard/assets/kaa-6HZHGXH3-SFvv5G_m.js +1 -0
- package/dist/dashboard/assets/kab-KAB-ZGHBKWFO-Cj4UcDfk.js +8 -0
- package/dist/dashboard/assets/{kanban-definition-6JOO6SKY-CgrwoTjI.js → kanban-definition-6JOO6SKY-cGLqpTJv.js} +1 -1
- package/dist/dashboard/assets/kk-KZ-P5N5QNE5-BVT_-IeJ.js +1 -0
- package/dist/dashboard/assets/km-KH-HSX4SM5Z-BpH_zneK.js +11 -0
- package/dist/dashboard/assets/ko-KR-MTYHY66A-BWtVfnIt.js +9 -0
- package/dist/dashboard/assets/ku-TR-6OUDTVRD-DKg1sGxW.js +9 -0
- package/dist/dashboard/assets/{layout-IaUxkFkm.js → layout-gKZamora.js} +1 -1
- package/dist/dashboard/assets/{linear-shc0iNFn.js → linear-DOSQ02II.js} +1 -1
- package/dist/dashboard/assets/lt-LT-XHIRWOB4-DFICvEvq.js +3 -0
- package/dist/dashboard/assets/lv-LV-5QDEKY6T-CeIp3POy.js +7 -0
- package/dist/dashboard/assets/min-BD9nuOyV.js +1 -0
- package/dist/dashboard/assets/{mindmap-definition-QFDTVHPH-GmO0JRtA.js → mindmap-definition-QFDTVHPH-DNg9-qbI.js} +7 -7
- package/dist/dashboard/assets/mr-IN-CRQNXWMA-CMiofriy.js +13 -0
- package/dist/dashboard/assets/my-MM-5M5IBNSE-WhbW1bUQ.js +1 -0
- package/dist/dashboard/assets/nb-NO-T6EIAALU-kvj3GRtN.js +10 -0
- package/dist/dashboard/assets/nl-NL-IS3SIHDZ-BdYOVmBV.js +8 -0
- package/dist/dashboard/assets/nn-NO-6E72VCQL-BAI1MQoO.js +8 -0
- package/dist/dashboard/assets/oc-FR-POXYY2M6-ChavJGTr.js +8 -0
- package/dist/dashboard/assets/pa-IN-N4M65BXN-AWmvLRaa.js +4 -0
- package/dist/dashboard/assets/percentages-BXMCSKIN-QYboYPU4.js +215 -0
- package/dist/dashboard/assets/pica-DzAIGpll.js +7 -0
- package/dist/dashboard/assets/{pieDiagram-DEJITSTG-Bfm_toYA.js → pieDiagram-DEJITSTG-C4ZFpuC0.js} +1 -1
- package/dist/dashboard/assets/pl-PL-T2D74RX3-C06LuJit.js +9 -0
- package/dist/dashboard/assets/pt-BR-5N22H2LF-DP-AXG39.js +9 -0
- package/dist/dashboard/assets/pt-PT-UZXXM6DQ-T-2COPG-.js +9 -0
- package/dist/dashboard/assets/{quadrantDiagram-34T5L4WZ-DIjTM3lm.js → quadrantDiagram-34T5L4WZ-yvEQRUmq.js} +1 -1
- package/dist/dashboard/assets/{requirementDiagram-MS252O5E-B9AEoGYh.js → requirementDiagram-MS252O5E-BkjbZkw1.js} +1 -1
- package/dist/dashboard/assets/ro-RO-JPDTUUEW-V8ShMmhj.js +11 -0
- package/dist/dashboard/assets/roundRect-0PYZxl1G.js +1 -0
- package/dist/dashboard/assets/ru-RU-B4JR7IUQ-Bk5QWjQg.js +9 -0
- package/dist/dashboard/assets/{sankeyDiagram-XADWPNL6-C_O5XLXm.js → sankeyDiagram-XADWPNL6-_6wZr0Q4.js} +1 -1
- package/dist/dashboard/assets/{sequenceDiagram-FGHM5R23-CHSxSvmJ.js → sequenceDiagram-FGHM5R23-B03MUCIO.js} +1 -1
- package/dist/dashboard/assets/si-LK-N5RQ5JYF-DrLyNjX_.js +1 -0
- package/dist/dashboard/assets/sk-SK-C5VTKIMK-MG2XJDav.js +6 -0
- package/dist/dashboard/assets/sl-SI-NN7IZMDC-DyF2jYcb.js +6 -0
- package/dist/dashboard/assets/{stateDiagram-FHFEXIEX-B0URrhHw.js → stateDiagram-FHFEXIEX-DW9W8hRb.js} +1 -1
- package/dist/dashboard/assets/stateDiagram-v2-QKLJ7IA2-DGLHFCvB.js +1 -0
- package/dist/dashboard/assets/subset-shared.chunk-DxnLkUkc.js +84 -0
- package/dist/dashboard/assets/subset-worker.chunk-BCGTROg4.js +1 -0
- package/dist/dashboard/assets/sv-SE-XGPEYMSR-BJOlh7Ta.js +10 -0
- package/dist/dashboard/assets/ta-IN-2NMHFXQM-DbvdzD_M.js +9 -0
- package/dist/dashboard/assets/th-TH-HPSO5L25-BZoStbWu.js +2 -0
- package/dist/dashboard/assets/{timeline-definition-GMOUNBTQ-CycU2gVC.js → timeline-definition-GMOUNBTQ-Cq3P6C9q.js} +1 -1
- package/dist/dashboard/assets/tr-TR-DEFEU3FU-D0q3dp6Y.js +7 -0
- package/dist/dashboard/assets/uk-UA-QMV73CPH-KQC76_Jh.js +6 -0
- package/dist/dashboard/assets/{vennDiagram-DHZGUBPP-DcU2536G.js → vennDiagram-DHZGUBPP-BO3azTdL.js} +1 -1
- package/dist/dashboard/assets/vi-VN-M7AON7JQ-BsyXHyOP.js +5 -0
- package/dist/dashboard/assets/{wardley-RL74JXVD-BWqqaX3S.js → wardley-RL74JXVD-BjaWFGf-.js} +1 -1
- package/dist/dashboard/assets/{wardleyDiagram-NUSXRM2D-CUnIHJgd.js → wardleyDiagram-NUSXRM2D-D0Ghwhzw.js} +1 -1
- package/dist/dashboard/assets/{xychartDiagram-5P7HB3ND-b9fcOYTj.js → xychartDiagram-5P7HB3ND-D1QYeAFg.js} +1 -1
- package/dist/dashboard/assets/zh-CN-LNUGB5OW-Cg54ea9D.js +10 -0
- package/dist/dashboard/assets/zh-HK-E62DVLB3-N7JXSzdg.js +1 -0
- package/dist/dashboard/assets/zh-TW-RAJ6MFWO-CTX6fzNp.js +9 -0
- package/dist/dashboard/index.html +2 -2
- package/dist/index.js +2519 -1270
- package/dist/skill-packs/excalidraw/SKILL.md +82 -3
- package/dist/skill-packs/video-watching/SKILL.md +54 -13
- package/dist/skill-packs/video-watching/scripts/build_frame_index.py +61 -10
- package/dist/skill-packs/video-watching/scripts/transcribe.sh +147 -54
- package/dist/templates/init/data-structures/default.md +26 -24
- package/package.json +1 -1
- package/skill/SKILL.md +48 -12
- package/skill-packs/excalidraw/SKILL.md +82 -3
- package/skill-packs/video-watching/SKILL.md +54 -13
- package/skill-packs/video-watching/scripts/build_frame_index.py +61 -10
- package/skill-packs/video-watching/scripts/transcribe.sh +147 -54
- package/dist/dashboard/assets/channel-zrLwggBX.js +0 -1
- package/dist/dashboard/assets/classDiagram-6PBFFD2Q-BvejNiwH.js +0 -1
- package/dist/dashboard/assets/classDiagram-v2-HSJHXN6E-BvejNiwH.js +0 -1
- package/dist/dashboard/assets/clone-BFVdml6g.js +0 -1
- package/dist/dashboard/assets/index-CqSkXBSu.css +0 -1
- package/dist/dashboard/assets/index-flRpQtDj.js +0 -476
- package/dist/dashboard/assets/min-BIL7YgTN.js +0 -1
- package/dist/dashboard/assets/stateDiagram-v2-QKLJ7IA2-FdAb9MBo.js +0 -1
- package/dist/skill-packs/video-watching/scripts/gap_fill.py +0 -45
- package/skill-packs/video-watching/scripts/gap_fill.py +0 -45
|
@@ -32,14 +32,90 @@ Write a spec JSON, then run it. The script prints `elements/images/texts` counts
|
|
|
32
32
|
|
|
33
33
|
### JS API (for pipelines that generate many boards)
|
|
34
34
|
```js
|
|
35
|
-
const
|
|
36
|
-
|
|
35
|
+
const path = require('path');
|
|
36
|
+
// skill lives at <project>/.claude/skills/excalidraw/ — adjust leading ../ count to match your script's depth from project root
|
|
37
|
+
const { buildExcalidraw, lane, grid } = require(path.resolve(__dirname, '../.claude/skills/excalidraw/scripts/build_excalidraw.js'));
|
|
38
|
+
buildExcalidraw({ out: path.resolve(__dirname, '../boards/Board.excalidraw.md'), elements: [ ...lane({ title, images, x, y, thumbW }) ] });
|
|
37
39
|
```
|
|
38
40
|
|
|
41
|
+
## File layout
|
|
42
|
+
|
|
43
|
+
### Single board (default)
|
|
44
|
+
Keep the spec next to the generated board but clearly separated:
|
|
45
|
+
```
|
|
46
|
+
boards/
|
|
47
|
+
├── MyBoard.excalidraw.md ← generated deliverable; do not hand-edit
|
|
48
|
+
└── _spec/
|
|
49
|
+
└── MyBoard.json ← source of truth; edit this, then regenerate
|
|
50
|
+
```
|
|
51
|
+
The `.excalidraw.md` is **disposable** — it is fully derived from the spec. If the two ever
|
|
52
|
+
disagree, the spec wins. Commit both (the board for Obsidian/GitHub preview, the spec for
|
|
53
|
+
reproducibility), but only edit the spec.
|
|
54
|
+
|
|
55
|
+
### Multi-board pipeline
|
|
56
|
+
When a single generator produces several boards, isolate it in a `pipeline/` folder so the
|
|
57
|
+
deliverable boards stay at the top of the project and are easy to open in Obsidian:
|
|
58
|
+
```
|
|
59
|
+
boards/
|
|
60
|
+
├── Overview.excalidraw.md ← generated
|
|
61
|
+
├── Funnel.excalidraw.md ← generated
|
|
62
|
+
├── Pricing.excalidraw.md ← generated
|
|
63
|
+
└── pipeline/
|
|
64
|
+
├── generate.js ← single regen entrypoint: `node pipeline/generate.js`
|
|
65
|
+
├── shared-style.js ← shared palette / helpers
|
|
66
|
+
└── spec/
|
|
67
|
+
├── Overview.json ← source spec for Overview board
|
|
68
|
+
├── Funnel.json ← source spec for Funnel board
|
|
69
|
+
└── Pricing.json ← source spec for Pricing board
|
|
70
|
+
```
|
|
71
|
+
- **Generated files** (`*.excalidraw.md`) live one level above `pipeline/` — open them in Obsidian without navigating into a sub-folder.
|
|
72
|
+
- **Source specs** live in `pipeline/spec/` — one JSON per board.
|
|
73
|
+
- **Single entrypoint**: `node pipeline/generate.js` rebuilds every board. No per-board manual commands.
|
|
74
|
+
|
|
75
|
+
### Many boards from shared data (recipe)
|
|
76
|
+
Use this pattern when multiple boards pull from the same data set (e.g. one board per product, per region, or per funnel step):
|
|
77
|
+
|
|
78
|
+
```js
|
|
79
|
+
// pipeline/generate.js (lives at boards/pipeline/generate.js)
|
|
80
|
+
const path = require('path');
|
|
81
|
+
const ROOT = path.resolve(__dirname, '..'); // boards/ directory
|
|
82
|
+
// ../../ = project root (boards/ → project/); adjust if boards/ is nested deeper
|
|
83
|
+
const { buildExcalidraw, lane } = require(path.resolve(__dirname, '../../.claude/skills/excalidraw/scripts/build_excalidraw.js'));
|
|
84
|
+
const style = require(path.resolve(__dirname, 'shared-style.js'));
|
|
85
|
+
const items = require(path.resolve(__dirname, 'spec/items.json')); // shared data
|
|
86
|
+
|
|
87
|
+
for (const item of items) {
|
|
88
|
+
const elements = style.buildItemBoard(item); // per-item spec logic
|
|
89
|
+
buildExcalidraw({
|
|
90
|
+
out: path.resolve(ROOT, `${item.slug}.excalidraw.md`),
|
|
91
|
+
elements,
|
|
92
|
+
});
|
|
93
|
+
console.log('wrote', item.slug);
|
|
94
|
+
}
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
```js
|
|
98
|
+
// pipeline/shared-style.js (lives at boards/pipeline/shared-style.js)
|
|
99
|
+
const path = require('path');
|
|
100
|
+
// ../../ = project root (boards/ → project/); adjust if boards/ is nested deeper
|
|
101
|
+
const { card, connector, sectionTitle } = require(path.resolve(__dirname, '../../.claude/skills/excalidraw/scripts/lib/style.js'));
|
|
102
|
+
|
|
103
|
+
exports.buildItemBoard = (item) => [
|
|
104
|
+
sectionTitle({ x: 0, y: 0, text: item.name, fontSize: 40 }),
|
|
105
|
+
// … common layout using item fields
|
|
106
|
+
];
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Key conventions:
|
|
110
|
+
- `ROOT = path.resolve(__dirname, '..')` pins paths relative to the generator file, not the working directory. The generator works correctly wherever it is invoked from.
|
|
111
|
+
- Each item produces exactly one board; the mapping is `items.json → <slug>.excalidraw.md`.
|
|
112
|
+
- `shared-style.js` owns the layout logic — boards stay visually consistent; change the style once, regenerate all.
|
|
113
|
+
- Add a `package.json` script or `Makefile` alias so the command is always `npm run boards` (or similar) and never has to be rediscovered.
|
|
114
|
+
|
|
39
115
|
## Spec schema
|
|
40
116
|
```jsonc
|
|
41
117
|
{
|
|
42
|
-
"out": "/
|
|
118
|
+
"out": "./boards/Board.excalidraw.md", // prefer __dirname-relative in JS generators; relative to cwd for CLI
|
|
43
119
|
"vaultRoot": "/abs/vault", // optional; auto-detected by walking up to `.obsidian`
|
|
44
120
|
"attachDir": "Attachments", // external images get copied here (relative to board dir)
|
|
45
121
|
"wikilinkMode": "basename", // "basename" (default) or "path" (vault-relative)
|
|
@@ -48,6 +124,9 @@ buildExcalidraw({ out, elements: [ ...lane({ title, images, x, y, thumbW }) ] })
|
|
|
48
124
|
}
|
|
49
125
|
```
|
|
50
126
|
|
|
127
|
+
**Paths in generators**: always use `path.resolve(__dirname, ...)` for `out` and image `path` fields — never
|
|
128
|
+
hardcode absolute paths. This keeps the generator portable: move the folder and it still runs.
|
|
129
|
+
|
|
51
130
|
### Element types
|
|
52
131
|
- `text` — `{ x, y, text, fontSize?, color?, width?, align?, fontFamily? }` (fontFamily 1=hand, 2=normal, 3=code). **Set `width` for any caption/label that must stay inside a column or card** → the text WRAPS to that width (autoResize off) and its height is computed from the wrapped line count. Omit `width` only for short single-line text you want sized to content (it renders on one line and will overlap neighbours if long).
|
|
53
132
|
- `image` — `{ x, y, path, width? , height? }` — give ONE of width/height; the other is derived from aspect. `path` is an absolute file path.
|
|
@@ -39,15 +39,36 @@ if it's installed.
|
|
|
39
39
|
# language auto-detects — no need to pass it. Override only if auto mislabels a
|
|
40
40
|
# short/ambiguous clip: ./transcribe.sh "/abs/path/clip.mp4" tr
|
|
41
41
|
```
|
|
42
|
+
|
|
43
|
+
**Pick the mode for the kind of video — this matters most for app/UI recordings:**
|
|
44
|
+
- **Talking-head / lecture / ad creative** → the default is right. Scene-detect + a
|
|
45
|
+
10s gap-fill catches the visuals.
|
|
46
|
+
- **App screen-recording / onboarding funnel / UI walkthrough** → add `--mode ui`.
|
|
47
|
+
App screens linger 3–5s and change by **text only** (a questionnaire step, a
|
|
48
|
+
paywall) — they don't move enough to trip scene-detect, so the 10s default
|
|
49
|
+
silently drops most of them. `--mode ui` samples every ~2.5s so each screen lands.
|
|
50
|
+
If you only need the screens (no narration), add `--frames-only` to skip whisper:
|
|
51
|
+
```bash
|
|
52
|
+
./scripts/transcribe.sh "/abs/path/onboarding.mp4" --mode ui --frames-only --contact-sheet
|
|
53
|
+
```
|
|
54
|
+
|
|
42
55
|
This writes everything into `<video_dir>/<slug>.media/`:
|
|
43
|
-
- `transcript.srt` / `.json` / `.txt` — timestamped transcript (large-v3-turbo)
|
|
56
|
+
- `transcript.srt` / `.json` / `.txt` — timestamped transcript (large-v3-turbo) — *skipped with `--frames-only`*
|
|
44
57
|
- `frames/anchor_first.jpg` / `anchor_last.jpg` — first + last frame, always captured
|
|
45
58
|
(short cut-heavy creatives carry the hook and CTA here; scene-detect misses both)
|
|
46
|
-
- `frames/
|
|
47
|
-
|
|
48
|
-
|
|
59
|
+
- `frames/frame_*.jpg` — frames selected in **one time-based pass**: a scene change
|
|
60
|
+
fired (slide/UI transition) **or** `MAX_GAP` seconds elapsed since the last frame,
|
|
61
|
+
whichever comes first. The gap rule is wall-clock based (`prev_selected_t`), so it
|
|
62
|
+
works on variable-frame-rate screen recordings where frame-number sampling breaks,
|
|
63
|
+
and it guarantees every static stretch (app demos, slides, CTAs) gets a frame.
|
|
64
|
+
- `frames/contact_sheet.jpg` — tiled montage of all frames, *only with `--contact-sheet`*
|
|
49
65
|
- `frames.json` — **the index you read**: `[{file, t, at, type}]`, sorted by time,
|
|
50
|
-
near-duplicate timestamps collapsed (anchors always kept)
|
|
66
|
+
near-duplicate timestamps collapsed (anchors always kept). `type` is `scene` (a
|
|
67
|
+
picture change fired), `gap` (a periodic fill at the `MAX_GAP` cadence), or `anchor`.
|
|
68
|
+
|
|
69
|
+
The engine prints a **coverage check** at the end: `longest unsampled gap = Xs`. If it
|
|
70
|
+
warns the gap is >2× `MAX_GAP`, frames are likely missing — re-run denser (`--max-gap`
|
|
71
|
+
lower, or `--mode ui`) **before** any expensive deep-analysis pass.
|
|
51
72
|
|
|
52
73
|
`audio.wav` is auto-deleted after transcription (it's a ~1.9MB/min whisper-only
|
|
53
74
|
intermediate). The whole `*.media/` dir is gitignored — it stays next to the video
|
|
@@ -69,7 +90,9 @@ nothing on screen need no frame.
|
|
|
69
90
|
> Heuristic: the two `anchor` frames (first/last) almost always matter — the hook
|
|
70
91
|
> and the CTA. Every `scene` frame is a candidate (the picture changed for a
|
|
71
92
|
> reason). `gap` frames cover static stretches scene-detect skipped — often the
|
|
72
|
-
> most informative part (an app demo or
|
|
93
|
+
> most informative part (an app demo or onboarding screen that doesn't "move"), so
|
|
94
|
+
> check them. In `--mode ui` runs most frames are `gap` — that's expected and you
|
|
95
|
+
> generally want to look at all of them, one per screen.
|
|
73
96
|
|
|
74
97
|
**Need a frame the index doesn't have? Grab it on demand.** Scene-detect fires on
|
|
75
98
|
motion, not on meaning — on fast-cut video the most informative moment often sits
|
|
@@ -116,6 +139,14 @@ relevant dreamcontext skill (per the skill-triage rule) and load it:
|
|
|
116
139
|
- **Knowledge / training** → if it should persist for the project, hand the transcript to `dreamcontext knowledge` so it becomes durable context
|
|
117
140
|
Don't guess the use-case — let the user direct it.
|
|
118
141
|
|
|
142
|
+
> **UI teardown (no transcript needed).** When the goal is purely the app flow —
|
|
143
|
+
> e.g. tearing down a competitor's onboarding funnel — run `--frames-only --mode ui`
|
|
144
|
+
> and skip the `.transcript.md` entirely. The deliverable becomes a **curated screen
|
|
145
|
+
> list** (one entry per onboarding step, in order, from `frames.json`), which feeds a
|
|
146
|
+
> board (`excalidraw`) or an `onboarding-design` analysis directly. Use `--contact-sheet`
|
|
147
|
+
> to eyeball coverage first, and trust the coverage warning — under-sampling here is
|
|
148
|
+
> exactly what makes a teardown wrongly report "screen X wasn't shown".
|
|
149
|
+
|
|
119
150
|
## What this skill is NOT
|
|
120
151
|
- Not a knowledge-base writer or ingestion pipeline. This skill produces the
|
|
121
152
|
transcript artifact; persisting it (chunking, embedding, ingest into a project's
|
|
@@ -140,12 +171,22 @@ brew install whisper-cpp ffmpeg # yt-dlp too, only for remote links
|
|
|
140
171
|
├── SKILL.md ← you are here
|
|
141
172
|
└── scripts/
|
|
142
173
|
├── transcribe.sh ← video → transcript + frames + frames.json (the engine)
|
|
143
|
-
|
|
144
|
-
└── build_frame_index.py ← frames/*.jpg → frames.json (pts_time index)
|
|
174
|
+
└── build_frame_index.py ← frames/*.jpg → frames.json (pts_time index + coverage check)
|
|
145
175
|
```
|
|
146
176
|
|
|
147
|
-
##
|
|
148
|
-
|
|
149
|
-
-
|
|
150
|
-
|
|
151
|
-
-
|
|
177
|
+
## Flags & tuning
|
|
178
|
+
Flags (each also settable as an env var, e.g. `MODE=ui`):
|
|
179
|
+
- `--mode ui|lecture` — sampling preset. `ui` = `MAX_GAP` 2.5s (app screens); `lecture`
|
|
180
|
+
(default) = 10s (talking-head). Explicit `--max-gap`/`--scene-threshold` always win.
|
|
181
|
+
- `--frames-only` — skip whisper; extract frames only (UI/UX teardowns).
|
|
182
|
+
- `--max-gap N` — guarantee a frame at least every N seconds.
|
|
183
|
+
- `--scene-threshold N` — scene-change sensitivity, lower = more frames.
|
|
184
|
+
- `--contact-sheet` — also emit `frames/contact_sheet.jpg` (coverage at a glance).
|
|
185
|
+
- `--lang CODE` — force the transcript language (default: auto-detect).
|
|
186
|
+
|
|
187
|
+
Common adjustments:
|
|
188
|
+
- Onboarding/UI recording under-sampled? `--mode ui` (or push further: `--max-gap 1.5`).
|
|
189
|
+
- Too few frames on a slide-heavy talk? `--scene-threshold 0.1`.
|
|
190
|
+
- Long static lecture over-sampled? `--max-gap 20`.
|
|
191
|
+
- Capturing too many near-identical frames? `DEDUPE_SEC=1.0 ./transcribe.sh …`.
|
|
192
|
+
- Force a model: `WHISPER_MODEL=/abs/ggml-large-v3.bin ./transcribe.sh …`.
|
|
@@ -1,14 +1,22 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
"""Build frames.json: map each extracted frame to its pts_time on the video timeline.
|
|
3
3
|
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
zero-padded frame file in selection order, adds the first/last anchor frames, drops
|
|
7
|
-
near-duplicate timestamps,
|
|
4
|
+
transcribe.sh selects frames in ONE time-based pass (scene change OR every MAX_GAP),
|
|
5
|
+
dumping each kept frame's pts_time to frame_times.txt. This pairs every timestamp with
|
|
6
|
+
its zero-padded frame file in selection order, adds the first/last anchor frames, drops
|
|
7
|
+
near-duplicate timestamps, labels each frame (scene vs gap vs anchor), and emits
|
|
8
|
+
frames.json sorted by time. It also prints a COVERAGE check — the largest unsampled
|
|
9
|
+
stretch — so a human can catch under-sampling before an expensive deep-analysis pass.
|
|
10
|
+
|
|
11
|
+
Type labels are reconstructed from inter-frame spacing (no second decode): a frame that
|
|
12
|
+
landed sooner than MAX_GAP after the previous one was triggered by a scene change
|
|
13
|
+
('scene'); one that landed at the MAX_GAP cadence is a periodic fill ('gap'). They are
|
|
14
|
+
hints for which frames to prioritize, not a hard contract.
|
|
8
15
|
|
|
9
16
|
Stdlib only — no venv needed.
|
|
10
17
|
Usage: build_frame_index.py <OUT_DIR> [DURATION_SEC]
|
|
11
18
|
Env: DEDUPE_SEC=0.4 collapse non-anchor frames closer than this to the kept one.
|
|
19
|
+
MAX_GAP=10 the gap target the engine sampled at (drives type + coverage warn).
|
|
12
20
|
"""
|
|
13
21
|
import json
|
|
14
22
|
import os
|
|
@@ -19,6 +27,7 @@ from pathlib import Path
|
|
|
19
27
|
# metadata=print header line, e.g.: "frame:0 pts:321024 pts_time:12.852500"
|
|
20
28
|
FRAME_RE = re.compile(r"frame:(\d+)\b.*?pts_time:([\d.]+)", re.DOTALL)
|
|
21
29
|
DEDUPE_SEC = float(os.environ.get("DEDUPE_SEC", "0.4"))
|
|
30
|
+
MAX_GAP = float(os.environ.get("MAX_GAP", "10"))
|
|
22
31
|
|
|
23
32
|
|
|
24
33
|
def parse_times(meta_file: Path) -> dict[int, float]:
|
|
@@ -29,11 +38,11 @@ def parse_times(meta_file: Path) -> dict[int, float]:
|
|
|
29
38
|
return {int(n): round(float(t), 2) for n, t in FRAME_RE.findall(text)}
|
|
30
39
|
|
|
31
40
|
|
|
32
|
-
def collect(out_dir: Path, prefix: str, meta_name: str
|
|
41
|
+
def collect(out_dir: Path, prefix: str, meta_name: str) -> list[dict]:
|
|
33
42
|
times = parse_times(out_dir / "frames" / meta_name)
|
|
34
43
|
frames = sorted((out_dir / "frames").glob(f"{prefix}_*.jpg"))
|
|
35
44
|
# transcribe.sh names files 1-based (%04d starts at 0001); metadata frame: is 0-based.
|
|
36
|
-
return [{"file": f"frames/{f.name}", "t": times.get(i), "type":
|
|
45
|
+
return [{"file": f"frames/{f.name}", "t": times.get(i), "type": "key"}
|
|
37
46
|
for i, f in enumerate(frames)]
|
|
38
47
|
|
|
39
48
|
|
|
@@ -69,6 +78,40 @@ def dedupe(rows: list[dict]) -> list[dict]:
|
|
|
69
78
|
return kept
|
|
70
79
|
|
|
71
80
|
|
|
81
|
+
def label_types(rows: list[dict]) -> None:
|
|
82
|
+
"""Reconstruct scene/gap from spacing. Frames spaced >= ~MAX_GAP are periodic
|
|
83
|
+
fills ('gap'); closer ones were pulled in early by a scene change ('scene')."""
|
|
84
|
+
near = MAX_GAP * 0.9
|
|
85
|
+
prev_t = 0.0
|
|
86
|
+
for r in rows:
|
|
87
|
+
if r["type"] == "anchor":
|
|
88
|
+
if r["t"] is not None:
|
|
89
|
+
prev_t = r["t"]
|
|
90
|
+
continue
|
|
91
|
+
if r["t"] is None:
|
|
92
|
+
r["type"] = "scene"
|
|
93
|
+
continue
|
|
94
|
+
r["type"] = "gap" if (r["t"] - prev_t) >= near else "scene"
|
|
95
|
+
prev_t = r["t"]
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def coverage(rows: list[dict], duration: float) -> tuple[float, float]:
|
|
99
|
+
"""Largest unsampled stretch (s) and where it starts, over [0, duration]."""
|
|
100
|
+
ts = sorted(r["t"] for r in rows if r["t"] is not None)
|
|
101
|
+
if not ts:
|
|
102
|
+
return (duration, 0.0)
|
|
103
|
+
# Bound the right edge with the real duration; if ffprobe couldn't read it (0), fall
|
|
104
|
+
# back to the last sampled time so the tail gap (last frame -> end) still surfaces
|
|
105
|
+
# rather than being silently dropped.
|
|
106
|
+
right = round(duration, 2) if duration > 0 else ts[-1]
|
|
107
|
+
bounds = [0.0] + ts + [right]
|
|
108
|
+
worst, at = 0.0, 0.0
|
|
109
|
+
for a, b in zip(bounds, bounds[1:]):
|
|
110
|
+
if b - a > worst:
|
|
111
|
+
worst, at = b - a, a
|
|
112
|
+
return (worst, at)
|
|
113
|
+
|
|
114
|
+
|
|
72
115
|
def main() -> int:
|
|
73
116
|
if len(sys.argv) < 2:
|
|
74
117
|
print("usage: build_frame_index.py <OUT_DIR> [DURATION_SEC]", file=sys.stderr)
|
|
@@ -76,18 +119,26 @@ def main() -> int:
|
|
|
76
119
|
out_dir = Path(sys.argv[1])
|
|
77
120
|
duration = float(sys.argv[2]) if len(sys.argv) > 2 else 0.0
|
|
78
121
|
|
|
79
|
-
rows = collect(out_dir, "
|
|
80
|
-
rows += collect(out_dir, "gap", "gap_times.txt", "gap")
|
|
122
|
+
rows = collect(out_dir, "frame", "frame_times.txt")
|
|
81
123
|
rows += anchors(out_dir, duration)
|
|
82
|
-
# Sort by time
|
|
83
|
-
|
|
124
|
+
# Sort by time (unknown timestamps sink to the end); on a tie, anchors sort FIRST so
|
|
125
|
+
# the eq(n,0) seed frame at t=0 (and any selected frame coincident with anchor_last)
|
|
126
|
+
# dedupes away against the identical anchor instead of producing a duplicate entry.
|
|
127
|
+
rows.sort(key=lambda r: (r["t"] is None, r["t"] or 0.0, r["type"] != "anchor"))
|
|
84
128
|
rows = dedupe(rows)
|
|
129
|
+
label_types(rows)
|
|
85
130
|
for r in rows:
|
|
86
131
|
r["at"] = fmt(r["t"])
|
|
87
132
|
|
|
88
133
|
(out_dir / "frames.json").write_text(json.dumps(rows, ensure_ascii=False, indent=2))
|
|
134
|
+
|
|
89
135
|
n_anchor = sum(r["type"] == "anchor" for r in rows)
|
|
136
|
+
worst, at = coverage(rows, duration)
|
|
90
137
|
print(f" frames.json: {len(rows)} frames indexed ({n_anchor} anchors, dedupe<{DEDUPE_SEC}s)")
|
|
138
|
+
print(f" coverage: longest unsampled gap = {worst:.1f}s at {fmt(at)} (target MAX_GAP={MAX_GAP:g}s)")
|
|
139
|
+
if duration > 0 and worst > MAX_GAP * 2:
|
|
140
|
+
print(f" ⚠ coverage gap is >2x MAX_GAP — frames may be missing around {fmt(at)}. "
|
|
141
|
+
f"Re-run denser: --max-gap {MAX_GAP / 2:g} (or --mode ui for app recordings).")
|
|
91
142
|
return 0
|
|
92
143
|
|
|
93
144
|
|
|
@@ -6,30 +6,92 @@
|
|
|
6
6
|
# video. Claude reads the outputs (transcript + frames.json), decides which frames
|
|
7
7
|
# matter, views them, and writes the curated <video>.transcript.md (see SKILL.md).
|
|
8
8
|
#
|
|
9
|
-
# Usage: ./transcribe.sh "/abs/path/to/video.mov" [lang]
|
|
9
|
+
# Usage: ./transcribe.sh "/abs/path/to/video.mov" [lang] [flags]
|
|
10
10
|
# lang: whisper language code. DEFAULT 'auto' — whisper detects it, no need to pass.
|
|
11
11
|
# Force only if auto mislabels a short/ambiguous clip (e.g. 'tr', 'en').
|
|
12
12
|
#
|
|
13
|
+
# Flags (also settable as env vars):
|
|
14
|
+
# --frames-only skip whisper entirely; extract frames only. For pure
|
|
15
|
+
# UI/UX teardowns where no transcript is needed. (FRAMES_ONLY=1)
|
|
16
|
+
# --mode ui|lecture sampling preset. 'ui' = MAX_GAP 2.5s (app screens linger
|
|
17
|
+
# 3-5s and change by text only, below scene-detect). 'lecture'
|
|
18
|
+
# (default) = MAX_GAP 10s for talking-head video. (MODE=ui)
|
|
19
|
+
# --max-gap N guarantee a frame at least every N seconds. Overrides the
|
|
20
|
+
# mode preset. (MAX_GAP=N)
|
|
21
|
+
# --scene-threshold N ffmpeg scene-change sensitivity, lower = more frames. (SCENE_THRESHOLD=N)
|
|
22
|
+
# --contact-sheet also emit frames/contact_sheet.jpg — a tiled montage of every
|
|
23
|
+
# frame, so coverage is verifiable in one glance. (CONTACT_SHEET=1)
|
|
24
|
+
# --lang CODE same as the positional lang arg.
|
|
25
|
+
#
|
|
13
26
|
# Env overrides:
|
|
14
27
|
# WHISPER_MODEL=/abs/path/ggml-*.bin force a specific model
|
|
15
28
|
# OUT_DIR=/abs/path where artifacts go (default: <video_dir>/<slug>.media)
|
|
16
|
-
#
|
|
17
|
-
#
|
|
29
|
+
# NOTE: avoid ':' in OUT_DIR — it is special inside the
|
|
30
|
+
# ffmpeg filter graph and would truncate the metadata path.
|
|
31
|
+
# DEDUPE_SEC=0.4 collapse non-anchor frames closer than this
|
|
18
32
|
#
|
|
19
33
|
# Outputs (in OUT_DIR):
|
|
20
|
-
# transcript.srt timestamped segments (human-friendly)
|
|
21
|
-
# transcript.json/.txt/.vtt
|
|
22
|
-
# frames/
|
|
23
|
-
# frames/gap_*.jpg fill frames where scene-detect left a stretch > MAX_GAP unsampled
|
|
34
|
+
# transcript.srt timestamped segments (human-friendly) [skipped with --frames-only]
|
|
35
|
+
# transcript.json/.txt/.vtt [skipped with --frames-only]
|
|
36
|
+
# frames/frame_*.jpg selected frames (scene change OR every MAX_GAP, whichever first)
|
|
24
37
|
# frames/anchor_first.jpg / anchor_last.jpg always-captured first + last frame
|
|
38
|
+
# frames/contact_sheet.jpg tiled montage of all frames [--contact-sheet only]
|
|
25
39
|
# frames.json [{file, t, at, type}] — each frame's pts_time, sorted (Claude reads this)
|
|
26
40
|
# source-meta.txt ffprobe dump (audio.wav is created then deleted after transcription)
|
|
27
41
|
set -euo pipefail
|
|
28
42
|
|
|
29
|
-
|
|
30
|
-
|
|
43
|
+
# --- arg + flag parsing ---------------------------------------------------------
|
|
44
|
+
# Positional: first non-flag = VIDEO, second non-flag = LANG_CODE (back-compatible).
|
|
45
|
+
VIDEO=""
|
|
46
|
+
LANG_CODE=""
|
|
47
|
+
FRAMES_ONLY="${FRAMES_ONLY:-0}"
|
|
48
|
+
MODE="${MODE:-lecture}"
|
|
49
|
+
CONTACT_SHEET="${CONTACT_SHEET:-0}"
|
|
50
|
+
# MAX_GAP / SCENE_THRESHOLD: track whether explicitly set so a CLI/env value beats the
|
|
51
|
+
# mode preset. Empty here means "fall back to the mode preset" computed below.
|
|
52
|
+
MAX_GAP="${MAX_GAP:-}"
|
|
53
|
+
SCENE_THRESHOLD="${SCENE_THRESHOLD:-}"
|
|
54
|
+
|
|
55
|
+
while [ $# -gt 0 ]; do
|
|
56
|
+
case "$1" in
|
|
57
|
+
--frames-only) FRAMES_ONLY=1 ;;
|
|
58
|
+
--contact-sheet) CONTACT_SHEET=1 ;;
|
|
59
|
+
--mode) MODE="${2:?--mode needs a value}"; shift ;;
|
|
60
|
+
--mode=*) MODE="${1#*=}" ;;
|
|
61
|
+
--max-gap) MAX_GAP="${2:?--max-gap needs a value}"; shift ;;
|
|
62
|
+
--max-gap=*) MAX_GAP="${1#*=}" ;;
|
|
63
|
+
--scene-threshold) SCENE_THRESHOLD="${2:?--scene-threshold needs a value}"; shift ;;
|
|
64
|
+
--scene-threshold=*) SCENE_THRESHOLD="${1#*=}" ;;
|
|
65
|
+
--lang) LANG_CODE="${2:?--lang needs a value}"; shift ;;
|
|
66
|
+
--lang=*) LANG_CODE="${1#*=}" ;;
|
|
67
|
+
--*) echo "unknown flag: $1" >&2; exit 2 ;;
|
|
68
|
+
*) if [ -z "$VIDEO" ]; then VIDEO="$1"
|
|
69
|
+
elif [ -z "$LANG_CODE" ]; then LANG_CODE="$1"
|
|
70
|
+
else echo "unexpected arg: $1" >&2; exit 2; fi ;;
|
|
71
|
+
esac
|
|
72
|
+
shift
|
|
73
|
+
done
|
|
74
|
+
|
|
75
|
+
[ -n "$VIDEO" ] || { echo "Usage: transcribe.sh <video> [lang] [--frames-only] [--mode ui] [--max-gap N]" >&2; exit 1; }
|
|
76
|
+
LANG_CODE="${LANG_CODE:-auto}" # whisper auto-detects the spoken language unless overridden
|
|
77
|
+
|
|
78
|
+
# Mode presets fill in only what the caller left unset (explicit CLI/env always wins).
|
|
79
|
+
case "$MODE" in
|
|
80
|
+
ui) MODE_MAX_GAP=2.5 ;;
|
|
81
|
+
lecture) MODE_MAX_GAP=10 ;;
|
|
82
|
+
*) echo "unknown --mode '$MODE' (use ui|lecture)" >&2; exit 2 ;;
|
|
83
|
+
esac
|
|
84
|
+
MAX_GAP="${MAX_GAP:-$MODE_MAX_GAP}"
|
|
31
85
|
SCENE_THRESHOLD="${SCENE_THRESHOLD:-0.15}"
|
|
32
|
-
|
|
86
|
+
|
|
87
|
+
# Guard the two numeric knobs before they reach the ffmpeg filter / Python: a non-number
|
|
88
|
+
# would crash build_frame_index (float()), and MAX_GAP=0 makes gte(t-prev,0) select EVERY
|
|
89
|
+
# frame (disk-fill footgun). Fail loudly instead.
|
|
90
|
+
is_num='^[0-9]+([.][0-9]+)?$'
|
|
91
|
+
[[ "$MAX_GAP" =~ $is_num ]] || { echo "--max-gap must be a number (got '$MAX_GAP')" >&2; exit 2; }
|
|
92
|
+
awk "BEGIN{exit !($MAX_GAP > 0)}" || { echo "--max-gap must be > 0 (got '$MAX_GAP')" >&2; exit 2; }
|
|
93
|
+
[[ "$SCENE_THRESHOLD" =~ $is_num ]] || { echo "--scene-threshold must be a number (got '$SCENE_THRESHOLD')" >&2; exit 2; }
|
|
94
|
+
|
|
33
95
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
34
96
|
|
|
35
97
|
# --- remote link? download with yt-dlp (local files skip this) ------------------
|
|
@@ -43,7 +105,10 @@ if [[ "$VIDEO" =~ ^https?:// ]]; then
|
|
|
43
105
|
fi
|
|
44
106
|
[ -f "$VIDEO" ] || { echo "video not found: $VIDEO" >&2; exit 1; }
|
|
45
107
|
|
|
108
|
+
command -v ffmpeg >/dev/null || { echo "ffmpeg not found (brew install ffmpeg)" >&2; exit 1; }
|
|
109
|
+
|
|
46
110
|
# --- model: prefer turbo, then large-v3, then medium (override with WHISPER_MODEL)
|
|
111
|
+
# Only needed for transcription — skip the whole resolution when --frames-only.
|
|
47
112
|
pick_model() {
|
|
48
113
|
if [ -n "${WHISPER_MODEL:-}" ]; then printf '%s' "$WHISPER_MODEL"; return; fi
|
|
49
114
|
local p
|
|
@@ -57,18 +122,19 @@ pick_model() {
|
|
|
57
122
|
done
|
|
58
123
|
printf ''
|
|
59
124
|
}
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
125
|
+
if [ "$FRAMES_ONLY" != "1" ]; then
|
|
126
|
+
MODEL="$(pick_model)"
|
|
127
|
+
[ -n "$MODEL" ] && [ -f "$MODEL" ] || {
|
|
128
|
+
echo "no whisper model found. Install one, e.g.:" >&2
|
|
129
|
+
echo " curl -L -o ~/.cache/whisper.cpp/models/ggml-large-v3-turbo.bin \\" >&2
|
|
130
|
+
echo " https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin" >&2
|
|
131
|
+
echo " or set WHISPER_MODEL=/abs/path/ggml-*.bin" >&2
|
|
132
|
+
echo " (or pass --frames-only to skip transcription entirely)" >&2
|
|
133
|
+
exit 1
|
|
134
|
+
}
|
|
135
|
+
WHISPER_BIN="$(command -v whisper-cli || command -v whisper-cpp || command -v main || true)"
|
|
136
|
+
[ -n "$WHISPER_BIN" ] || { echo "whisper.cpp binary not found (brew install whisper-cpp). Or pass --frames-only." >&2; exit 1; }
|
|
137
|
+
fi
|
|
72
138
|
|
|
73
139
|
# --- where artifacts go: next to the video by default ---------------------------
|
|
74
140
|
SRC_DIR="$(cd "$(dirname "$VIDEO")" && pwd)"
|
|
@@ -79,28 +145,51 @@ mkdir -p "$OUT/frames"
|
|
|
79
145
|
# even-dim scaling keeps the mjpeg encoder happy on odd-width sources
|
|
80
146
|
SCALE="scale=trunc(iw/2)*2:trunc(ih/2)*2"
|
|
81
147
|
|
|
82
|
-
echo "==> [$SLUG] probing"
|
|
148
|
+
echo "==> [$SLUG] probing (mode=$MODE, max_gap=${MAX_GAP}s, scene>$SCENE_THRESHOLD, frames_only=$FRAMES_ONLY)"
|
|
83
149
|
ffprobe -v error -show_entries format=duration,size:stream=codec_type,codec_name,width,height \
|
|
84
150
|
-of default=noprint_wrappers=1 "$VIDEO" | tee "$OUT/source-meta.txt"
|
|
85
151
|
DURATION="$(ffprobe -v error -show_entries format=duration -of csv=p=0 "$VIDEO" 2>/dev/null | head -1)"
|
|
86
152
|
DURATION="${DURATION:-0}"
|
|
87
153
|
|
|
88
|
-
|
|
89
|
-
|
|
154
|
+
if [ "$FRAMES_ONLY" = "1" ]; then
|
|
155
|
+
echo "==> [$SLUG] --frames-only: skipping audio + transcription"
|
|
156
|
+
else
|
|
157
|
+
echo "==> [$SLUG] extracting 16kHz mono audio"
|
|
158
|
+
ffmpeg -y -loglevel error -i "$VIDEO" -ar 16000 -ac 1 -c:a pcm_s16le "$OUT/audio.wav"
|
|
90
159
|
|
|
91
|
-
echo "==> [$SLUG] transcribing with $(basename "$MODEL") (lang=$LANG_CODE)"
|
|
92
|
-
"$WHISPER_BIN" -m "$MODEL" -f "$OUT/audio.wav" -l "$LANG_CODE" \
|
|
93
|
-
|
|
94
|
-
# audio.wav is a pure whisper intermediate (~1.9MB/min) — drop it once the transcript exists.
|
|
95
|
-
rm -f "$OUT/audio.wav"
|
|
160
|
+
echo "==> [$SLUG] transcribing with $(basename "$MODEL") (lang=$LANG_CODE)"
|
|
161
|
+
"$WHISPER_BIN" -m "$MODEL" -f "$OUT/audio.wav" -l "$LANG_CODE" \
|
|
162
|
+
--output-txt --output-srt --output-vtt --output-json -of "$OUT/transcript" -pp
|
|
163
|
+
# audio.wav is a pure whisper intermediate (~1.9MB/min) — drop it once the transcript exists.
|
|
164
|
+
rm -f "$OUT/audio.wav"
|
|
165
|
+
fi
|
|
96
166
|
|
|
97
|
-
echo "==> [$SLUG] extracting
|
|
167
|
+
echo "==> [$SLUG] extracting frames (scene change OR every ${MAX_GAP}s, whichever fires first)"
|
|
168
|
+
# ONE decode pass, time-based and frame-rate-independent:
|
|
169
|
+
# eq(n,0) always seed the first frame (also primes prev_selected_t)
|
|
170
|
+
# gt(scene,$SCENE_THRESHOLD) a scene change (slide/UI transition) fired
|
|
171
|
+
# gte(t-prev_selected_t,MAX_GAP) MAX_GAP elapsed since the last KEPT frame — guarantees
|
|
172
|
+
# coverage of static stretches scene-detect misses (app
|
|
173
|
+
# screens that change by text only). prev_selected_t is
|
|
174
|
+
# wall-clock seconds, so this is robust on variable-frame-rate
|
|
175
|
+
# screen recordings where frame-number sampling (mod(n,N)) breaks.
|
|
98
176
|
# pix_fmt yuvj420p avoids mjpeg "non full-range YUV" failures on screen recordings;
|
|
99
|
-
# even-dim scaling keeps the encoder happy on odd-width sources.
|
|
100
177
|
# metadata=print dumps each kept frame's pts_time so frames map back to the timeline.
|
|
101
178
|
ffmpeg -y -loglevel error -i "$VIDEO" \
|
|
102
|
-
-vf "select='gt(scene,$SCENE_THRESHOLD)',metadata=print:file=$OUT/frames/
|
|
103
|
-
-fps_mode vfr -pix_fmt yuvj420p -q:v 3 "$OUT/frames/
|
|
179
|
+
-vf "select='eq(n,0)+gt(scene,$SCENE_THRESHOLD)+gte(t-prev_selected_t,$MAX_GAP)',metadata=print:file=$OUT/frames/frame_times.txt,scale=trunc(iw/2)*2:trunc(ih/2)*2" \
|
|
180
|
+
-fps_mode vfr -pix_fmt yuvj420p -q:v 3 "$OUT/frames/frame_%04d.jpg" || true
|
|
181
|
+
|
|
182
|
+
# The || true above keeps a partial result usable, but zero frames means the decode
|
|
183
|
+
# produced nothing (no decodable video stream, unsupported codec, write failure). That
|
|
184
|
+
# must fail loudly — silently leaving an empty frames.json is the exact under-sampling
|
|
185
|
+
# trap issue #15 is about. Count safely (|| true so a no-match never trips set -e).
|
|
186
|
+
NSEL="$(find "$OUT/frames" -name 'frame_*.jpg' 2>/dev/null | wc -l | tr -d ' ')"
|
|
187
|
+
if [ "$NSEL" -eq 0 ]; then
|
|
188
|
+
echo "ERROR: ffmpeg extracted 0 frames from '$VIDEO'." >&2
|
|
189
|
+
echo " The file may have no decodable video stream (audio-only?), an unsupported codec," >&2
|
|
190
|
+
echo " or the output dir isn't writable. No frames.json written." >&2
|
|
191
|
+
exit 1
|
|
192
|
+
fi
|
|
104
193
|
|
|
105
194
|
echo "==> [$SLUG] capturing first + last anchor frames"
|
|
106
195
|
# Short, cut-heavy creatives carry the most information in the opening hook and the
|
|
@@ -110,27 +199,31 @@ ffmpeg -y -loglevel error -i "$VIDEO" -vf "$SCALE" -frames:v 1 -q:v 3 \
|
|
|
110
199
|
ffmpeg -y -loglevel error -sseof -0.5 -i "$VIDEO" -vf "$SCALE" -update 1 -frames:v 1 -q:v 3 \
|
|
111
200
|
"$OUT/frames/anchor_last.jpg" || true
|
|
112
201
|
|
|
113
|
-
echo "==> [$SLUG]
|
|
114
|
-
|
|
115
|
-
# (app demo, slide, talking head) gets no frame. gap_fill.py returns timestamps that
|
|
116
|
-
# keep every unsampled stretch <= MAX_GAP; we grab each and log its known pts_time.
|
|
117
|
-
: > "$OUT/frames/gap_times.txt"
|
|
118
|
-
gi=0
|
|
119
|
-
while IFS= read -r t; do
|
|
120
|
-
[ -z "$t" ] && continue
|
|
121
|
-
printf -v fn "gap_%04d.jpg" "$((gi + 1))"
|
|
122
|
-
if ffmpeg -y -loglevel error -ss "$t" -i "$VIDEO" -vf "$SCALE" -frames:v 1 -q:v 3 "$OUT/frames/$fn"; then
|
|
123
|
-
echo "frame:$gi pts_time:$t" >> "$OUT/frames/gap_times.txt"
|
|
124
|
-
gi=$((gi + 1))
|
|
125
|
-
fi
|
|
126
|
-
done < <(python3 "$SCRIPT_DIR/gap_fill.py" "$OUT" "$DURATION" "$MAX_GAP")
|
|
127
|
-
echo " gap frames added: $gi"
|
|
202
|
+
echo "==> [$SLUG] building frames.json (timestamp index, dedup near-dups, coverage check)"
|
|
203
|
+
MAX_GAP="$MAX_GAP" python3 "$SCRIPT_DIR/build_frame_index.py" "$OUT" "$DURATION"
|
|
128
204
|
|
|
129
|
-
|
|
130
|
-
|
|
205
|
+
# --- optional contact sheet: one glance to verify coverage before deep analysis --
|
|
206
|
+
if [ "$CONTACT_SHEET" = "1" ]; then
|
|
207
|
+
echo "==> [$SLUG] building contact sheet (coverage montage)"
|
|
208
|
+
NF="$(ls -1 "$OUT/frames"/frame_*.jpg 2>/dev/null | wc -l | tr -d ' ')"
|
|
209
|
+
if [ "$NF" -gt 0 ]; then
|
|
210
|
+
COLS="$(awk -v n="$NF" 'BEGIN{c=int(sqrt(n)); if(c*c<n)c++; print c}')"
|
|
211
|
+
ROWS="$(awk -v n="$NF" -v c="$COLS" 'BEGIN{r=int(n/c); if(r*c<n)r++; print r}')"
|
|
212
|
+
ffmpeg -y -loglevel error -pattern_type glob -i "$OUT/frames/frame_*.jpg" \
|
|
213
|
+
-vf "scale=240:-1,tile=${COLS}x${ROWS}:padding=4:margin=4" -frames:v 1 -q:v 4 \
|
|
214
|
+
"$OUT/frames/contact_sheet.jpg" \
|
|
215
|
+
&& echo " contact sheet: $OUT/frames/contact_sheet.jpg (${COLS}x${ROWS})" \
|
|
216
|
+
|| echo " contact sheet skipped (montage failed)"
|
|
217
|
+
fi
|
|
218
|
+
fi
|
|
131
219
|
|
|
132
|
-
|
|
220
|
+
# find -not -name keeps the count set -e-safe (grep -v exits 1 on no-match, tripping pipefail).
|
|
221
|
+
NFRAMES="$(find "$OUT/frames" -name '*.jpg' ! -name 'contact_sheet.jpg' 2>/dev/null | wc -l | tr -d ' ')"
|
|
133
222
|
echo "==> [$SLUG] DONE"
|
|
134
|
-
echo " transcript: $OUT/transcript.srt"
|
|
223
|
+
[ "$FRAMES_ONLY" = "1" ] || echo " transcript: $OUT/transcript.srt"
|
|
135
224
|
echo " frames: $NFRAMES (index: $OUT/frames.json)"
|
|
136
|
-
|
|
225
|
+
if [ "$FRAMES_ONLY" = "1" ]; then
|
|
226
|
+
echo " next: Claude reads frames.json, views key frames, writes the curated screen list / transcript.md"
|
|
227
|
+
else
|
|
228
|
+
echo " next: Claude reads transcript + frames.json, views key frames, writes the curated transcript.md"
|
|
229
|
+
fi
|