localcaption 0.2.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {localcaption-0.2.0 → localcaption-0.4.0}/CHANGELOG.md +50 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/CONTRIBUTING.md +4 -1
- {localcaption-0.2.0 → localcaption-0.4.0}/PKG-INFO +141 -38
- {localcaption-0.2.0 → localcaption-0.4.0}/README.md +137 -36
- {localcaption-0.2.0 → localcaption-0.4.0}/docs/diagrams/architecture.mmd +4 -2
- localcaption-0.4.0/docs/diagrams/architecture.png +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/docs/diagrams/pipeline.mmd +1 -1
- localcaption-0.4.0/docs/diagrams/pipeline.png +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/docs/diagrams/sequence.mmd +5 -4
- localcaption-0.4.0/docs/diagrams/sequence.png +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/pyproject.toml +4 -1
- {localcaption-0.2.0 → localcaption-0.4.0}/scripts/install.sh +2 -2
- {localcaption-0.2.0 → localcaption-0.4.0}/scripts/setup.sh +2 -2
- localcaption-0.4.0/src/localcaption/backends/__init__.py +1 -0
- localcaption-0.4.0/src/localcaption/backends/faster_whisper.py +120 -0
- localcaption-0.4.0/src/localcaption/backends/whisper_cpp.py +58 -0
- localcaption-0.4.0/src/localcaption/batch.py +254 -0
- localcaption-0.4.0/src/localcaption/chapters.py +244 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/cli.py +160 -20
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/download.py +11 -3
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/errors.py +1 -1
- localcaption-0.4.0/src/localcaption/index.py +172 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/models.py +2 -2
- localcaption-0.4.0/src/localcaption/pipeline.py +237 -0
- localcaption-0.4.0/src/localcaption/summary.py +163 -0
- localcaption-0.4.0/src/localcaption/summary_prompt.txt +18 -0
- localcaption-0.4.0/src/localcaption/whisper.py +148 -0
- localcaption-0.4.0/tests/conftest.py +22 -0
- localcaption-0.4.0/tests/test_backends.py +286 -0
- localcaption-0.4.0/tests/test_batch.py +274 -0
- localcaption-0.4.0/tests/test_chapters.py +163 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/test_cli_doctor.py +33 -0
- localcaption-0.4.0/tests/test_cli_search.py +86 -0
- localcaption-0.4.0/tests/test_cli_transcribe.py +328 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/test_imports.py +1 -1
- localcaption-0.4.0/tests/test_index.py +193 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/test_models.py +5 -0
- localcaption-0.4.0/tests/test_pipeline.py +719 -0
- localcaption-0.4.0/tests/test_summary.py +305 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/test_whisper_paths.py +20 -1
- localcaption-0.2.0/docs/diagrams/architecture.png +0 -0
- localcaption-0.2.0/docs/diagrams/pipeline.png +0 -0
- localcaption-0.2.0/docs/diagrams/sequence.png +0 -0
- localcaption-0.2.0/src/localcaption/pipeline.py +0 -82
- localcaption-0.2.0/src/localcaption/whisper.py +0 -103
- {localcaption-0.2.0 → localcaption-0.4.0}/.github/FUNDING.yml +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/.github/ISSUE_TEMPLATE/config.yml +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/.github/pull_request_template.md +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/.github/workflows/ci.yml +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/.github/workflows/release.yml +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/.gitignore +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/CODE_OF_CONDUCT.md +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/LICENSE +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/SECURITY.md +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/docs/RELEASING.md +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/scripts/uninstall.sh +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/__init__.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/__main__.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/_logging.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/audio.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/src/localcaption/installer.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/__init__.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/test_cli_model.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/test_installer.py +0 -0
- {localcaption-0.2.0 → localcaption-0.4.0}/tests/test_models_download.py +0 -0
|
@@ -7,6 +7,56 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.4.0] - 2026-09-05
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
- Pluggable transcription backends. `whisper.cpp` stays the default;
|
|
14
|
+
`faster-whisper` is available via `pip install localcaption[faster]`.
|
|
15
|
+
Select with `--backend {whisper-cpp,faster-whisper}` or
|
|
16
|
+
`$LOCALCAPTION_BACKEND`. `whisper_dir` is only required for whisper.cpp.
|
|
17
|
+
- **Batch mode.** `localcaption --batch urls.txt` transcribes each non-empty,
|
|
18
|
+
non-`#` line sequentially (URLs or local files). Output goes to
|
|
19
|
+
`<out>/<videoId>/` so items cannot clobber each other; if
|
|
20
|
+
`<videoId>/<videoId>.txt` already exists the item is skipped. A summary
|
|
21
|
+
table is printed at the end. Exit 0 if every item succeeded or was
|
|
22
|
+
skipped, 1 if any failed. Python helper: `localcaption.batch.transcribe_urls`.
|
|
23
|
+
- **Local auto-summary via Ollama.** `localcaption <url-or-file> --summary`
|
|
24
|
+
POSTs the `.txt` transcript to a local Ollama instance
|
|
25
|
+
(`http://localhost:11434/api/generate`) and writes `<id>.summary.md`
|
|
26
|
+
next to it. `--summary-model` (default `llama3.1:8b`) and
|
|
27
|
+
`--summary-prompt PATH` override the model and the built-in prompt
|
|
28
|
+
(TL;DR, key points, notable quotes, action items). If Ollama is down,
|
|
29
|
+
times out, or returns a bad response, the pipeline still succeeds and
|
|
30
|
+
prints a warning. `HTTP_PROXY` is ignored for this call.
|
|
31
|
+
- **YouTube chapters.** When yt-dlp reports chapter markers, write
|
|
32
|
+
`<id>.chapters.json` and a `<id>.chaptered.md` with headings like
|
|
33
|
+
`## 00:00 Intro`. The raw whisper `<id>.txt` is unchanged.
|
|
34
|
+
- **`localcaption search <term>`.** Grep previously transcribed videos via a
|
|
35
|
+
JSONL index at `~/.local/share/localcaption/index.jsonl` (override with
|
|
36
|
+
`LOCALCAPTION_INDEX_PATH`). Matches are ranked by hit count and printed
|
|
37
|
+
with timestamps when whisper JSON/SRT is available.
|
|
38
|
+
|
|
39
|
+
### Changed
|
|
40
|
+
- Default whisper model is now `small.en` instead of `base.en`. Better
|
|
41
|
+
accuracy on accents and proper nouns; `tiny.en` remains the fast
|
|
42
|
+
fallback via `--model tiny.en`. Installer, `setup.sh`, and
|
|
43
|
+
`doctor --fix` follow the same default.
|
|
44
|
+
|
|
45
|
+
## [0.3.0] - 2026-08-16
|
|
46
|
+
|
|
47
|
+
### Added
|
|
48
|
+
- **Local video and audio files.** `localcaption /path/to/video.mp4` (or a
|
|
49
|
+
wav / mp3 / mkv / similar) runs the same ffmpeg + whisper.cpp path
|
|
50
|
+
without calling yt-dlp. Output names come from the input filename.
|
|
51
|
+
|
|
52
|
+
### Changed
|
|
53
|
+
- CLI help, `doctor`, and the README now say `<url-or-file>` instead of
|
|
54
|
+
treating this as URL-only.
|
|
55
|
+
|
|
56
|
+
### Fixed
|
|
57
|
+
- Architecture, pipeline, and sequence diagrams render with a white
|
|
58
|
+
background so they stay readable on GitHub dark mode.
|
|
59
|
+
|
|
10
60
|
## [0.2.0] - 2026-06-06
|
|
11
61
|
|
|
12
62
|
### Added
|
|
@@ -24,11 +24,14 @@ src/localcaption/
|
|
|
24
24
|
├── __main__.py # python -m localcaption
|
|
25
25
|
├── _logging.py # tiny stdout logger (no logging-module config)
|
|
26
26
|
├── audio.py # ffmpeg → 16 kHz mono WAV (stage 2)
|
|
27
|
+
├── batch.py # public Python API: transcribe_urls(...)
|
|
27
28
|
├── cli.py # argparse entry point (the `localcaption` script)
|
|
28
29
|
├── download.py # yt-dlp Python API wrapper (stage 1)
|
|
29
30
|
├── errors.py # exception hierarchy
|
|
30
31
|
├── pipeline.py # public Python API: transcribe_url(...)
|
|
31
|
-
|
|
32
|
+
├── summary.py # optional local Ollama summary
|
|
33
|
+
├── whisper.py # Backend protocol + transcribe() dispatcher
|
|
34
|
+
└── backends/ # whisper.cpp (default) and faster-whisper
|
|
32
35
|
scripts/
|
|
33
36
|
└── setup.sh # bootstraps whisper.cpp + venv + model
|
|
34
37
|
tests/ # pytest suite
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: localcaption
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Fully-local video → transcript pipeline using yt-dlp, ffmpeg, and whisper.cpp. Supports YouTube, Vimeo, Twitch, and 1000+ sites. No API keys.
|
|
5
5
|
Project-URL: Homepage, https://github.com/jatinkrmalik/localcaption
|
|
6
6
|
Project-URL: Repository, https://github.com/jatinkrmalik/localcaption
|
|
@@ -51,6 +51,8 @@ Provides-Extra: dev
|
|
|
51
51
|
Requires-Dist: pytest-cov>=4; extra == 'dev'
|
|
52
52
|
Requires-Dist: pytest>=7; extra == 'dev'
|
|
53
53
|
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
54
|
+
Provides-Extra: faster
|
|
55
|
+
Requires-Dist: faster-whisper>=1.0; extra == 'faster'
|
|
54
56
|
Description-Content-Type: text/markdown
|
|
55
57
|
|
|
56
58
|
# localcaption
|
|
@@ -81,7 +83,7 @@ Description-Content-Type: text/markdown
|
|
|
81
83
|
|---|---|
|
|
82
84
|
| Download best audio | [`yt-dlp`](https://github.com/yt-dlp/yt-dlp) (YouTube, Vimeo, Twitch, 1000+ sites) |
|
|
83
85
|
| Re-encode to 16 kHz mono WAV | [`ffmpeg`](https://ffmpeg.org/) |
|
|
84
|
-
| Transcribe locally | [`whisper.cpp`](https://github.com/ggerganov/whisper.cpp) |
|
|
86
|
+
| Transcribe locally | [`whisper.cpp`](https://github.com/ggerganov/whisper.cpp) (default) or [`faster-whisper`](https://github.com/SYSTRAN/faster-whisper) |
|
|
85
87
|
|
|
86
88
|
Nothing is uploaded to a third-party service. No OpenAI / Google / DeepL keys
|
|
87
89
|
required. Runs happily on a laptop.
|
|
@@ -118,8 +120,8 @@ localcaption doctor --fix # ~2 min on an M-series Mac
|
|
|
118
120
|
`doctor --fix` is idempotent and end-to-end: it installs missing system
|
|
119
121
|
tools (`ffmpeg`/`cmake` via `brew`/`apt`), clones + builds whisper.cpp at
|
|
120
122
|
the canonical XDG location, downloads the default model, and re-runs the
|
|
121
|
-
diagnostics to confirm everything works. Pick a
|
|
122
|
-
`--model
|
|
123
|
+
diagnostics to confirm everything works. Pick a faster model with
|
|
124
|
+
`--model tiny.en`.
|
|
123
125
|
|
|
124
126
|
Prefer to do it yourself? Two equivalent options:
|
|
125
127
|
|
|
@@ -130,13 +132,13 @@ curl -fsSL https://raw.githubusercontent.com/jatinkrmalik/localcaption/main/scri
|
|
|
130
132
|
# Option B: DIY, anywhere you like:
|
|
131
133
|
git clone https://github.com/ggerganov/whisper.cpp /path/to/whisper.cpp
|
|
132
134
|
cd /path/to/whisper.cpp && cmake -B build && cmake --build build -j --config Release
|
|
133
|
-
bash models/download-ggml-model.sh
|
|
135
|
+
bash models/download-ggml-model.sh small.en
|
|
134
136
|
export LOCALCAPTION_WHISPER_DIR=/path/to/whisper.cpp # add to your shell rc
|
|
135
137
|
```
|
|
136
138
|
|
|
137
139
|
> 💡 The `install.sh` bootstrap is just `pipx install localcaption` followed
|
|
138
140
|
> by `localcaption doctor --fix`, same logic, single source of truth.
|
|
139
|
-
> Override the default model with `WHISPER_MODEL=
|
|
141
|
+
> Override the default model with `WHISPER_MODEL=tiny.en bash install.sh`.
|
|
140
142
|
|
|
141
143
|
After install, verify everything is wired up:
|
|
142
144
|
|
|
@@ -148,7 +150,7 @@ localcaption doctor --fix # diagnostic + auto-repair anything missing
|
|
|
148
150
|
### Uninstall
|
|
149
151
|
|
|
150
152
|
To completely remove `localcaption` and everything it installed (the
|
|
151
|
-
binary, whisper.cpp build, and ggml models (about
|
|
153
|
+
binary, whisper.cpp build, and ggml models (about 500 MB total):
|
|
152
154
|
|
|
153
155
|
```bash
|
|
154
156
|
# pipx + whisper.cpp + models, with confirmation prompts:
|
|
@@ -159,7 +161,7 @@ bash scripts/uninstall.sh
|
|
|
159
161
|
```
|
|
160
162
|
|
|
161
163
|
Useful flags: `--dry-run` (preview), `--yes` (skip prompts),
|
|
162
|
-
`--keep-models` (uninstall the binary but keep the
|
|
164
|
+
`--keep-models` (uninstall the binary but keep the ~500 MB whisper.cpp +
|
|
163
165
|
models cache for next time).
|
|
164
166
|
|
|
165
167
|
Sample output:
|
|
@@ -180,7 +182,7 @@ whisper.cpp:
|
|
|
180
182
|
searching: /Users/you/.local/share/localcaption/whisper.cpp
|
|
181
183
|
✅ directory exists
|
|
182
184
|
✅ binary built (.../build/bin/whisper-cli)
|
|
183
|
-
✅ models present (ggml-
|
|
185
|
+
✅ models present (ggml-small.en.bin)
|
|
184
186
|
|
|
185
187
|
All checks passed. You're good to go: localcaption <url>
|
|
186
188
|
```
|
|
@@ -191,7 +193,7 @@ download the default model, then re-verify:
|
|
|
191
193
|
|
|
192
194
|
```bash
|
|
193
195
|
localcaption doctor --fix # repair everything
|
|
194
|
-
localcaption doctor --fix --model
|
|
196
|
+
localcaption doctor --fix --model tiny.en # …with a faster/smaller model
|
|
195
197
|
```
|
|
196
198
|
|
|
197
199
|
### Dev install (contributors)
|
|
@@ -219,16 +221,32 @@ localcaption "https://www.youtube.com/watch?v=dQw4w9WgXcQ"
|
|
|
219
221
|
|
|
220
222
|
# Vimeo, Twitch, Twitter/X, and 1000+ other sites work too
|
|
221
223
|
localcaption "https://vimeo.com/148751763"
|
|
224
|
+
|
|
225
|
+
# Local video/audio files
|
|
226
|
+
localcaption /path/to/video.mp4
|
|
227
|
+
localcaption ./recording.wav
|
|
228
|
+
|
|
229
|
+
# Batch: one URL or local path per line (# comments and blank lines ignored)
|
|
230
|
+
localcaption --batch urls.txt -o transcripts/ -m small.en
|
|
231
|
+
|
|
232
|
+
# Transcript + local summary (requires a running Ollama)
|
|
233
|
+
localcaption "https://www.youtube.com/watch?v=dQw4w9WgXcQ" --summary
|
|
234
|
+
localcaption ./talk.mp4 --summary --summary-model llama3.1:8b
|
|
222
235
|
```
|
|
223
236
|
|
|
224
237
|
| flag | default | what it does |
|
|
225
238
|
|---|---|---|
|
|
226
|
-
| `-m`, `--model` | `
|
|
239
|
+
| `-m`, `--model` | `small.en` | whisper model name (`tiny.en`, `base.en`, `small.en`, `medium.en`, `large-v3`, …) |
|
|
227
240
|
| `-o`, `--out` | `./transcripts` | output directory |
|
|
228
241
|
| `-l`, `--language` | `auto` | ISO language code, or `auto` to let whisper detect it |
|
|
229
|
-
| `--
|
|
242
|
+
| `--backend` | `whisper-cpp` | transcription backend: `whisper-cpp` or `faster-whisper`. `$LOCALCAPTION_BACKEND` if the flag is omitted |
|
|
243
|
+
| `--whisper-dir` | auto-detect¹ | path to a built whisper.cpp checkout (whisper-cpp backend) |
|
|
230
244
|
| `--keep-audio` | off | keep the downloaded audio + intermediate WAV in `<out>/.work/` |
|
|
231
245
|
| `--no-print` | off | don't echo the transcript to stdout |
|
|
246
|
+
| `--batch FILE` | off | transcribe each non-empty, non-`#` line in FILE sequentially |
|
|
247
|
+
| `--summary` | off | after transcription, write `<id>.summary.md` via local Ollama |
|
|
248
|
+
| `--summary-model` | `llama3.1:8b` | Ollama model used by `--summary` |
|
|
249
|
+
| `--summary-prompt` | built-in | path to a prompt template (`{transcript}` is replaced if present) |
|
|
232
250
|
|
|
233
251
|
¹ `--whisper-dir` resolution order:
|
|
234
252
|
1. The explicit flag value, if given.
|
|
@@ -236,27 +254,88 @@ localcaption "https://vimeo.com/148751763"
|
|
|
236
254
|
3. `./whisper.cpp` (dev checkout).
|
|
237
255
|
4. `~/.local/share/localcaption/whisper.cpp` (where `install.sh` puts it).
|
|
238
256
|
|
|
239
|
-
Outputs `<videoId>.txt`, `.srt`, `.vtt`, and `.json` in the chosen directory.
|
|
257
|
+
Outputs `<videoId>.txt`, `.srt`, `.vtt`, and `.json` in the chosen directory. For local files, the output filename is derived from the input file's name. With `--summary`, also writes `<videoId>.summary.md`. When the source has chapter markers (typical on YouTube), also writes `<videoId>.chapters.json` and `<videoId>.chaptered.md`. The raw whisper `.txt` is left unchanged.
|
|
258
|
+
|
|
259
|
+
`--batch FILE` writes each item into `<out>/<videoId>/` and skips any video
|
|
260
|
+
whose `.txt` is already there, so you can re-run a list after a failure.
|
|
261
|
+
Local paths are relative to the list file (and `~` is expanded). The process
|
|
262
|
+
is sequential (whisper.cpp already saturates the machine). Exit 0 if
|
|
263
|
+
everything succeeded or was skipped, 1 otherwise.
|
|
264
|
+
|
|
265
|
+
You can also invoke it as a module: `python -m localcaption <url-or-file>`.
|
|
266
|
+
|
|
267
|
+
### faster-whisper (optional)
|
|
268
|
+
|
|
269
|
+
`whisper.cpp` is the default backend and needs no extra Python packages.
|
|
270
|
+
To use [faster-whisper](https://github.com/SYSTRAN/faster-whisper) instead
|
|
271
|
+
(CTranslate2, typically faster on CPU/CUDA, including Windows):
|
|
272
|
+
|
|
273
|
+
```bash
|
|
274
|
+
pip install 'localcaption[faster]'
|
|
275
|
+
# pipx:
|
|
276
|
+
pipx inject localcaption faster-whisper
|
|
277
|
+
|
|
278
|
+
localcaption --backend faster-whisper "https://www.youtube.com/watch?v=..."
|
|
279
|
+
# or:
|
|
280
|
+
export LOCALCAPTION_BACKEND=faster-whisper
|
|
281
|
+
```
|
|
282
|
+
|
|
283
|
+
faster-whisper downloads its own CTranslate2 weights on first use; it does
|
|
284
|
+
not read ggml files from `--whisper-dir`. `--model` names (`base.en`,
|
|
285
|
+
`small.en`, `large-v3`, ...) match the usual Whisper sizes.
|
|
286
|
+
|
|
287
|
+
### Summaries (optional)
|
|
288
|
+
|
|
289
|
+
If [Ollama](https://ollama.com) is running locally, `--summary` sends the
|
|
290
|
+
`.txt` transcript to `http://localhost:11434/api/generate` and writes
|
|
291
|
+
`<id>.summary.md` next to it. The built-in prompt asks for a TL;DR, key
|
|
292
|
+
points, notable quotes, and action items.
|
|
293
|
+
|
|
294
|
+
```bash
|
|
295
|
+
localcaption <url-or-file> --summary
|
|
296
|
+
localcaption <url-or-file> --summary --summary-model mistral
|
|
297
|
+
localcaption <url-or-file> --summary --summary-prompt ./my_prompt.txt
|
|
298
|
+
```
|
|
240
299
|
|
|
241
|
-
|
|
300
|
+
If Ollama isn't reachable, localcaption logs a warning and still exits 0.
|
|
301
|
+
The transcript files are unchanged.
|
|
242
302
|
|
|
243
303
|
### Subcommands
|
|
244
304
|
|
|
245
305
|
| Subcommand | What it does |
|
|
246
306
|
|---|---|
|
|
247
|
-
| _(default)_ `localcaption <url>` | Transcribe a
|
|
307
|
+
| _(default)_ `localcaption <url-or-file>` | Transcribe a URL or local video/audio file. |
|
|
248
308
|
| `localcaption doctor` | Read-only diagnostic: prereqs, whisper.cpp, available models. Useful before filing a bug. |
|
|
249
309
|
| `localcaption doctor --fix` | Self-heal: install missing system deps, clone+build whisper.cpp, download the default model, then re-verify. Idempotent. |
|
|
250
310
|
| `localcaption model list` | List every supported whisper model with size + install status. |
|
|
251
311
|
| `localcaption model info <name>` | Show metadata about a single model. |
|
|
252
312
|
| `localcaption model download <name>` | Download a model with progress bar + atomic writes. |
|
|
253
313
|
| `localcaption model rm <name>` | Remove an installed model to free disk space. |
|
|
314
|
+
| `localcaption search <term>` | Grep previously transcribed videos. Ranked matches with timestamps. |
|
|
315
|
+
|
|
316
|
+
### Search past transcripts
|
|
317
|
+
|
|
318
|
+
Each successful transcription upserts one JSON line in
|
|
319
|
+
`~/.local/share/localcaption/index.jsonl` (`id`, `url`, `title`, `duration`,
|
|
320
|
+
`chapters`, `transcript`). Re-running the same id replaces that row. Override
|
|
321
|
+
the path with `LOCALCAPTION_INDEX_PATH`.
|
|
322
|
+
|
|
323
|
+
```bash
|
|
324
|
+
localcaption search "install"
|
|
325
|
+
# vid123 02:30 First, let's install pip
|
|
326
|
+
# Lecture on Python tooling
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
Search is a case-insensitive substring. Hits are ranked by how often the term
|
|
330
|
+
appears (title matches get a small boost). Timestamps come from the sibling
|
|
331
|
+
whisper `.json` or `.srt` when those files are still next to the `.txt`.
|
|
254
332
|
|
|
255
333
|
### Managing models
|
|
256
334
|
|
|
257
|
-
`localcaption`
|
|
258
|
-
|
|
259
|
-
|
|
335
|
+
`localcaption` defaults to `small.en` (~466 MB), downloaded by
|
|
336
|
+
`doctor --fix` or on first use. For a faster run use `--model tiny.en`;
|
|
337
|
+
for non-English audio, pick a multilingual model. If the model isn't
|
|
338
|
+
already installed, you'll be prompted to download it:
|
|
260
339
|
|
|
261
340
|
```bash
|
|
262
341
|
$ localcaption --model small.en "https://www.youtube.com/watch?v=..."
|
|
@@ -285,9 +364,9 @@ localcaption --model small.en --auto-download "https://www.youtube.com/..."
|
|
|
285
364
|
|
|
286
365
|
| Model | Size | Best for |
|
|
287
366
|
|---|---|---|
|
|
288
|
-
| `tiny.en` | 75 MB |
|
|
289
|
-
| `base.en` | 142 MB |
|
|
290
|
-
| `small.en` | 466 MB | **
|
|
367
|
+
| `tiny.en` | 75 MB | Fast fallback, English only, low-resource environments |
|
|
368
|
+
| `base.en` | 142 MB | Faster than `small.en`, lower accuracy |
|
|
369
|
+
| `small.en` | 466 MB | **Install default**, English, accuracy/speed balance |
|
|
291
370
|
| `medium.en` | 1.5 GB | High accuracy English, ~3× slower than `small.en` |
|
|
292
371
|
| `large-v3` | 3.0 GB | Best accuracy, multilingual, slow |
|
|
293
372
|
| `large-v3-turbo` | 1.6 GB | Near-large quality at ~half the size, great compromise |
|
|
@@ -304,17 +383,37 @@ result = transcribe_url(
|
|
|
304
383
|
"https://www.youtube.com/watch?v=dQw4w9WgXcQ",
|
|
305
384
|
out_dir=Path("transcripts"),
|
|
306
385
|
whisper_dir=Path("whisper.cpp"),
|
|
307
|
-
model="
|
|
386
|
+
model="small.en",
|
|
387
|
+
summary=True, # optional; writes .summary.md via local Ollama
|
|
308
388
|
)
|
|
309
389
|
print(result.transcripts.txt.read_text())
|
|
390
|
+
|
|
391
|
+
# faster-whisper (pip install 'localcaption[faster]') does not need whisper_dir:
|
|
392
|
+
# transcribe_url(url, out_dir=Path("transcripts"), backend="faster-whisper")
|
|
393
|
+
```
|
|
394
|
+
|
|
395
|
+
Batch from Python:
|
|
396
|
+
|
|
397
|
+
```python
|
|
398
|
+
from pathlib import Path
|
|
399
|
+
from localcaption.batch import read_url_list, transcribe_urls
|
|
400
|
+
|
|
401
|
+
result = transcribe_urls(
|
|
402
|
+
read_url_list(Path("urls.txt")),
|
|
403
|
+
out_dir=Path("transcripts"),
|
|
404
|
+
whisper_dir=Path("whisper.cpp"),
|
|
405
|
+
model="small.en",
|
|
406
|
+
)
|
|
407
|
+
print(result.summary())
|
|
310
408
|
```
|
|
311
409
|
|
|
312
410
|
## Architecture
|
|
313
411
|
|
|
314
412
|
`localcaption` is intentionally tiny: an orchestrator (`pipeline.py`) drives
|
|
315
413
|
three single-responsibility stages, each wrapping one external tool. The
|
|
316
|
-
|
|
317
|
-
|
|
414
|
+
transcribe stage is a small `Backend` protocol; `whisper.cpp` is the default
|
|
415
|
+
implementation and `faster-whisper` is an optional extra. Swapping backends
|
|
416
|
+
does not touch `download.py` or `audio.py`.
|
|
318
417
|
|
|
319
418
|
### Module map
|
|
320
419
|
|
|
@@ -323,8 +422,9 @@ for `faster-whisper` without touching `download.py` or `audio.py`.
|
|
|
323
422
|
| Layer | Files | Responsibility |
|
|
324
423
|
|---|---|---|
|
|
325
424
|
| Entry points | `cli.py`, `__main__.py` | argparse, exit codes, stdout formatting |
|
|
326
|
-
| Orchestration | `pipeline.py` | public Python API: `transcribe_url(...)` |
|
|
327
|
-
| Pipeline stages | `download.py`, `audio.py`, `whisper.py` |
|
|
425
|
+
| Orchestration | `pipeline.py`, `batch.py` | public Python API: `transcribe_url(...)`, `transcribe_urls(...)` |
|
|
426
|
+
| Pipeline stages | `download.py`, `audio.py`, `whisper.py`, `backends/`, `summary.py` | download, re-encode, transcribe (pluggable), optional Ollama summary |
|
|
427
|
+
| Chapters & search | `chapters.py`, `index.py` | YouTube chapter sidecars + JSONL search index |
|
|
328
428
|
| Support | `errors.py`, `_logging.py` | exception hierarchy, tiny logger |
|
|
329
429
|
|
|
330
430
|
### Runtime sequence
|
|
@@ -339,15 +439,17 @@ the subprocess hops to yt-dlp, ffmpeg, and whisper.cpp. The intermediate
|
|
|
339
439
|
> files alongside the rendered PNGs. Regenerate with:
|
|
340
440
|
> ```bash
|
|
341
441
|
> mmdc -i docs/diagrams/<name>.mmd -o docs/diagrams/<name>.png \
|
|
342
|
-
> -t default -b
|
|
442
|
+
> -t default -b white --width 1600 --scale 2
|
|
343
443
|
> ```
|
|
344
444
|
|
|
345
445
|
## Benchmarks
|
|
346
446
|
|
|
347
447
|
Wall-clock times for the **complete** pipeline (yt-dlp download → ffmpeg
|
|
348
|
-
re-encode → whisper.cpp transcription), measured with
|
|
349
|
-
|
|
350
|
-
|
|
448
|
+
re-encode → whisper.cpp transcription), measured with `base.en` (the
|
|
449
|
+
previous default). These have **not** been re-run on `small.en`; expect
|
|
450
|
+
transcription to take longer. Numbers will vary with your network speed
|
|
451
|
+
and CPU/GPU; treat them as order-of-magnitude reference, not a
|
|
452
|
+
competitive benchmark.
|
|
351
453
|
|
|
352
454
|
| Video | Length | Wall-clock | Speed vs. realtime | Hardware |
|
|
353
455
|
|---|---|---|---|---|
|
|
@@ -360,15 +462,16 @@ order-of-magnitude reference, not a competitive benchmark.
|
|
|
360
462
|
|
|
361
463
|
```bash
|
|
362
464
|
# Apple Silicon, macOS, whisper.cpp built with Metal,
|
|
363
|
-
# model: ggml-base.en
|
|
465
|
+
# model: ggml-base.en (matches the table above; not the current default),
|
|
466
|
+
# language: auto, no other heavy processes.
|
|
364
467
|
|
|
365
|
-
time localcaption --no-print -o /tmp/lc-bench-1 \
|
|
468
|
+
time localcaption --model base.en --no-print -o /tmp/lc-bench-1 \
|
|
366
469
|
"https://www.youtube.com/watch?v=PSRJfaAYkW4"
|
|
367
470
|
|
|
368
|
-
time localcaption --no-print -o /tmp/lc-bench-2 \
|
|
471
|
+
time localcaption --model base.en --no-print -o /tmp/lc-bench-2 \
|
|
369
472
|
"https://www.youtube.com/watch?v=aircAruvnKk"
|
|
370
473
|
|
|
371
|
-
time localcaption --no-print -o /tmp/lc-bench-3 \
|
|
474
|
+
time localcaption --model base.en --no-print -o /tmp/lc-bench-3 \
|
|
372
475
|
"https://www.youtube.com/watch?v=BYizgB2FcAQ"
|
|
373
476
|
```
|
|
374
477
|
|
|
@@ -379,8 +482,8 @@ hardware in the **Hardware** column.
|
|
|
379
482
|
|
|
380
483
|
## Notes
|
|
381
484
|
|
|
382
|
-
- Bigger models = better quality but slower. `
|
|
383
|
-
|
|
485
|
+
- Bigger models = better quality but slower. `small.en` is the default;
|
|
486
|
+
use `--model tiny.en` when you want speed over accuracy.
|
|
384
487
|
- Apple Silicon: whisper.cpp's CMake build uses Metal automatically, you'll
|
|
385
488
|
see `ggml_metal_init` in the logs.
|
|
386
489
|
- The pipeline accepts any URL `yt-dlp` supports (Vimeo, Twitch VODs, Twitter/X,
|
|
@@ -407,7 +510,7 @@ criteria, and discussion):
|
|
|
407
510
|
| [#4](https://github.com/jatinkrmalik/localcaption/issues/4) | Speaker diarization with pyannote.audio (`--diarize`) | `stretch`, `help wanted` |
|
|
408
511
|
| [#5](https://github.com/jatinkrmalik/localcaption/issues/5) | YouTube chapters & grep-able search index | `enhancement` |
|
|
409
512
|
| [#6](https://github.com/jatinkrmalik/localcaption/issues/6) | Pluggable transcription backends (faster-whisper / MLX) | `help wanted` |
|
|
410
|
-
|
|
|
513
|
+
| [#1](https://github.com/jatinkrmalik/localcaption/issues/1) | Switch default model from `base.en` to `small.en` | _unreleased_ ✅ |
|
|
411
514
|
|
|
412
515
|
**Have an idea?** Open a
|
|
413
516
|
[feature request](https://github.com/jatinkrmalik/localcaption/issues/new/choose),
|