omni-transcriber 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 omni-transcriber contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,178 @@
1
+ Metadata-Version: 2.4
2
+ Name: omni-transcriber
3
+ Version: 0.1.0
4
+ Summary: Transcribe anything: YouTube/Instagram/any URL, local media files, or whole folders - locally via Whisper.
5
+ License-Expression: MIT
6
+ License-File: LICENSE
7
+ Requires-Dist: typer>=0.15
8
+ Requires-Dist: rich>=13.7
9
+ Requires-Dist: faster-whisper>=1.2.0
10
+ Requires-Dist: ctranslate2>=4.6.3
11
+ Requires-Dist: yt-dlp[default]>=2026.1
12
+ Requires-Dist: onnxruntime<1.24 ; python_full_version < '3.11'
13
+ Requires-Dist: static-ffmpeg>=3.0
14
+ Requires-Dist: nvidia-cublas-cu12>=12.8 ; sys_platform != 'darwin' and extra == 'gpu'
15
+ Requires-Dist: nvidia-cudnn-cu12>=9.8,<10 ; sys_platform != 'darwin' and extra == 'gpu'
16
+ Requires-Python: >=3.10
17
+ Provides-Extra: gpu
18
+ Description-Content-Type: text/markdown
19
+
20
+ # omni-transcriber
21
+
22
+ Transcribe **anything** from the command line: a YouTube link, an Instagram
23
+ reel, any yt-dlp-supported URL, a local video or audio file, or a whole folder
24
+ of media. Transcription runs locally via [faster-whisper][fw] — no API key, no
25
+ audio leaves your machine.
26
+
27
+ ```console
28
+ $ transcriber https://youtu.be/jNQXAC9IVRw
29
+ Alright, so here we are, one of the elephants...
30
+
31
+ $ transcriber lecture.mp4 -f srt -o lecture.srt
32
+ $ transcriber ~/podcasts -r -m turbo -o transcripts/
33
+ ```
34
+
35
+ ## Install
36
+
37
+ Needs Python 3.10 or newer. [uv][uv] is the smoothest route, but pipx and pip
38
+ work too.
39
+
40
+ ```console
41
+ $ uv tool install omni-transcriber # or: pipx install omni-transcriber
42
+ $ uvx omni-transcriber --help # or run it without installing
43
+ ```
44
+
45
+ **With NVIDIA GPU acceleration:**
46
+
47
+ ```console
48
+ $ uv tool install "omni-transcriber[gpu]"
49
+ ```
50
+
51
+ | Install | Download size | Notes |
52
+ | --- | --- | --- |
53
+ | default (CPU) | ~450 MB | Works everywhere. Fine for short clips. |
54
+ | `[gpu]` extra | ~2.7 GB | Adds the CUDA cuBLAS/cuDNN wheels. NVIDIA only. |
55
+
56
+ The `[gpu]` extra is never required. On a machine with an NVIDIA card but no
57
+ `[gpu]` install, `--device cuda` prints a warning and falls back to the CPU
58
+ rather than failing.
59
+
60
+ ### Optional companions
61
+
62
+ - **A JavaScript runtime** — YouTube increasingly needs one. [Deno][deno] is
63
+ yt-dlp's preferred runtime; Node.js also works. `transcriber` enables whichever
64
+ it finds on your `PATH` automatically. Without one, some videos lose formats
65
+ or fail outright.
66
+ - **ffmpeg** — *not* used for transcription (the bundled PyAV decoder reads audio
67
+ straight out of any container). yt-dlp needs it only for a few stream types;
68
+ if it is missing when that happens, a static build is fetched automatically.
69
+
70
+ ## Usage
71
+
72
+ ```
73
+ transcriber INPUTS... files, folders, and/or URLs — mix freely
74
+ ```
75
+
76
+ `ots` is installed as a shorter alias for the same command — `ots lecture.mp4`
77
+ does exactly what `transcriber lecture.mp4` does. (If you also use
78
+ [OpenTimestamps][ots], which ships its own `ots`, one will shadow the other on
79
+ your `PATH`; `transcriber` always works.)
80
+
81
+ | Flag | Meaning |
82
+ | --- | --- |
83
+ | `-f, --format` | `txt` (default), `srt`, `vtt`, `json`, `md` |
84
+ | `-t, --timestamps` | Timestamps in `txt`/`md` (`srt`/`vtt` always have them) |
85
+ | `-m, --model` | Whisper model — see the table below (default `small`) |
86
+ | `-o, --output` | A file for one input, or a directory (`-o out/`) for many |
87
+ | `-l, --language` | Force a language (`en`, `fr`, ...); default is auto-detect |
88
+ | `-r, --recursive` | Recurse into subfolders |
89
+ | `--device` | `auto` (default), `cpu`, `cuda` |
90
+ | `--compute-type` | `auto` (default), `int8`, `float16`, `int8_float16` |
91
+ | `--batch-size N` | Decoding batch size; default 8 on CUDA, sequential on CPU |
92
+ | `--no-vad` | Disable voice-activity filtering (keeps silence) |
93
+ | `--playlist` / `--limit N` | Expand a playlist link; cap how many entries |
94
+ | `--cookies-from-browser` / `--cookies` | Authenticate for private/login-walled media |
95
+ | `--keep-audio` | Keep the downloaded audio next to the transcript |
96
+ | `--overwrite` | Re-transcribe items that already have a transcript |
97
+ | `-v, --verbose` | yt-dlp and engine detail on stderr |
98
+
99
+ ### Where output goes
100
+
101
+ - **One input, no `-o`** → the transcript goes to **stdout**, so it pipes:
102
+ `transcriber talk.mp4 | wc -w`. Progress bars and warnings all go to stderr.
103
+ - **`-o FILE`** → that exact file.
104
+ - **Several inputs, a folder, a playlist, or `-o DIR/`** → one
105
+ `<title>.<format>` file per item.
106
+
107
+ Existing transcripts are **skipped** with a notice rather than overwritten, so
108
+ re-running an interrupted batch resumes it for free. Pass `--overwrite` to redo
109
+ them. A batch keeps going when one item fails and prints a summary at the end;
110
+ the exit code is 1 if anything failed.
111
+
112
+ ## Models
113
+
114
+ Whisper models are downloaded once, on first use, to the HuggingFace cache
115
+ (`~/.cache/huggingface`).
116
+
117
+ | Model | Download | Good for |
118
+ | --- | --- | --- |
119
+ | `tiny` | 75 MB | Quick drafts, testing |
120
+ | `base` | 145 MB | Clear speech, low resources |
121
+ | `small` | 484 MB | **Default** — the accuracy/speed sweet spot on CPU |
122
+ | `medium` | 1.5 GB | Better accuracy, noticeably slower on CPU |
123
+ | `large-v3` | 3.1 GB | Best accuracy. GPU strongly recommended |
124
+ | `turbo` | 1.6 GB | Near-`large-v3` accuracy, much faster. **Best GPU pick** |
125
+ | `distil-large-v3` | 1.5 GB | Distilled, English-only, very fast |
126
+
127
+ Add `.en` to `tiny`/`base`/`small`/`medium` for slightly better English-only
128
+ accuracy. Any HuggingFace CTranslate2 repo id or local model directory also
129
+ works.
130
+
131
+ **Speed:** on an RTX 5070 Ti, `small` transcribes 468 s of audio in 3.7 s;
132
+ the same job on CPU takes 122 s.
133
+
134
+ ## Authentication for private media
135
+
136
+ Instagram reels, members-only videos, and anything else behind a login need
137
+ your browser's cookies:
138
+
139
+ ```console
140
+ $ transcriber <reel-url> --cookies-from-browser firefox -f md -t
141
+ $ transcriber <reel-url> --cookies cookies.txt
142
+ ```
143
+
144
+ > **Windows note:** Chrome's app-bound cookie encryption usually defeats
145
+ > `--cookies-from-browser chrome`. Use Firefox, or export a `cookies.txt` with a
146
+ > browser extension and pass `--cookies`.
147
+
148
+ ## Exit codes
149
+
150
+ | Code | Meaning |
151
+ | --- | --- |
152
+ | 0 | Everything transcribed (or was already done) |
153
+ | 1 | At least one item failed — the message says why |
154
+ | 2 | Bad command-line usage |
155
+ | 130 | Interrupted with Ctrl-C |
156
+
157
+ ## Development
158
+
159
+ From a checkout:
160
+
161
+ ```console
162
+ $ uv sync # dev environment
163
+ $ uv run pytest # fast tests: no network, no models
164
+ $ uv run pytest -m slow # real tiny-model smoke test (~75 MB, once)
165
+ $ uv run ruff check . && uv run ruff format .
166
+ ```
167
+
168
+ CI runs the suite on Linux, macOS, and Windows across Python 3.10 and 3.12.
169
+ Python 3.10 through 3.14 are verified locally on Linux.
170
+
171
+ ## License
172
+
173
+ MIT — see [LICENSE](LICENSE).
174
+
175
+ [fw]: https://github.com/SYSTRAN/faster-whisper
176
+ [uv]: https://docs.astral.sh/uv/
177
+ [deno]: https://deno.com/
178
+ [ots]: https://opentimestamps.org/
@@ -0,0 +1,159 @@
1
+ # omni-transcriber
2
+
3
+ Transcribe **anything** from the command line: a YouTube link, an Instagram
4
+ reel, any yt-dlp-supported URL, a local video or audio file, or a whole folder
5
+ of media. Transcription runs locally via [faster-whisper][fw] — no API key, no
6
+ audio leaves your machine.
7
+
8
+ ```console
9
+ $ transcriber https://youtu.be/jNQXAC9IVRw
10
+ Alright, so here we are, one of the elephants...
11
+
12
+ $ transcriber lecture.mp4 -f srt -o lecture.srt
13
+ $ transcriber ~/podcasts -r -m turbo -o transcripts/
14
+ ```
15
+
16
+ ## Install
17
+
18
+ Needs Python 3.10 or newer. [uv][uv] is the smoothest route, but pipx and pip
19
+ work too.
20
+
21
+ ```console
22
+ $ uv tool install omni-transcriber # or: pipx install omni-transcriber
23
+ $ uvx omni-transcriber --help # or run it without installing
24
+ ```
25
+
26
+ **With NVIDIA GPU acceleration:**
27
+
28
+ ```console
29
+ $ uv tool install "omni-transcriber[gpu]"
30
+ ```
31
+
32
+ | Install | Download size | Notes |
33
+ | --- | --- | --- |
34
+ | default (CPU) | ~450 MB | Works everywhere. Fine for short clips. |
35
+ | `[gpu]` extra | ~2.7 GB | Adds the CUDA cuBLAS/cuDNN wheels. NVIDIA only. |
36
+
37
+ The `[gpu]` extra is never required. On a machine with an NVIDIA card but no
38
+ `[gpu]` install, `--device cuda` prints a warning and falls back to the CPU
39
+ rather than failing.
40
+
41
+ ### Optional companions
42
+
43
+ - **A JavaScript runtime** — YouTube increasingly needs one. [Deno][deno] is
44
+ yt-dlp's preferred runtime; Node.js also works. `transcriber` enables whichever
45
+ it finds on your `PATH` automatically. Without one, some videos lose formats
46
+ or fail outright.
47
+ - **ffmpeg** — *not* used for transcription (the bundled PyAV decoder reads audio
48
+ straight out of any container). yt-dlp needs it only for a few stream types;
49
+ if it is missing when that happens, a static build is fetched automatically.
50
+
51
+ ## Usage
52
+
53
+ ```
54
+ transcriber INPUTS... files, folders, and/or URLs — mix freely
55
+ ```
56
+
57
+ `ots` is installed as a shorter alias for the same command — `ots lecture.mp4`
58
+ does exactly what `transcriber lecture.mp4` does. (If you also use
59
+ [OpenTimestamps][ots], which ships its own `ots`, one will shadow the other on
60
+ your `PATH`; `transcriber` always works.)
61
+
62
+ | Flag | Meaning |
63
+ | --- | --- |
64
+ | `-f, --format` | `txt` (default), `srt`, `vtt`, `json`, `md` |
65
+ | `-t, --timestamps` | Timestamps in `txt`/`md` (`srt`/`vtt` always have them) |
66
+ | `-m, --model` | Whisper model — see the table below (default `small`) |
67
+ | `-o, --output` | A file for one input, or a directory (`-o out/`) for many |
68
+ | `-l, --language` | Force a language (`en`, `fr`, ...); default is auto-detect |
69
+ | `-r, --recursive` | Recurse into subfolders |
70
+ | `--device` | `auto` (default), `cpu`, `cuda` |
71
+ | `--compute-type` | `auto` (default), `int8`, `float16`, `int8_float16` |
72
+ | `--batch-size N` | Decoding batch size; default 8 on CUDA, sequential on CPU |
73
+ | `--no-vad` | Disable voice-activity filtering (keeps silence) |
74
+ | `--playlist` / `--limit N` | Expand a playlist link; cap how many entries |
75
+ | `--cookies-from-browser` / `--cookies` | Authenticate for private/login-walled media |
76
+ | `--keep-audio` | Keep the downloaded audio next to the transcript |
77
+ | `--overwrite` | Re-transcribe items that already have a transcript |
78
+ | `-v, --verbose` | yt-dlp and engine detail on stderr |
79
+
80
+ ### Where output goes
81
+
82
+ - **One input, no `-o`** → the transcript goes to **stdout**, so it pipes:
83
+ `transcriber talk.mp4 | wc -w`. Progress bars and warnings all go to stderr.
84
+ - **`-o FILE`** → that exact file.
85
+ - **Several inputs, a folder, a playlist, or `-o DIR/`** → one
86
+ `<title>.<format>` file per item.
87
+
88
+ Existing transcripts are **skipped** with a notice rather than overwritten, so
89
+ re-running an interrupted batch resumes it for free. Pass `--overwrite` to redo
90
+ them. A batch keeps going when one item fails and prints a summary at the end;
91
+ the exit code is 1 if anything failed.
92
+
93
+ ## Models
94
+
95
+ Whisper models are downloaded once, on first use, to the HuggingFace cache
96
+ (`~/.cache/huggingface`).
97
+
98
+ | Model | Download | Good for |
99
+ | --- | --- | --- |
100
+ | `tiny` | 75 MB | Quick drafts, testing |
101
+ | `base` | 145 MB | Clear speech, low resources |
102
+ | `small` | 484 MB | **Default** — the accuracy/speed sweet spot on CPU |
103
+ | `medium` | 1.5 GB | Better accuracy, noticeably slower on CPU |
104
+ | `large-v3` | 3.1 GB | Best accuracy. GPU strongly recommended |
105
+ | `turbo` | 1.6 GB | Near-`large-v3` accuracy, much faster. **Best GPU pick** |
106
+ | `distil-large-v3` | 1.5 GB | Distilled, English-only, very fast |
107
+
108
+ Add `.en` to `tiny`/`base`/`small`/`medium` for slightly better English-only
109
+ accuracy. Any HuggingFace CTranslate2 repo id or local model directory also
110
+ works.
111
+
112
+ **Speed:** on an RTX 5070 Ti, `small` transcribes 468 s of audio in 3.7 s;
113
+ the same job on CPU takes 122 s.
114
+
115
+ ## Authentication for private media
116
+
117
+ Instagram reels, members-only videos, and anything else behind a login need
118
+ your browser's cookies:
119
+
120
+ ```console
121
+ $ transcriber <reel-url> --cookies-from-browser firefox -f md -t
122
+ $ transcriber <reel-url> --cookies cookies.txt
123
+ ```
124
+
125
+ > **Windows note:** Chrome's app-bound cookie encryption usually defeats
126
+ > `--cookies-from-browser chrome`. Use Firefox, or export a `cookies.txt` with a
127
+ > browser extension and pass `--cookies`.
128
+
129
+ ## Exit codes
130
+
131
+ | Code | Meaning |
132
+ | --- | --- |
133
+ | 0 | Everything transcribed (or was already done) |
134
+ | 1 | At least one item failed — the message says why |
135
+ | 2 | Bad command-line usage |
136
+ | 130 | Interrupted with Ctrl-C |
137
+
138
+ ## Development
139
+
140
+ From a checkout:
141
+
142
+ ```console
143
+ $ uv sync # dev environment
144
+ $ uv run pytest # fast tests: no network, no models
145
+ $ uv run pytest -m slow # real tiny-model smoke test (~75 MB, once)
146
+ $ uv run ruff check . && uv run ruff format .
147
+ ```
148
+
149
+ CI runs the suite on Linux, macOS, and Windows across Python 3.10 and 3.12.
150
+ Python 3.10 through 3.14 are verified locally on Linux.
151
+
152
+ ## License
153
+
154
+ MIT — see [LICENSE](LICENSE).
155
+
156
+ [fw]: https://github.com/SYSTRAN/faster-whisper
157
+ [uv]: https://docs.astral.sh/uv/
158
+ [deno]: https://deno.com/
159
+ [ots]: https://opentimestamps.org/
@@ -0,0 +1,59 @@
1
+ [project]
2
+ name = "omni-transcriber"
3
+ version = "0.1.0"
4
+ description = "Transcribe anything: YouTube/Instagram/any URL, local media files, or whole folders - locally via Whisper."
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ license = "MIT"
8
+ license-files = ["LICENSE"]
9
+ dependencies = [
10
+ "typer>=0.15",
11
+ "rich>=13.7",
12
+ "faster-whisper>=1.2.0",
13
+ "ctranslate2>=4.6.3",
14
+ "yt-dlp[default]>=2026.1",
15
+ # Not a new dependency - faster-whisper already pulls onnxruntime for VAD.
16
+ # Constrained because 1.24+ publishes no cp310 wheels despite claiming
17
+ # requires-python >=3.10, which would break this package's own 3.10 floor.
18
+ "onnxruntime<1.24; python_full_version < '3.11'",
19
+ "static-ffmpeg>=3.0",
20
+ ]
21
+
22
+ [project.optional-dependencies]
23
+ gpu = [
24
+ "nvidia-cublas-cu12>=12.8; platform_system != 'Darwin'",
25
+ "nvidia-cudnn-cu12>=9.8,<10; platform_system != 'Darwin'",
26
+ ]
27
+
28
+ [project.scripts]
29
+ transcriber = "transcriber.cli:main"
30
+ # Short alias for daily use. `transcriber` stays the canonical name: the
31
+ # opentimestamps-client package also ships an `ots` script, so on a machine with
32
+ # both installed one shadows the other and the long name is the way back in.
33
+ ots = "transcriber.cli:main"
34
+
35
+ [build-system]
36
+ requires = ["uv_build>=0.8,<0.9"]
37
+ build-backend = "uv_build"
38
+
39
+ [tool.uv.build-backend]
40
+ module-name = "transcriber"
41
+
42
+ [dependency-groups]
43
+ dev = ["pytest>=8", "ruff>=0.6"]
44
+
45
+ [tool.pytest.ini_options]
46
+ markers = ["slow: real-model tests"]
47
+ addopts = "-m 'not slow'"
48
+
49
+ [[tool.uv.index]]
50
+ name = "testpypi"
51
+ url = "https://test.pypi.org/simple/"
52
+ publish-url = "https://test.pypi.org/legacy/"
53
+ explicit = true
54
+
55
+ [tool.ruff]
56
+ line-length = 100
57
+
58
+ [tool.ruff.lint]
59
+ select = ["E", "F", "I", "UP", "B"]
@@ -0,0 +1,10 @@
1
+ """omni-transcriber: transcribe anything, locally."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ try:
6
+ __version__ = version("omni-transcriber")
7
+ except PackageNotFoundError: # pragma: no cover - source checkout without install
8
+ __version__ = "0.0.0.dev0"
9
+
10
+ __all__ = ["__version__"]
@@ -0,0 +1,4 @@
1
+ from transcriber.cli import main
2
+
3
+ if __name__ == "__main__":
4
+ main()