omni-transcriber 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- omni_transcriber-0.1.0/LICENSE +21 -0
- omni_transcriber-0.1.0/PKG-INFO +178 -0
- omni_transcriber-0.1.0/README.md +159 -0
- omni_transcriber-0.1.0/pyproject.toml +59 -0
- omni_transcriber-0.1.0/src/transcriber/__init__.py +10 -0
- omni_transcriber-0.1.0/src/transcriber/__main__.py +4 -0
- omni_transcriber-0.1.0/src/transcriber/cli.py +444 -0
- omni_transcriber-0.1.0/src/transcriber/downloader.py +345 -0
- omni_transcriber-0.1.0/src/transcriber/engines/__init__.py +40 -0
- omni_transcriber-0.1.0/src/transcriber/engines/base.py +74 -0
- omni_transcriber-0.1.0/src/transcriber/engines/local_whisper.py +232 -0
- omni_transcriber-0.1.0/src/transcriber/errors.py +34 -0
- omni_transcriber-0.1.0/src/transcriber/ffmpeg.py +40 -0
- omni_transcriber-0.1.0/src/transcriber/formats.py +170 -0
- omni_transcriber-0.1.0/src/transcriber/gpu.py +109 -0
- omni_transcriber-0.1.0/src/transcriber/naming.py +68 -0
- omni_transcriber-0.1.0/src/transcriber/progress.py +151 -0
- omni_transcriber-0.1.0/src/transcriber/resolver.py +134 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 omni-transcriber contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: omni-transcriber
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Transcribe anything: YouTube/Instagram/any URL, local media files, or whole folders - locally via Whisper.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Dist: typer>=0.15
|
|
8
|
+
Requires-Dist: rich>=13.7
|
|
9
|
+
Requires-Dist: faster-whisper>=1.2.0
|
|
10
|
+
Requires-Dist: ctranslate2>=4.6.3
|
|
11
|
+
Requires-Dist: yt-dlp[default]>=2026.1
|
|
12
|
+
Requires-Dist: onnxruntime<1.24 ; python_full_version < '3.11'
|
|
13
|
+
Requires-Dist: static-ffmpeg>=3.0
|
|
14
|
+
Requires-Dist: nvidia-cublas-cu12>=12.8 ; sys_platform != 'darwin' and extra == 'gpu'
|
|
15
|
+
Requires-Dist: nvidia-cudnn-cu12>=9.8,<10 ; sys_platform != 'darwin' and extra == 'gpu'
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Provides-Extra: gpu
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# omni-transcriber
|
|
21
|
+
|
|
22
|
+
Transcribe **anything** from the command line: a YouTube link, an Instagram
|
|
23
|
+
reel, any yt-dlp-supported URL, a local video or audio file, or a whole folder
|
|
24
|
+
of media. Transcription runs locally via [faster-whisper][fw] — no API key, no
|
|
25
|
+
audio leaves your machine.
|
|
26
|
+
|
|
27
|
+
```console
|
|
28
|
+
$ transcriber https://youtu.be/jNQXAC9IVRw
|
|
29
|
+
Alright, so here we are, one of the elephants...
|
|
30
|
+
|
|
31
|
+
$ transcriber lecture.mp4 -f srt -o lecture.srt
|
|
32
|
+
$ transcriber ~/podcasts -r -m turbo -o transcripts/
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
Needs Python 3.10 or newer. [uv][uv] is the smoothest route, but pipx and pip
|
|
38
|
+
work too.
|
|
39
|
+
|
|
40
|
+
```console
|
|
41
|
+
$ uv tool install omni-transcriber # or: pipx install omni-transcriber
|
|
42
|
+
$ uvx omni-transcriber --help # or run it without installing
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
**With NVIDIA GPU acceleration:**
|
|
46
|
+
|
|
47
|
+
```console
|
|
48
|
+
$ uv tool install "omni-transcriber[gpu]"
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
| Install | Download size | Notes |
|
|
52
|
+
| --- | --- | --- |
|
|
53
|
+
| default (CPU) | ~450 MB | Works everywhere. Fine for short clips. |
|
|
54
|
+
| `[gpu]` extra | ~2.7 GB | Adds the CUDA cuBLAS/cuDNN wheels. NVIDIA only. |
|
|
55
|
+
|
|
56
|
+
The `[gpu]` extra is never required. On a machine with an NVIDIA card but no
|
|
57
|
+
`[gpu]` install, `--device cuda` prints a warning and falls back to the CPU
|
|
58
|
+
rather than failing.
|
|
59
|
+
|
|
60
|
+
### Optional companions
|
|
61
|
+
|
|
62
|
+
- **A JavaScript runtime** — YouTube increasingly needs one. [Deno][deno] is
|
|
63
|
+
yt-dlp's preferred runtime; Node.js also works. `transcriber` enables whichever
|
|
64
|
+
it finds on your `PATH` automatically. Without one, some videos lose formats
|
|
65
|
+
or fail outright.
|
|
66
|
+
- **ffmpeg** — *not* used for transcription (the bundled PyAV decoder reads audio
|
|
67
|
+
straight out of any container). yt-dlp needs it only for a few stream types;
|
|
68
|
+
if it is missing when that happens, a static build is fetched automatically.
|
|
69
|
+
|
|
70
|
+
## Usage
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
transcriber INPUTS... files, folders, and/or URLs — mix freely
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
`ots` is installed as a shorter alias for the same command — `ots lecture.mp4`
|
|
77
|
+
does exactly what `transcriber lecture.mp4` does. (If you also use
|
|
78
|
+
[OpenTimestamps][ots], which ships its own `ots`, one will shadow the other on
|
|
79
|
+
your `PATH`; `transcriber` always works.)
|
|
80
|
+
|
|
81
|
+
| Flag | Meaning |
|
|
82
|
+
| --- | --- |
|
|
83
|
+
| `-f, --format` | `txt` (default), `srt`, `vtt`, `json`, `md` |
|
|
84
|
+
| `-t, --timestamps` | Timestamps in `txt`/`md` (`srt`/`vtt` always have them) |
|
|
85
|
+
| `-m, --model` | Whisper model — see the table below (default `small`) |
|
|
86
|
+
| `-o, --output` | A file for one input, or a directory (`-o out/`) for many |
|
|
87
|
+
| `-l, --language` | Force a language (`en`, `fr`, ...); default is auto-detect |
|
|
88
|
+
| `-r, --recursive` | Recurse into subfolders |
|
|
89
|
+
| `--device` | `auto` (default), `cpu`, `cuda` |
|
|
90
|
+
| `--compute-type` | `auto` (default), `int8`, `float16`, `int8_float16` |
|
|
91
|
+
| `--batch-size N` | Decoding batch size; default 8 on CUDA, sequential on CPU |
|
|
92
|
+
| `--no-vad` | Disable voice-activity filtering (keeps silence) |
|
|
93
|
+
| `--playlist` / `--limit N` | Expand a playlist link; cap how many entries |
|
|
94
|
+
| `--cookies-from-browser` / `--cookies` | Authenticate for private/login-walled media |
|
|
95
|
+
| `--keep-audio` | Keep the downloaded audio next to the transcript |
|
|
96
|
+
| `--overwrite` | Re-transcribe items that already have a transcript |
|
|
97
|
+
| `-v, --verbose` | yt-dlp and engine detail on stderr |
|
|
98
|
+
|
|
99
|
+
### Where output goes
|
|
100
|
+
|
|
101
|
+
- **One input, no `-o`** → the transcript goes to **stdout**, so it pipes:
|
|
102
|
+
`transcriber talk.mp4 | wc -w`. Progress bars and warnings all go to stderr.
|
|
103
|
+
- **`-o FILE`** → that exact file.
|
|
104
|
+
- **Several inputs, a folder, a playlist, or `-o DIR/`** → one
|
|
105
|
+
`<title>.<format>` file per item.
|
|
106
|
+
|
|
107
|
+
Existing transcripts are **skipped** with a notice rather than overwritten, so
|
|
108
|
+
re-running an interrupted batch resumes it for free. Pass `--overwrite` to redo
|
|
109
|
+
them. A batch keeps going when one item fails and prints a summary at the end;
|
|
110
|
+
the exit code is 1 if anything failed.
|
|
111
|
+
|
|
112
|
+
## Models
|
|
113
|
+
|
|
114
|
+
Whisper models are downloaded once, on first use, to the HuggingFace cache
|
|
115
|
+
(`~/.cache/huggingface`).
|
|
116
|
+
|
|
117
|
+
| Model | Download | Good for |
|
|
118
|
+
| --- | --- | --- |
|
|
119
|
+
| `tiny` | 75 MB | Quick drafts, testing |
|
|
120
|
+
| `base` | 145 MB | Clear speech, low resources |
|
|
121
|
+
| `small` | 484 MB | **Default** — the accuracy/speed sweet spot on CPU |
|
|
122
|
+
| `medium` | 1.5 GB | Better accuracy, noticeably slower on CPU |
|
|
123
|
+
| `large-v3` | 3.1 GB | Best accuracy. GPU strongly recommended |
|
|
124
|
+
| `turbo` | 1.6 GB | Near-`large-v3` accuracy, much faster. **Best GPU pick** |
|
|
125
|
+
| `distil-large-v3` | 1.5 GB | Distilled, English-only, very fast |
|
|
126
|
+
|
|
127
|
+
Add `.en` to `tiny`/`base`/`small`/`medium` for slightly better English-only
|
|
128
|
+
accuracy. Any HuggingFace CTranslate2 repo id or local model directory also
|
|
129
|
+
works.
|
|
130
|
+
|
|
131
|
+
**Speed:** on an RTX 5070 Ti, `small` transcribes 468 s of audio in 3.7 s;
|
|
132
|
+
the same job on CPU takes 122 s.
|
|
133
|
+
|
|
134
|
+
## Authentication for private media
|
|
135
|
+
|
|
136
|
+
Instagram reels, members-only videos, and anything else behind a login need
|
|
137
|
+
your browser's cookies:
|
|
138
|
+
|
|
139
|
+
```console
|
|
140
|
+
$ transcriber <reel-url> --cookies-from-browser firefox -f md -t
|
|
141
|
+
$ transcriber <reel-url> --cookies cookies.txt
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
> **Windows note:** Chrome's app-bound cookie encryption usually defeats
|
|
145
|
+
> `--cookies-from-browser chrome`. Use Firefox, or export a `cookies.txt` with a
|
|
146
|
+
> browser extension and pass `--cookies`.
|
|
147
|
+
|
|
148
|
+
## Exit codes
|
|
149
|
+
|
|
150
|
+
| Code | Meaning |
|
|
151
|
+
| --- | --- |
|
|
152
|
+
| 0 | Everything transcribed (or was already done) |
|
|
153
|
+
| 1 | At least one item failed — the message says why |
|
|
154
|
+
| 2 | Bad command-line usage |
|
|
155
|
+
| 130 | Interrupted with Ctrl-C |
|
|
156
|
+
|
|
157
|
+
## Development
|
|
158
|
+
|
|
159
|
+
From a checkout:
|
|
160
|
+
|
|
161
|
+
```console
|
|
162
|
+
$ uv sync # dev environment
|
|
163
|
+
$ uv run pytest # fast tests: no network, no models
|
|
164
|
+
$ uv run pytest -m slow # real tiny-model smoke test (~75 MB, once)
|
|
165
|
+
$ uv run ruff check . && uv run ruff format .
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
CI runs the suite on Linux, macOS, and Windows across Python 3.10 and 3.12.
|
|
169
|
+
Python 3.10 through 3.14 are verified locally on Linux.
|
|
170
|
+
|
|
171
|
+
## License
|
|
172
|
+
|
|
173
|
+
MIT — see [LICENSE](LICENSE).
|
|
174
|
+
|
|
175
|
+
[fw]: https://github.com/SYSTRAN/faster-whisper
|
|
176
|
+
[uv]: https://docs.astral.sh/uv/
|
|
177
|
+
[deno]: https://deno.com/
|
|
178
|
+
[ots]: https://opentimestamps.org/
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
# omni-transcriber
|
|
2
|
+
|
|
3
|
+
Transcribe **anything** from the command line: a YouTube link, an Instagram
|
|
4
|
+
reel, any yt-dlp-supported URL, a local video or audio file, or a whole folder
|
|
5
|
+
of media. Transcription runs locally via [faster-whisper][fw] — no API key, no
|
|
6
|
+
audio leaves your machine.
|
|
7
|
+
|
|
8
|
+
```console
|
|
9
|
+
$ transcriber https://youtu.be/jNQXAC9IVRw
|
|
10
|
+
Alright, so here we are, one of the elephants...
|
|
11
|
+
|
|
12
|
+
$ transcriber lecture.mp4 -f srt -o lecture.srt
|
|
13
|
+
$ transcriber ~/podcasts -r -m turbo -o transcripts/
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
## Install
|
|
17
|
+
|
|
18
|
+
Needs Python 3.10 or newer. [uv][uv] is the smoothest route, but pipx and pip
|
|
19
|
+
work too.
|
|
20
|
+
|
|
21
|
+
```console
|
|
22
|
+
$ uv tool install omni-transcriber # or: pipx install omni-transcriber
|
|
23
|
+
$ uvx omni-transcriber --help # or run it without installing
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
**With NVIDIA GPU acceleration:**
|
|
27
|
+
|
|
28
|
+
```console
|
|
29
|
+
$ uv tool install "omni-transcriber[gpu]"
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
| Install | Download size | Notes |
|
|
33
|
+
| --- | --- | --- |
|
|
34
|
+
| default (CPU) | ~450 MB | Works everywhere. Fine for short clips. |
|
|
35
|
+
| `[gpu]` extra | ~2.7 GB | Adds the CUDA cuBLAS/cuDNN wheels. NVIDIA only. |
|
|
36
|
+
|
|
37
|
+
The `[gpu]` extra is never required. On a machine with an NVIDIA card but no
|
|
38
|
+
`[gpu]` install, `--device cuda` prints a warning and falls back to the CPU
|
|
39
|
+
rather than failing.
|
|
40
|
+
|
|
41
|
+
### Optional companions
|
|
42
|
+
|
|
43
|
+
- **A JavaScript runtime** — YouTube increasingly needs one. [Deno][deno] is
|
|
44
|
+
yt-dlp's preferred runtime; Node.js also works. `transcriber` enables whichever
|
|
45
|
+
it finds on your `PATH` automatically. Without one, some videos lose formats
|
|
46
|
+
or fail outright.
|
|
47
|
+
- **ffmpeg** — *not* used for transcription (the bundled PyAV decoder reads audio
|
|
48
|
+
straight out of any container). yt-dlp needs it only for a few stream types;
|
|
49
|
+
if it is missing when that happens, a static build is fetched automatically.
|
|
50
|
+
|
|
51
|
+
## Usage
|
|
52
|
+
|
|
53
|
+
```
|
|
54
|
+
transcriber INPUTS... files, folders, and/or URLs — mix freely
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
`ots` is installed as a shorter alias for the same command — `ots lecture.mp4`
|
|
58
|
+
does exactly what `transcriber lecture.mp4` does. (If you also use
|
|
59
|
+
[OpenTimestamps][ots], which ships its own `ots`, one will shadow the other on
|
|
60
|
+
your `PATH`; `transcriber` always works.)
|
|
61
|
+
|
|
62
|
+
| Flag | Meaning |
|
|
63
|
+
| --- | --- |
|
|
64
|
+
| `-f, --format` | `txt` (default), `srt`, `vtt`, `json`, `md` |
|
|
65
|
+
| `-t, --timestamps` | Timestamps in `txt`/`md` (`srt`/`vtt` always have them) |
|
|
66
|
+
| `-m, --model` | Whisper model — see the table below (default `small`) |
|
|
67
|
+
| `-o, --output` | A file for one input, or a directory (`-o out/`) for many |
|
|
68
|
+
| `-l, --language` | Force a language (`en`, `fr`, ...); default is auto-detect |
|
|
69
|
+
| `-r, --recursive` | Recurse into subfolders |
|
|
70
|
+
| `--device` | `auto` (default), `cpu`, `cuda` |
|
|
71
|
+
| `--compute-type` | `auto` (default), `int8`, `float16`, `int8_float16` |
|
|
72
|
+
| `--batch-size N` | Decoding batch size; default 8 on CUDA, sequential on CPU |
|
|
73
|
+
| `--no-vad` | Disable voice-activity filtering (keeps silence) |
|
|
74
|
+
| `--playlist` / `--limit N` | Expand a playlist link; cap how many entries |
|
|
75
|
+
| `--cookies-from-browser` / `--cookies` | Authenticate for private/login-walled media |
|
|
76
|
+
| `--keep-audio` | Keep the downloaded audio next to the transcript |
|
|
77
|
+
| `--overwrite` | Re-transcribe items that already have a transcript |
|
|
78
|
+
| `-v, --verbose` | yt-dlp and engine detail on stderr |
|
|
79
|
+
|
|
80
|
+
### Where output goes
|
|
81
|
+
|
|
82
|
+
- **One input, no `-o`** → the transcript goes to **stdout**, so it pipes:
|
|
83
|
+
`transcriber talk.mp4 | wc -w`. Progress bars and warnings all go to stderr.
|
|
84
|
+
- **`-o FILE`** → that exact file.
|
|
85
|
+
- **Several inputs, a folder, a playlist, or `-o DIR/`** → one
|
|
86
|
+
`<title>.<format>` file per item.
|
|
87
|
+
|
|
88
|
+
Existing transcripts are **skipped** with a notice rather than overwritten, so
|
|
89
|
+
re-running an interrupted batch resumes it for free. Pass `--overwrite` to redo
|
|
90
|
+
them. A batch keeps going when one item fails and prints a summary at the end;
|
|
91
|
+
the exit code is 1 if anything failed.
|
|
92
|
+
|
|
93
|
+
## Models
|
|
94
|
+
|
|
95
|
+
Whisper models are downloaded once, on first use, to the HuggingFace cache
|
|
96
|
+
(`~/.cache/huggingface`).
|
|
97
|
+
|
|
98
|
+
| Model | Download | Good for |
|
|
99
|
+
| --- | --- | --- |
|
|
100
|
+
| `tiny` | 75 MB | Quick drafts, testing |
|
|
101
|
+
| `base` | 145 MB | Clear speech, low resources |
|
|
102
|
+
| `small` | 484 MB | **Default** — the accuracy/speed sweet spot on CPU |
|
|
103
|
+
| `medium` | 1.5 GB | Better accuracy, noticeably slower on CPU |
|
|
104
|
+
| `large-v3` | 3.1 GB | Best accuracy. GPU strongly recommended |
|
|
105
|
+
| `turbo` | 1.6 GB | Near-`large-v3` accuracy, much faster. **Best GPU pick** |
|
|
106
|
+
| `distil-large-v3` | 1.5 GB | Distilled, English-only, very fast |
|
|
107
|
+
|
|
108
|
+
Add `.en` to `tiny`/`base`/`small`/`medium` for slightly better English-only
|
|
109
|
+
accuracy. Any HuggingFace CTranslate2 repo id or local model directory also
|
|
110
|
+
works.
|
|
111
|
+
|
|
112
|
+
**Speed:** on an RTX 5070 Ti, `small` transcribes 468 s of audio in 3.7 s;
|
|
113
|
+
the same job on CPU takes 122 s.
|
|
114
|
+
|
|
115
|
+
## Authentication for private media
|
|
116
|
+
|
|
117
|
+
Instagram reels, members-only videos, and anything else behind a login need
|
|
118
|
+
your browser's cookies:
|
|
119
|
+
|
|
120
|
+
```console
|
|
121
|
+
$ transcriber <reel-url> --cookies-from-browser firefox -f md -t
|
|
122
|
+
$ transcriber <reel-url> --cookies cookies.txt
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
> **Windows note:** Chrome's app-bound cookie encryption usually defeats
|
|
126
|
+
> `--cookies-from-browser chrome`. Use Firefox, or export a `cookies.txt` with a
|
|
127
|
+
> browser extension and pass `--cookies`.
|
|
128
|
+
|
|
129
|
+
## Exit codes
|
|
130
|
+
|
|
131
|
+
| Code | Meaning |
|
|
132
|
+
| --- | --- |
|
|
133
|
+
| 0 | Everything transcribed (or was already done) |
|
|
134
|
+
| 1 | At least one item failed — the message says why |
|
|
135
|
+
| 2 | Bad command-line usage |
|
|
136
|
+
| 130 | Interrupted with Ctrl-C |
|
|
137
|
+
|
|
138
|
+
## Development
|
|
139
|
+
|
|
140
|
+
From a checkout:
|
|
141
|
+
|
|
142
|
+
```console
|
|
143
|
+
$ uv sync # dev environment
|
|
144
|
+
$ uv run pytest # fast tests: no network, no models
|
|
145
|
+
$ uv run pytest -m slow # real tiny-model smoke test (~75 MB, once)
|
|
146
|
+
$ uv run ruff check . && uv run ruff format .
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
CI runs the suite on Linux, macOS, and Windows across Python 3.10 and 3.12.
|
|
150
|
+
Python 3.10 through 3.14 are verified locally on Linux.
|
|
151
|
+
|
|
152
|
+
## License
|
|
153
|
+
|
|
154
|
+
MIT — see [LICENSE](LICENSE).
|
|
155
|
+
|
|
156
|
+
[fw]: https://github.com/SYSTRAN/faster-whisper
|
|
157
|
+
[uv]: https://docs.astral.sh/uv/
|
|
158
|
+
[deno]: https://deno.com/
|
|
159
|
+
[ots]: https://opentimestamps.org/
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "omni-transcriber"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Transcribe anything: YouTube/Instagram/any URL, local media files, or whole folders - locally via Whisper."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
license-files = ["LICENSE"]
|
|
9
|
+
dependencies = [
|
|
10
|
+
"typer>=0.15",
|
|
11
|
+
"rich>=13.7",
|
|
12
|
+
"faster-whisper>=1.2.0",
|
|
13
|
+
"ctranslate2>=4.6.3",
|
|
14
|
+
"yt-dlp[default]>=2026.1",
|
|
15
|
+
# Not a new dependency - faster-whisper already pulls onnxruntime for VAD.
|
|
16
|
+
# Constrained because 1.24+ publishes no cp310 wheels despite claiming
|
|
17
|
+
# requires-python >=3.10, which would break this package's own 3.10 floor.
|
|
18
|
+
"onnxruntime<1.24; python_full_version < '3.11'",
|
|
19
|
+
"static-ffmpeg>=3.0",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
gpu = [
|
|
24
|
+
"nvidia-cublas-cu12>=12.8; platform_system != 'Darwin'",
|
|
25
|
+
"nvidia-cudnn-cu12>=9.8,<10; platform_system != 'Darwin'",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.scripts]
|
|
29
|
+
transcriber = "transcriber.cli:main"
|
|
30
|
+
# Short alias for daily use. `transcriber` stays the canonical name: the
|
|
31
|
+
# opentimestamps-client package also ships an `ots` script, so on a machine with
|
|
32
|
+
# both installed one shadows the other and the long name is the way back in.
|
|
33
|
+
ots = "transcriber.cli:main"
|
|
34
|
+
|
|
35
|
+
[build-system]
|
|
36
|
+
requires = ["uv_build>=0.8,<0.9"]
|
|
37
|
+
build-backend = "uv_build"
|
|
38
|
+
|
|
39
|
+
[tool.uv.build-backend]
|
|
40
|
+
module-name = "transcriber"
|
|
41
|
+
|
|
42
|
+
[dependency-groups]
|
|
43
|
+
dev = ["pytest>=8", "ruff>=0.6"]
|
|
44
|
+
|
|
45
|
+
[tool.pytest.ini_options]
|
|
46
|
+
markers = ["slow: real-model tests"]
|
|
47
|
+
addopts = "-m 'not slow'"
|
|
48
|
+
|
|
49
|
+
[[tool.uv.index]]
|
|
50
|
+
name = "testpypi"
|
|
51
|
+
url = "https://test.pypi.org/simple/"
|
|
52
|
+
publish-url = "https://test.pypi.org/legacy/"
|
|
53
|
+
explicit = true
|
|
54
|
+
|
|
55
|
+
[tool.ruff]
|
|
56
|
+
line-length = 100
|
|
57
|
+
|
|
58
|
+
[tool.ruff.lint]
|
|
59
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""omni-transcriber: transcribe anything, locally."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
__version__ = version("omni-transcriber")
|
|
7
|
+
except PackageNotFoundError: # pragma: no cover - source checkout without install
|
|
8
|
+
__version__ = "0.0.0.dev0"
|
|
9
|
+
|
|
10
|
+
__all__ = ["__version__"]
|