generic-audio-transcriber 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- generic_audio_transcriber-0.1.0/LICENSE +21 -0
- generic_audio_transcriber-0.1.0/PKG-INFO +164 -0
- generic_audio_transcriber-0.1.0/README.md +144 -0
- generic_audio_transcriber-0.1.0/pyproject.toml +39 -0
- generic_audio_transcriber-0.1.0/setup.cfg +4 -0
- generic_audio_transcriber-0.1.0/src/audio_transcriber/__init__.py +6 -0
- generic_audio_transcriber-0.1.0/src/audio_transcriber/_model.py +32 -0
- generic_audio_transcriber-0.1.0/src/audio_transcriber/cli.py +39 -0
- generic_audio_transcriber-0.1.0/src/audio_transcriber/py.typed +0 -0
- generic_audio_transcriber-0.1.0/src/audio_transcriber/result.py +36 -0
- generic_audio_transcriber-0.1.0/src/audio_transcriber/transcriber.py +82 -0
- generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/PKG-INFO +164 -0
- generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/SOURCES.txt +20 -0
- generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/dependency_links.txt +1 -0
- generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/entry_points.txt +2 -0
- generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/requires.txt +5 -0
- generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/top_level.txt +1 -0
- generic_audio_transcriber-0.1.0/tests/test_cli.py +62 -0
- generic_audio_transcriber-0.1.0/tests/test_integration.py +57 -0
- generic_audio_transcriber-0.1.0/tests/test_model.py +48 -0
- generic_audio_transcriber-0.1.0/tests/test_result.py +26 -0
- generic_audio_transcriber-0.1.0/tests/test_transcribe.py +111 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Eduardo Milani
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: generic-audio-transcriber
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Drop-in speech-to-text for any Python project: audio bytes in, JSON-ready text out.
|
|
5
|
+
Author: Eduardo Milani
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/EduardoMilani8/generic-audio-transcriber
|
|
8
|
+
Keywords: speech-to-text,transcription,whisper,audio,asr
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
12
|
+
Requires-Python: >=3.9
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
License-File: LICENSE
|
|
15
|
+
Requires-Dist: faster-whisper>=1.0
|
|
16
|
+
Requires-Dist: av<19,>=11
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# generic-audio-transcriber
|
|
22
|
+
|
|
23
|
+
Drop-in speech-to-text for any Python project. Audio bytes in, JSON-ready text out.
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
from audio_transcriber import transcribe
|
|
27
|
+
|
|
28
|
+
result = transcribe(audio_bytes) # mp3, wav, webm, ogg, m4a... detected automatically
|
|
29
|
+
print(result.text) # "Hello, this is a test."
|
|
30
|
+
print(result.to_json()) # {"text": "...", "language": "en", ...}
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
No API keys, no servers, no cloud. It runs locally on the CPU, and the model is
|
|
34
|
+
downloaded automatically the first time you use it.
|
|
35
|
+
|
|
36
|
+
## Why
|
|
37
|
+
|
|
38
|
+
Many apps need the same small feature: the user sends or records audio, and the app
|
|
39
|
+
needs the text. This package is that feature as one function call, so you can add it to
|
|
40
|
+
a project without learning anything about speech recognition.
|
|
41
|
+
|
|
42
|
+
Under the hood it uses [faster-whisper](https://github.com/SYSTRAN/faster-whisper)
|
|
43
|
+
(OpenAI's Whisper model), which is accurate, supports about 99 languages, and runs well
|
|
44
|
+
on a plain CPU. You never have to train or tune a model.
|
|
45
|
+
|
|
46
|
+
## Install
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install git+https://github.com/EduardoMilani8/generic-audio-transcriber.git
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Requires Python 3.9+. You do **not** need to install ffmpeg: audio decoding is bundled.
|
|
53
|
+
|
|
54
|
+
## Usage
|
|
55
|
+
|
|
56
|
+
### From bytes (the main use case)
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
from audio_transcriber import transcribe
|
|
60
|
+
|
|
61
|
+
result = transcribe(audio_bytes)
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### From a file path or file object
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
transcribe("recording.mp3")
|
|
68
|
+
|
|
69
|
+
with open("recording.wav", "rb") as f:
|
|
70
|
+
transcribe(f)
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### Inside a web endpoint
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
from fastapi import FastAPI, UploadFile
|
|
77
|
+
from audio_transcriber import transcribe
|
|
78
|
+
|
|
79
|
+
app = FastAPI()
|
|
80
|
+
|
|
81
|
+
@app.post("/transcribe")
|
|
82
|
+
async def endpoint(file: UploadFile):
|
|
83
|
+
return transcribe(await file.read()).to_dict()
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
### The result
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
result.text # full transcript
|
|
90
|
+
result.language # detected language code, e.g. "pt"
|
|
91
|
+
result.language_probability # confidence of the language detection
|
|
92
|
+
result.duration # audio length in seconds
|
|
93
|
+
result.segments # timed pieces: .start, .end, .text
|
|
94
|
+
result.to_dict() # plain dict
|
|
95
|
+
result.to_json() # JSON string (non-ASCII kept readable)
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Silent audio gives an empty `text`, not an error.
|
|
99
|
+
|
|
100
|
+
## Options
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
transcribe(
|
|
104
|
+
audio,
|
|
105
|
+
model="small", # tiny | base | small | medium | large-v3
|
|
106
|
+
language=None, # "pt", "en", ... None = auto-detect
|
|
107
|
+
device="cpu", # "cpu" | "cuda" | "auto"
|
|
108
|
+
compute_type="int8", # "int8" is light; "float16" suits GPUs
|
|
109
|
+
beam_size=5,
|
|
110
|
+
vad_filter=True, # skip silence, reduces made-up text
|
|
111
|
+
)
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
Choosing a model is a trade-off between weight and accuracy:
|
|
115
|
+
|
|
116
|
+
| Model | Size on disk | Speed | Accuracy |
|
|
117
|
+
|---|---|---|---|
|
|
118
|
+
| `tiny` | smallest | fastest | lowest |
|
|
119
|
+
| `base` | small | fast | fair |
|
|
120
|
+
| `small` (default) | medium | good | good |
|
|
121
|
+
| `medium` / `large-v3` | large | slow on CPU | best |
|
|
122
|
+
|
|
123
|
+
The model is loaded once per process and reused, so only the first call is slow.
|
|
124
|
+
Setting `language` explicitly is faster and more accurate than auto-detection.
|
|
125
|
+
|
|
126
|
+
## Errors
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
from audio_transcriber import transcribe, TranscriptionError
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
transcribe(data)
|
|
133
|
+
except TranscriptionError as e:
|
|
134
|
+
... # empty input or audio that cannot be decoded
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## Command line
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
audio-transcriber recording.mp3 # prints JSON
|
|
141
|
+
audio-transcriber recording.mp3 --text # prints plain text
|
|
142
|
+
audio-transcriber recording.mp3 --model base --language pt
|
|
143
|
+
cat recording.mp3 | audio-transcriber - # read bytes from stdin
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
## Development
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
python -m venv .venv && source .venv/bin/activate
|
|
150
|
+
pip install -e ".[dev]"
|
|
151
|
+
pytest # fast unit tests, no model needed
|
|
152
|
+
RUN_SLOW=1 pytest -m slow # end-to-end tests, downloads the tiny model
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Notes
|
|
156
|
+
|
|
157
|
+
- The default device is `cpu` because it works on every machine. Pass `device="cuda"`
|
|
158
|
+
if you have a working CUDA setup and want GPU speed.
|
|
159
|
+
- `av` is pinned below version 19 because faster-whisper still uses an argument that
|
|
160
|
+
newer releases removed. The pin can be lifted once faster-whisper fixes it.
|
|
161
|
+
|
|
162
|
+
## License
|
|
163
|
+
|
|
164
|
+
MIT
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
# generic-audio-transcriber
|
|
2
|
+
|
|
3
|
+
Drop-in speech-to-text for any Python project. Audio bytes in, JSON-ready text out.
|
|
4
|
+
|
|
5
|
+
```python
|
|
6
|
+
from audio_transcriber import transcribe
|
|
7
|
+
|
|
8
|
+
result = transcribe(audio_bytes) # mp3, wav, webm, ogg, m4a... detected automatically
|
|
9
|
+
print(result.text) # "Hello, this is a test."
|
|
10
|
+
print(result.to_json()) # {"text": "...", "language": "en", ...}
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
No API keys, no servers, no cloud. It runs locally on the CPU, and the model is
|
|
14
|
+
downloaded automatically the first time you use it.
|
|
15
|
+
|
|
16
|
+
## Why
|
|
17
|
+
|
|
18
|
+
Many apps need the same small feature: the user sends or records audio, and the app
|
|
19
|
+
needs the text. This package is that feature as one function call, so you can add it to
|
|
20
|
+
a project without learning anything about speech recognition.
|
|
21
|
+
|
|
22
|
+
Under the hood it uses [faster-whisper](https://github.com/SYSTRAN/faster-whisper)
|
|
23
|
+
(OpenAI's Whisper model), which is accurate, supports about 99 languages, and runs well
|
|
24
|
+
on a plain CPU. You never have to train or tune a model.
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install git+https://github.com/EduardoMilani8/generic-audio-transcriber.git
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Requires Python 3.9+. You do **not** need to install ffmpeg: audio decoding is bundled.
|
|
33
|
+
|
|
34
|
+
## Usage
|
|
35
|
+
|
|
36
|
+
### From bytes (the main use case)
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
from audio_transcriber import transcribe
|
|
40
|
+
|
|
41
|
+
result = transcribe(audio_bytes)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
### From a file path or file object
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
transcribe("recording.mp3")
|
|
48
|
+
|
|
49
|
+
with open("recording.wav", "rb") as f:
|
|
50
|
+
transcribe(f)
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
### Inside a web endpoint
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from fastapi import FastAPI, UploadFile
|
|
57
|
+
from audio_transcriber import transcribe
|
|
58
|
+
|
|
59
|
+
app = FastAPI()
|
|
60
|
+
|
|
61
|
+
@app.post("/transcribe")
|
|
62
|
+
async def endpoint(file: UploadFile):
|
|
63
|
+
return transcribe(await file.read()).to_dict()
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
### The result
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
result.text # full transcript
|
|
70
|
+
result.language # detected language code, e.g. "pt"
|
|
71
|
+
result.language_probability # confidence of the language detection
|
|
72
|
+
result.duration # audio length in seconds
|
|
73
|
+
result.segments # timed pieces: .start, .end, .text
|
|
74
|
+
result.to_dict() # plain dict
|
|
75
|
+
result.to_json() # JSON string (non-ASCII kept readable)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Silent audio gives an empty `text`, not an error.
|
|
79
|
+
|
|
80
|
+
## Options
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
transcribe(
|
|
84
|
+
audio,
|
|
85
|
+
model="small", # tiny | base | small | medium | large-v3
|
|
86
|
+
language=None, # "pt", "en", ... None = auto-detect
|
|
87
|
+
device="cpu", # "cpu" | "cuda" | "auto"
|
|
88
|
+
compute_type="int8", # "int8" is light; "float16" suits GPUs
|
|
89
|
+
beam_size=5,
|
|
90
|
+
vad_filter=True, # skip silence, reduces made-up text
|
|
91
|
+
)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Choosing a model is a trade-off between weight and accuracy:
|
|
95
|
+
|
|
96
|
+
| Model | Size on disk | Speed | Accuracy |
|
|
97
|
+
|---|---|---|---|
|
|
98
|
+
| `tiny` | smallest | fastest | lowest |
|
|
99
|
+
| `base` | small | fast | fair |
|
|
100
|
+
| `small` (default) | medium | good | good |
|
|
101
|
+
| `medium` / `large-v3` | large | slow on CPU | best |
|
|
102
|
+
|
|
103
|
+
The model is loaded once per process and reused, so only the first call is slow.
|
|
104
|
+
Setting `language` explicitly is faster and more accurate than auto-detection.
|
|
105
|
+
|
|
106
|
+
## Errors
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
from audio_transcriber import transcribe, TranscriptionError
|
|
110
|
+
|
|
111
|
+
try:
|
|
112
|
+
transcribe(data)
|
|
113
|
+
except TranscriptionError as e:
|
|
114
|
+
... # empty input or audio that cannot be decoded
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
## Command line
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
audio-transcriber recording.mp3 # prints JSON
|
|
121
|
+
audio-transcriber recording.mp3 --text # prints plain text
|
|
122
|
+
audio-transcriber recording.mp3 --model base --language pt
|
|
123
|
+
cat recording.mp3 | audio-transcriber - # read bytes from stdin
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
## Development
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
python -m venv .venv && source .venv/bin/activate
|
|
130
|
+
pip install -e ".[dev]"
|
|
131
|
+
pytest # fast unit tests, no model needed
|
|
132
|
+
RUN_SLOW=1 pytest -m slow # end-to-end tests, downloads the tiny model
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
## Notes
|
|
136
|
+
|
|
137
|
+
- The default device is `cpu` because it works on every machine. Pass `device="cuda"`
|
|
138
|
+
if you have a working CUDA setup and want GPU speed.
|
|
139
|
+
- `av` is pinned below version 19 because faster-whisper still uses an argument that
|
|
140
|
+
newer releases removed. The pin can be lifted once faster-whisper fixes it.
|
|
141
|
+
|
|
142
|
+
## License
|
|
143
|
+
|
|
144
|
+
MIT
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "generic-audio-transcriber"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Drop-in speech-to-text for any Python project: audio bytes in, JSON-ready text out."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
authors = [{ name = "Eduardo Milani" }]
|
|
13
|
+
keywords = ["speech-to-text", "transcription", "whisper", "audio", "asr"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"License :: OSI Approved :: MIT License",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
18
|
+
]
|
|
19
|
+
# av<19: faster-whisper still calls av.open(metadata_errors=...), which av 19 rejects.
|
|
20
|
+
dependencies = ["faster-whisper>=1.0", "av>=11,<19"]
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
dev = ["pytest>=8"]
|
|
24
|
+
|
|
25
|
+
[project.scripts]
|
|
26
|
+
audio-transcriber = "audio_transcriber.cli:main"
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Repository = "https://github.com/EduardoMilani8/generic-audio-transcriber"
|
|
30
|
+
|
|
31
|
+
[tool.setuptools.packages.find]
|
|
32
|
+
where = ["src"]
|
|
33
|
+
|
|
34
|
+
[tool.setuptools.package-data]
|
|
35
|
+
audio_transcriber = ["py.typed"]
|
|
36
|
+
|
|
37
|
+
[tool.pytest.ini_options]
|
|
38
|
+
testpaths = ["tests"]
|
|
39
|
+
markers = ["slow: downloads a real model and runs real inference (set RUN_SLOW=1)"]
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Lazy, cached loading of the speech recognition model."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import threading
|
|
6
|
+
from typing import Any, Dict, Tuple
|
|
7
|
+
|
|
8
|
+
_cache: Dict[Tuple[str, str, str], Any] = {}
|
|
9
|
+
_lock = threading.Lock()
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def get_model(name: str, device: str, compute_type: str) -> Any:
|
|
13
|
+
"""Return a loaded model, loading (and downloading) it only the first time.
|
|
14
|
+
|
|
15
|
+
``faster_whisper`` is imported here rather than at module level so that
|
|
16
|
+
importing this package stays instant and tests can run without it.
|
|
17
|
+
"""
|
|
18
|
+
key = (name, device, compute_type)
|
|
19
|
+
with _lock:
|
|
20
|
+
model = _cache.get(key)
|
|
21
|
+
if model is None:
|
|
22
|
+
from faster_whisper import WhisperModel
|
|
23
|
+
|
|
24
|
+
model = WhisperModel(name, device=device, compute_type=compute_type)
|
|
25
|
+
_cache[key] = model
|
|
26
|
+
return model
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def clear_cache() -> None:
|
|
30
|
+
"""Drop every loaded model so its memory can be reclaimed."""
|
|
31
|
+
with _lock:
|
|
32
|
+
_cache.clear()
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Command line entry point: ``audio-transcriber FILE`` prints JSON."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from typing import List, Optional
|
|
8
|
+
|
|
9
|
+
from .transcriber import DEFAULT_MODEL, TranscriptionError, transcribe
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
13
|
+
parser = argparse.ArgumentParser(
|
|
14
|
+
prog="audio-transcriber",
|
|
15
|
+
description="Transcribe an audio file to text and print the result as JSON.",
|
|
16
|
+
)
|
|
17
|
+
parser.add_argument("audio", help="Path to an audio file, or '-' to read bytes from stdin.")
|
|
18
|
+
parser.add_argument("--model", default=DEFAULT_MODEL, help=f"Model size (default: {DEFAULT_MODEL}).")
|
|
19
|
+
parser.add_argument("--language", default=None, help="Language code such as 'pt'. Auto-detected if omitted.")
|
|
20
|
+
parser.add_argument("--text", action="store_true", help="Print only the plain text instead of JSON.")
|
|
21
|
+
return parser
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def main(argv: Optional[List[str]] = None) -> int:
|
|
25
|
+
args = build_parser().parse_args(argv)
|
|
26
|
+
audio = sys.stdin.buffer.read() if args.audio == "-" else args.audio
|
|
27
|
+
|
|
28
|
+
try:
|
|
29
|
+
result = transcribe(audio, model=args.model, language=args.language)
|
|
30
|
+
except TranscriptionError as exc:
|
|
31
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
32
|
+
return 1
|
|
33
|
+
|
|
34
|
+
print(result.text if args.text else result.to_json(indent=2))
|
|
35
|
+
return 0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
if __name__ == "__main__":
|
|
39
|
+
sys.exit(main())
|
|
File without changes
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Plain data types returned by :func:`audio_transcriber.transcribe`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from dataclasses import asdict, dataclass, field
|
|
7
|
+
from typing import Any, Dict, Tuple
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True)
|
|
11
|
+
class Segment:
|
|
12
|
+
"""A timed piece of the transcript. Times are in seconds."""
|
|
13
|
+
|
|
14
|
+
start: float
|
|
15
|
+
end: float
|
|
16
|
+
text: str
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class TranscriptionResult:
|
|
21
|
+
"""The outcome of a transcription, easy to turn into JSON."""
|
|
22
|
+
|
|
23
|
+
text: str
|
|
24
|
+
language: str
|
|
25
|
+
language_probability: float
|
|
26
|
+
duration: float
|
|
27
|
+
segments: Tuple[Segment, ...] = field(default_factory=tuple)
|
|
28
|
+
|
|
29
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
30
|
+
data = asdict(self)
|
|
31
|
+
data["segments"] = [asdict(s) for s in self.segments]
|
|
32
|
+
return data
|
|
33
|
+
|
|
34
|
+
def to_json(self, **kwargs: Any) -> str:
|
|
35
|
+
kwargs.setdefault("ensure_ascii", False)
|
|
36
|
+
return json.dumps(self.to_dict(), **kwargs)
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""The public ``transcribe`` function."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import io
|
|
6
|
+
import os
|
|
7
|
+
from typing import BinaryIO, Optional, Union
|
|
8
|
+
|
|
9
|
+
from av.error import FFmpegError
|
|
10
|
+
|
|
11
|
+
from . import _model
|
|
12
|
+
from .result import Segment, TranscriptionResult
|
|
13
|
+
|
|
14
|
+
AudioInput = Union[bytes, bytearray, memoryview, str, "os.PathLike[str]", BinaryIO]
|
|
15
|
+
|
|
16
|
+
DEFAULT_MODEL = "small"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class TranscriptionError(Exception):
|
|
20
|
+
"""Raised when the audio is empty or cannot be decoded."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _prepare_audio(audio: AudioInput) -> Union[str, BinaryIO]:
|
|
24
|
+
if isinstance(audio, (bytes, bytearray, memoryview)):
|
|
25
|
+
data = bytes(audio)
|
|
26
|
+
if not data:
|
|
27
|
+
raise TranscriptionError("Audio is empty.")
|
|
28
|
+
return io.BytesIO(data)
|
|
29
|
+
if isinstance(audio, (str, os.PathLike)):
|
|
30
|
+
return os.fspath(audio)
|
|
31
|
+
return audio
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def transcribe(
|
|
35
|
+
audio: AudioInput,
|
|
36
|
+
*,
|
|
37
|
+
model: str = DEFAULT_MODEL,
|
|
38
|
+
language: Optional[str] = None,
|
|
39
|
+
device: str = "cpu",
|
|
40
|
+
compute_type: str = "int8",
|
|
41
|
+
beam_size: int = 5,
|
|
42
|
+
vad_filter: bool = True,
|
|
43
|
+
) -> TranscriptionResult:
|
|
44
|
+
"""Convert speech to text.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
audio: Raw audio bytes (mp3, wav, webm, ogg, m4a, ...), a file path,
|
|
48
|
+
or a binary file-like object. The format is detected automatically.
|
|
49
|
+
model: Model size or name: ``tiny``, ``base``, ``small``, ``medium``,
|
|
50
|
+
``large-v3``, ... The model is downloaded on first use and cached.
|
|
51
|
+
language: ISO code such as ``"pt"`` or ``"en"``. ``None`` detects it.
|
|
52
|
+
device: ``"cpu"`` (works everywhere), ``"cuda"`` or ``"auto"``.
|
|
53
|
+
compute_type: Quantization, e.g. ``"int8"`` (light) or ``"float16"``.
|
|
54
|
+
beam_size: Higher is slightly more accurate and slower.
|
|
55
|
+
vad_filter: Skip silent parts, which also reduces hallucinated text.
|
|
56
|
+
|
|
57
|
+
Raises:
|
|
58
|
+
TranscriptionError: If the audio is empty or cannot be decoded.
|
|
59
|
+
"""
|
|
60
|
+
source = _prepare_audio(audio)
|
|
61
|
+
whisper = _model.get_model(model, device, compute_type)
|
|
62
|
+
|
|
63
|
+
try:
|
|
64
|
+
raw_segments, info = whisper.transcribe(
|
|
65
|
+
source,
|
|
66
|
+
language=language,
|
|
67
|
+
beam_size=beam_size,
|
|
68
|
+
vad_filter=vad_filter,
|
|
69
|
+
)
|
|
70
|
+
segments = tuple(
|
|
71
|
+
Segment(start=s.start, end=s.end, text=s.text.strip()) for s in raw_segments
|
|
72
|
+
)
|
|
73
|
+
except FFmpegError as exc:
|
|
74
|
+
raise TranscriptionError(f"Could not decode audio: {exc}") from exc
|
|
75
|
+
|
|
76
|
+
return TranscriptionResult(
|
|
77
|
+
text=" ".join(s.text for s in segments if s.text),
|
|
78
|
+
language=info.language,
|
|
79
|
+
language_probability=info.language_probability,
|
|
80
|
+
duration=info.duration,
|
|
81
|
+
segments=segments,
|
|
82
|
+
)
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: generic-audio-transcriber
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Drop-in speech-to-text for any Python project: audio bytes in, JSON-ready text out.
|
|
5
|
+
Author: Eduardo Milani
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/EduardoMilani8/generic-audio-transcriber
|
|
8
|
+
Keywords: speech-to-text,transcription,whisper,audio,asr
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
12
|
+
Requires-Python: >=3.9
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
License-File: LICENSE
|
|
15
|
+
Requires-Dist: faster-whisper>=1.0
|
|
16
|
+
Requires-Dist: av<19,>=11
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# generic-audio-transcriber
|
|
22
|
+
|
|
23
|
+
Drop-in speech-to-text for any Python project. Audio bytes in, JSON-ready text out.
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
from audio_transcriber import transcribe
|
|
27
|
+
|
|
28
|
+
result = transcribe(audio_bytes) # mp3, wav, webm, ogg, m4a... detected automatically
|
|
29
|
+
print(result.text) # "Hello, this is a test."
|
|
30
|
+
print(result.to_json()) # {"text": "...", "language": "en", ...}
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
No API keys, no servers, no cloud. It runs locally on the CPU, and the model is
|
|
34
|
+
downloaded automatically the first time you use it.
|
|
35
|
+
|
|
36
|
+
## Why
|
|
37
|
+
|
|
38
|
+
Many apps need the same small feature: the user sends or records audio, and the app
|
|
39
|
+
needs the text. This package is that feature as one function call, so you can add it to
|
|
40
|
+
a project without learning anything about speech recognition.
|
|
41
|
+
|
|
42
|
+
Under the hood it uses [faster-whisper](https://github.com/SYSTRAN/faster-whisper)
|
|
43
|
+
(OpenAI's Whisper model), which is accurate, supports about 99 languages, and runs well
|
|
44
|
+
on a plain CPU. You never have to train or tune a model.
|
|
45
|
+
|
|
46
|
+
## Install
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install git+https://github.com/EduardoMilani8/generic-audio-transcriber.git
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Requires Python 3.9+. You do **not** need to install ffmpeg: audio decoding is bundled.
|
|
53
|
+
|
|
54
|
+
## Usage
|
|
55
|
+
|
|
56
|
+
### From bytes (the main use case)
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
from audio_transcriber import transcribe
|
|
60
|
+
|
|
61
|
+
result = transcribe(audio_bytes)
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### From a file path or file object
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
transcribe("recording.mp3")
|
|
68
|
+
|
|
69
|
+
with open("recording.wav", "rb") as f:
|
|
70
|
+
transcribe(f)
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### Inside a web endpoint
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
from fastapi import FastAPI, UploadFile
|
|
77
|
+
from audio_transcriber import transcribe
|
|
78
|
+
|
|
79
|
+
app = FastAPI()
|
|
80
|
+
|
|
81
|
+
@app.post("/transcribe")
|
|
82
|
+
async def endpoint(file: UploadFile):
|
|
83
|
+
return transcribe(await file.read()).to_dict()
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
### The result
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
result.text # full transcript
|
|
90
|
+
result.language # detected language code, e.g. "pt"
|
|
91
|
+
result.language_probability # confidence of the language detection
|
|
92
|
+
result.duration # audio length in seconds
|
|
93
|
+
result.segments # timed pieces: .start, .end, .text
|
|
94
|
+
result.to_dict() # plain dict
|
|
95
|
+
result.to_json() # JSON string (non-ASCII kept readable)
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Silent audio gives an empty `text`, not an error.
|
|
99
|
+
|
|
100
|
+
## Options
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
transcribe(
|
|
104
|
+
audio,
|
|
105
|
+
model="small", # tiny | base | small | medium | large-v3
|
|
106
|
+
language=None, # "pt", "en", ... None = auto-detect
|
|
107
|
+
device="cpu", # "cpu" | "cuda" | "auto"
|
|
108
|
+
compute_type="int8", # "int8" is light; "float16" suits GPUs
|
|
109
|
+
beam_size=5,
|
|
110
|
+
vad_filter=True, # skip silence, reduces made-up text
|
|
111
|
+
)
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
Choosing a model is a trade-off between weight and accuracy:
|
|
115
|
+
|
|
116
|
+
| Model | Size on disk | Speed | Accuracy |
|
|
117
|
+
|---|---|---|---|
|
|
118
|
+
| `tiny` | smallest | fastest | lowest |
|
|
119
|
+
| `base` | small | fast | fair |
|
|
120
|
+
| `small` (default) | medium | good | good |
|
|
121
|
+
| `medium` / `large-v3` | large | slow on CPU | best |
|
|
122
|
+
|
|
123
|
+
The model is loaded once per process and reused, so only the first call is slow.
|
|
124
|
+
Setting `language` explicitly is faster and more accurate than auto-detection.
|
|
125
|
+
|
|
126
|
+
## Errors
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
from audio_transcriber import transcribe, TranscriptionError
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
transcribe(data)
|
|
133
|
+
except TranscriptionError as e:
|
|
134
|
+
... # empty input or audio that cannot be decoded
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## Command line
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
audio-transcriber recording.mp3 # prints JSON
|
|
141
|
+
audio-transcriber recording.mp3 --text # prints plain text
|
|
142
|
+
audio-transcriber recording.mp3 --model base --language pt
|
|
143
|
+
cat recording.mp3 | audio-transcriber - # read bytes from stdin
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
## Development
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
python -m venv .venv && source .venv/bin/activate
|
|
150
|
+
pip install -e ".[dev]"
|
|
151
|
+
pytest # fast unit tests, no model needed
|
|
152
|
+
RUN_SLOW=1 pytest -m slow # end-to-end tests, downloads the tiny model
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Notes
|
|
156
|
+
|
|
157
|
+
- The default device is `cpu` because it works on every machine. Pass `device="cuda"`
|
|
158
|
+
if you have a working CUDA setup and want GPU speed.
|
|
159
|
+
- `av` is pinned below version 19 because faster-whisper still uses an argument that
|
|
160
|
+
newer releases removed. The pin can be lifted once faster-whisper fixes it.
|
|
161
|
+
|
|
162
|
+
## License
|
|
163
|
+
|
|
164
|
+
MIT
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
src/audio_transcriber/__init__.py
|
|
5
|
+
src/audio_transcriber/_model.py
|
|
6
|
+
src/audio_transcriber/cli.py
|
|
7
|
+
src/audio_transcriber/py.typed
|
|
8
|
+
src/audio_transcriber/result.py
|
|
9
|
+
src/audio_transcriber/transcriber.py
|
|
10
|
+
src/generic_audio_transcriber.egg-info/PKG-INFO
|
|
11
|
+
src/generic_audio_transcriber.egg-info/SOURCES.txt
|
|
12
|
+
src/generic_audio_transcriber.egg-info/dependency_links.txt
|
|
13
|
+
src/generic_audio_transcriber.egg-info/entry_points.txt
|
|
14
|
+
src/generic_audio_transcriber.egg-info/requires.txt
|
|
15
|
+
src/generic_audio_transcriber.egg-info/top_level.txt
|
|
16
|
+
tests/test_cli.py
|
|
17
|
+
tests/test_integration.py
|
|
18
|
+
tests/test_model.py
|
|
19
|
+
tests/test_result.py
|
|
20
|
+
tests/test_transcribe.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
audio_transcriber
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import json
|
|
3
|
+
|
|
4
|
+
from audio_transcriber import cli
|
|
5
|
+
from audio_transcriber.result import Segment, TranscriptionResult
|
|
6
|
+
from audio_transcriber.transcriber import TranscriptionError
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def fake_result():
|
|
10
|
+
return TranscriptionResult(
|
|
11
|
+
text="hi there",
|
|
12
|
+
language="en",
|
|
13
|
+
language_probability=0.9,
|
|
14
|
+
duration=1.0,
|
|
15
|
+
segments=(Segment(0.0, 1.0, "hi there"),),
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_prints_json_by_default(monkeypatch, capsys):
|
|
20
|
+
monkeypatch.setattr(cli, "transcribe", lambda audio, **kw: fake_result())
|
|
21
|
+
assert cli.main(["a.wav"]) == 0
|
|
22
|
+
assert json.loads(capsys.readouterr().out)["text"] == "hi there"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def test_text_flag_prints_plain_text(monkeypatch, capsys):
|
|
26
|
+
monkeypatch.setattr(cli, "transcribe", lambda audio, **kw: fake_result())
|
|
27
|
+
assert cli.main(["a.wav", "--text"]) == 0
|
|
28
|
+
assert capsys.readouterr().out.strip() == "hi there"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_dash_reads_bytes_from_stdin(monkeypatch):
|
|
32
|
+
seen = {}
|
|
33
|
+
|
|
34
|
+
def fake_transcribe(audio, **kw):
|
|
35
|
+
seen["audio"] = audio
|
|
36
|
+
return fake_result()
|
|
37
|
+
|
|
38
|
+
monkeypatch.setattr(cli, "transcribe", fake_transcribe)
|
|
39
|
+
monkeypatch.setattr("sys.stdin", type("S", (), {"buffer": io.BytesIO(b"raw")})())
|
|
40
|
+
cli.main(["-"])
|
|
41
|
+
assert seen["audio"] == b"raw"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_options_are_forwarded(monkeypatch):
|
|
45
|
+
seen = {}
|
|
46
|
+
|
|
47
|
+
def fake_transcribe(audio, **kw):
|
|
48
|
+
seen.update(kw)
|
|
49
|
+
return fake_result()
|
|
50
|
+
|
|
51
|
+
monkeypatch.setattr(cli, "transcribe", fake_transcribe)
|
|
52
|
+
cli.main(["a.wav", "--model", "base", "--language", "pt"])
|
|
53
|
+
assert seen == {"model": "base", "language": "pt"}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def test_errors_exit_with_status_1(monkeypatch, capsys):
|
|
57
|
+
def boom(audio, **kw):
|
|
58
|
+
raise TranscriptionError("bad audio")
|
|
59
|
+
|
|
60
|
+
monkeypatch.setattr(cli, "transcribe", boom)
|
|
61
|
+
assert cli.main(["a.wav"]) == 1
|
|
62
|
+
assert "bad audio" in capsys.readouterr().err
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""End-to-end checks with the real model. Run with: RUN_SLOW=1 pytest -m slow"""
|
|
2
|
+
|
|
3
|
+
import io
|
|
4
|
+
import math
|
|
5
|
+
import os
|
|
6
|
+
import shutil
|
|
7
|
+
import struct
|
|
8
|
+
import subprocess
|
|
9
|
+
import wave
|
|
10
|
+
|
|
11
|
+
import pytest
|
|
12
|
+
|
|
13
|
+
from audio_transcriber import TranscriptionError, TranscriptionResult, transcribe
|
|
14
|
+
|
|
15
|
+
pytestmark = [
|
|
16
|
+
pytest.mark.slow,
|
|
17
|
+
pytest.mark.skipif(not os.environ.get("RUN_SLOW"), reason="set RUN_SLOW=1 to run"),
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def tone_wav(seconds=1.0, rate=16000, freq=440.0) -> bytes:
|
|
22
|
+
frames = b"".join(
|
|
23
|
+
struct.pack("<h", int(8000 * math.sin(2 * math.pi * freq * i / rate)))
|
|
24
|
+
for i in range(int(seconds * rate))
|
|
25
|
+
)
|
|
26
|
+
buffer = io.BytesIO()
|
|
27
|
+
with wave.open(buffer, "wb") as wav:
|
|
28
|
+
wav.setnchannels(1)
|
|
29
|
+
wav.setsampwidth(2)
|
|
30
|
+
wav.setframerate(rate)
|
|
31
|
+
wav.writeframes(frames)
|
|
32
|
+
return buffer.getvalue()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_wav_bytes_are_decoded_and_transcribed():
|
|
36
|
+
result = transcribe(tone_wav(), model="tiny", compute_type="int8")
|
|
37
|
+
assert isinstance(result, TranscriptionResult)
|
|
38
|
+
assert result.duration == pytest.approx(1.0, abs=0.1)
|
|
39
|
+
assert isinstance(result.language, str)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@pytest.mark.skipif(shutil.which("ffmpeg") is None, reason="ffmpeg needed to build an mp3")
|
|
43
|
+
def test_mp3_bytes_are_decoded_too(tmp_path):
|
|
44
|
+
wav_path = tmp_path / "tone.wav"
|
|
45
|
+
wav_path.write_bytes(tone_wav())
|
|
46
|
+
mp3 = subprocess.run(
|
|
47
|
+
["ffmpeg", "-loglevel", "error", "-i", str(wav_path), "-f", "mp3", "-"],
|
|
48
|
+
check=True,
|
|
49
|
+
capture_output=True,
|
|
50
|
+
).stdout
|
|
51
|
+
result = transcribe(mp3, model="tiny", compute_type="int8")
|
|
52
|
+
assert result.duration == pytest.approx(1.0, abs=0.2)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def test_garbage_bytes_raise_a_clear_error():
|
|
56
|
+
with pytest.raises(TranscriptionError):
|
|
57
|
+
transcribe(b"this is definitely not audio" * 100, model="tiny", compute_type="int8")
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import types
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from audio_transcriber import _model
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@pytest.fixture(autouse=True)
|
|
10
|
+
def fresh_cache():
|
|
11
|
+
_model.clear_cache()
|
|
12
|
+
yield
|
|
13
|
+
_model.clear_cache()
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@pytest.fixture
|
|
17
|
+
def fake_faster_whisper(monkeypatch):
|
|
18
|
+
created = []
|
|
19
|
+
|
|
20
|
+
class FakeWhisperModel:
|
|
21
|
+
def __init__(self, name, device, compute_type):
|
|
22
|
+
created.append((name, device, compute_type))
|
|
23
|
+
|
|
24
|
+
module = types.ModuleType("faster_whisper")
|
|
25
|
+
module.WhisperModel = FakeWhisperModel
|
|
26
|
+
monkeypatch.setitem(sys.modules, "faster_whisper", module)
|
|
27
|
+
return created
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_model_is_loaded_once_per_configuration(fake_faster_whisper):
|
|
31
|
+
first = _model.get_model("small", "auto", "int8")
|
|
32
|
+
second = _model.get_model("small", "auto", "int8")
|
|
33
|
+
assert first is second
|
|
34
|
+
assert fake_faster_whisper == [("small", "auto", "int8")]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_different_configurations_load_different_models(fake_faster_whisper):
|
|
38
|
+
small = _model.get_model("small", "auto", "int8")
|
|
39
|
+
base = _model.get_model("base", "auto", "int8")
|
|
40
|
+
assert small is not base
|
|
41
|
+
assert len(fake_faster_whisper) == 2
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_clear_cache_forces_a_reload(fake_faster_whisper):
|
|
45
|
+
_model.get_model("small", "auto", "int8")
|
|
46
|
+
_model.clear_cache()
|
|
47
|
+
_model.get_model("small", "auto", "int8")
|
|
48
|
+
assert len(fake_faster_whisper) == 2
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from audio_transcriber.result import Segment, TranscriptionResult
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def make_result():
|
|
7
|
+
return TranscriptionResult(
|
|
8
|
+
text="olá mundo",
|
|
9
|
+
language="pt",
|
|
10
|
+
language_probability=0.98,
|
|
11
|
+
duration=1.5,
|
|
12
|
+
segments=(Segment(start=0.0, end=1.5, text="olá mundo"),),
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_to_dict_is_json_serializable_and_nested():
|
|
17
|
+
data = make_result().to_dict()
|
|
18
|
+
assert data["text"] == "olá mundo"
|
|
19
|
+
assert data["segments"] == [{"start": 0.0, "end": 1.5, "text": "olá mundo"}]
|
|
20
|
+
json.dumps(data)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_to_json_keeps_non_ascii_characters():
|
|
24
|
+
raw = make_result().to_json()
|
|
25
|
+
assert "olá mundo" in raw
|
|
26
|
+
assert json.loads(raw)["language"] == "pt"
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
import io
|
|
2
|
+
from types import SimpleNamespace
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
from av.error import InvalidDataError
|
|
6
|
+
|
|
7
|
+
from audio_transcriber import _model, transcriber
|
|
8
|
+
from audio_transcriber.transcriber import TranscriptionError, transcribe
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class FakeModel:
|
|
12
|
+
def __init__(self, segments=None, error=None):
|
|
13
|
+
self.calls = []
|
|
14
|
+
self._segments = segments if segments is not None else [
|
|
15
|
+
SimpleNamespace(start=0.0, end=1.0, text=" hello"),
|
|
16
|
+
SimpleNamespace(start=1.0, end=2.0, text=" world "),
|
|
17
|
+
]
|
|
18
|
+
self._error = error
|
|
19
|
+
|
|
20
|
+
def transcribe(self, source, **kwargs):
|
|
21
|
+
self.calls.append((source, kwargs))
|
|
22
|
+
if self._error:
|
|
23
|
+
raise self._error
|
|
24
|
+
info = SimpleNamespace(language="en", language_probability=0.9, duration=2.0)
|
|
25
|
+
return iter(self._segments), info
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@pytest.fixture
|
|
29
|
+
def fake(monkeypatch):
|
|
30
|
+
model = FakeModel()
|
|
31
|
+
monkeypatch.setattr(_model, "get_model", lambda name, device, ct: model)
|
|
32
|
+
return model
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_bytes_are_wrapped_in_a_file_like_object(fake):
|
|
36
|
+
transcribe(b"fake-audio")
|
|
37
|
+
source, _ = fake.calls[0]
|
|
38
|
+
assert isinstance(source, io.BytesIO)
|
|
39
|
+
assert source.read() == b"fake-audio"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def test_path_is_passed_as_string(fake, tmp_path):
|
|
43
|
+
path = tmp_path / "a.wav"
|
|
44
|
+
transcribe(path)
|
|
45
|
+
assert fake.calls[0][0] == str(path)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_file_like_is_passed_through(fake):
|
|
49
|
+
buffer = io.BytesIO(b"x")
|
|
50
|
+
transcribe(buffer)
|
|
51
|
+
assert fake.calls[0][0] is buffer
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_result_joins_and_strips_segment_text(fake):
|
|
55
|
+
result = transcribe(b"x")
|
|
56
|
+
assert result.text == "hello world"
|
|
57
|
+
assert [s.text for s in result.segments] == ["hello", "world"]
|
|
58
|
+
assert result.language == "en"
|
|
59
|
+
assert result.duration == 2.0
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_silence_gives_empty_text(monkeypatch):
|
|
63
|
+
monkeypatch.setattr(_model, "get_model", lambda *a: FakeModel(segments=[]))
|
|
64
|
+
result = transcribe(b"x")
|
|
65
|
+
assert result.text == ""
|
|
66
|
+
assert result.segments == ()
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_options_are_forwarded(fake):
|
|
70
|
+
transcribe(b"x", language="pt", beam_size=1, vad_filter=False)
|
|
71
|
+
assert fake.calls[0][1] == {"language": "pt", "beam_size": 1, "vad_filter": False}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def test_model_options_select_the_model(monkeypatch):
|
|
75
|
+
seen = []
|
|
76
|
+
monkeypatch.setattr(
|
|
77
|
+
_model, "get_model", lambda *args: seen.append(args) or FakeModel()
|
|
78
|
+
)
|
|
79
|
+
transcribe(b"x", model="base", device="cpu", compute_type="float32")
|
|
80
|
+
assert seen == [("base", "cpu", "float32")]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_empty_bytes_raise(fake):
|
|
84
|
+
with pytest.raises(TranscriptionError, match="empty"):
|
|
85
|
+
transcribe(b"")
|
|
86
|
+
assert fake.calls == []
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_decoder_failures_become_transcription_errors(monkeypatch):
|
|
90
|
+
broken = FakeModel(error=InvalidDataError(1, "invalid data"))
|
|
91
|
+
monkeypatch.setattr(_model, "get_model", lambda *a: broken)
|
|
92
|
+
with pytest.raises(TranscriptionError, match="invalid data"):
|
|
93
|
+
transcribe(b"not audio")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def test_default_model_is_small():
|
|
97
|
+
assert transcriber.DEFAULT_MODEL == "small"
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def test_non_decoding_failures_are_not_disguised(monkeypatch):
|
|
101
|
+
broken = FakeModel(error=RuntimeError("cuda library missing"))
|
|
102
|
+
monkeypatch.setattr(_model, "get_model", lambda *a: broken)
|
|
103
|
+
with pytest.raises(RuntimeError, match="cuda library missing"):
|
|
104
|
+
transcribe(b"x")
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def test_device_defaults_to_cpu(monkeypatch):
|
|
108
|
+
seen = []
|
|
109
|
+
monkeypatch.setattr(_model, "get_model", lambda *args: seen.append(args) or FakeModel())
|
|
110
|
+
transcribe(b"x")
|
|
111
|
+
assert seen == [("small", "cpu", "int8")]
|