generic-audio-transcriber 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. generic_audio_transcriber-0.1.0/LICENSE +21 -0
  2. generic_audio_transcriber-0.1.0/PKG-INFO +164 -0
  3. generic_audio_transcriber-0.1.0/README.md +144 -0
  4. generic_audio_transcriber-0.1.0/pyproject.toml +39 -0
  5. generic_audio_transcriber-0.1.0/setup.cfg +4 -0
  6. generic_audio_transcriber-0.1.0/src/audio_transcriber/__init__.py +6 -0
  7. generic_audio_transcriber-0.1.0/src/audio_transcriber/_model.py +32 -0
  8. generic_audio_transcriber-0.1.0/src/audio_transcriber/cli.py +39 -0
  9. generic_audio_transcriber-0.1.0/src/audio_transcriber/py.typed +0 -0
  10. generic_audio_transcriber-0.1.0/src/audio_transcriber/result.py +36 -0
  11. generic_audio_transcriber-0.1.0/src/audio_transcriber/transcriber.py +82 -0
  12. generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/PKG-INFO +164 -0
  13. generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/SOURCES.txt +20 -0
  14. generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/dependency_links.txt +1 -0
  15. generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/entry_points.txt +2 -0
  16. generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/requires.txt +5 -0
  17. generic_audio_transcriber-0.1.0/src/generic_audio_transcriber.egg-info/top_level.txt +1 -0
  18. generic_audio_transcriber-0.1.0/tests/test_cli.py +62 -0
  19. generic_audio_transcriber-0.1.0/tests/test_integration.py +57 -0
  20. generic_audio_transcriber-0.1.0/tests/test_model.py +48 -0
  21. generic_audio_transcriber-0.1.0/tests/test_result.py +26 -0
  22. generic_audio_transcriber-0.1.0/tests/test_transcribe.py +111 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Eduardo Milani
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,164 @@
1
+ Metadata-Version: 2.4
2
+ Name: generic-audio-transcriber
3
+ Version: 0.1.0
4
+ Summary: Drop-in speech-to-text for any Python project: audio bytes in, JSON-ready text out.
5
+ Author: Eduardo Milani
6
+ License: MIT
7
+ Project-URL: Repository, https://github.com/EduardoMilani8/generic-audio-transcriber
8
+ Keywords: speech-to-text,transcription,whisper,audio,asr
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
12
+ Requires-Python: >=3.9
13
+ Description-Content-Type: text/markdown
14
+ License-File: LICENSE
15
+ Requires-Dist: faster-whisper>=1.0
16
+ Requires-Dist: av<19,>=11
17
+ Provides-Extra: dev
18
+ Requires-Dist: pytest>=8; extra == "dev"
19
+ Dynamic: license-file
20
+
21
+ # generic-audio-transcriber
22
+
23
+ Drop-in speech-to-text for any Python project. Audio bytes in, JSON-ready text out.
24
+
25
+ ```python
26
+ from audio_transcriber import transcribe
27
+
28
+ result = transcribe(audio_bytes) # mp3, wav, webm, ogg, m4a... detected automatically
29
+ print(result.text) # "Hello, this is a test."
30
+ print(result.to_json()) # {"text": "...", "language": "en", ...}
31
+ ```
32
+
33
+ No API keys, no servers, no cloud. It runs locally on the CPU, and the model is
34
+ downloaded automatically the first time you use it.
35
+
36
+ ## Why
37
+
38
+ Many apps need the same small feature: the user sends or records audio, and the app
39
+ needs the text. This package is that feature as one function call, so you can add it to
40
+ a project without learning anything about speech recognition.
41
+
42
+ Under the hood it uses [faster-whisper](https://github.com/SYSTRAN/faster-whisper)
43
+ (OpenAI's Whisper model), which is accurate, supports about 99 languages, and runs well
44
+ on a plain CPU. You never have to train or tune a model.
45
+
46
+ ## Install
47
+
48
+ ```bash
49
+ pip install git+https://github.com/EduardoMilani8/generic-audio-transcriber.git
50
+ ```
51
+
52
+ Requires Python 3.9+. You do **not** need to install ffmpeg: audio decoding is bundled.
53
+
54
+ ## Usage
55
+
56
+ ### From bytes (the main use case)
57
+
58
+ ```python
59
+ from audio_transcriber import transcribe
60
+
61
+ result = transcribe(audio_bytes)
62
+ ```
63
+
64
+ ### From a file path or file object
65
+
66
+ ```python
67
+ transcribe("recording.mp3")
68
+
69
+ with open("recording.wav", "rb") as f:
70
+ transcribe(f)
71
+ ```
72
+
73
+ ### Inside a web endpoint
74
+
75
+ ```python
76
+ from fastapi import FastAPI, UploadFile
77
+ from audio_transcriber import transcribe
78
+
79
+ app = FastAPI()
80
+
81
+ @app.post("/transcribe")
82
+ async def endpoint(file: UploadFile):
83
+ return transcribe(await file.read()).to_dict()
84
+ ```
85
+
86
+ ### The result
87
+
88
+ ```python
89
+ result.text # full transcript
90
+ result.language # detected language code, e.g. "pt"
91
+ result.language_probability # confidence of the language detection
92
+ result.duration # audio length in seconds
93
+ result.segments # timed pieces: .start, .end, .text
94
+ result.to_dict() # plain dict
95
+ result.to_json() # JSON string (non-ASCII kept readable)
96
+ ```
97
+
98
+ Silent audio gives an empty `text`, not an error.
99
+
100
+ ## Options
101
+
102
+ ```python
103
+ transcribe(
104
+ audio,
105
+ model="small", # tiny | base | small | medium | large-v3
106
+ language=None, # "pt", "en", ... None = auto-detect
107
+ device="cpu", # "cpu" | "cuda" | "auto"
108
+ compute_type="int8", # "int8" is light; "float16" suits GPUs
109
+ beam_size=5,
110
+ vad_filter=True, # skip silence, reduces made-up text
111
+ )
112
+ ```
113
+
114
+ Choosing a model is a trade-off between weight and accuracy:
115
+
116
+ | Model | Size on disk | Speed | Accuracy |
117
+ |---|---|---|---|
118
+ | `tiny` | smallest | fastest | lowest |
119
+ | `base` | small | fast | fair |
120
+ | `small` (default) | medium | good | good |
121
+ | `medium` / `large-v3` | large | slow on CPU | best |
122
+
123
+ The model is loaded once per process and reused, so only the first call is slow.
124
+ Setting `language` explicitly is faster and more accurate than auto-detection.
125
+
126
+ ## Errors
127
+
128
+ ```python
129
+ from audio_transcriber import transcribe, TranscriptionError
130
+
131
+ try:
132
+ transcribe(data)
133
+ except TranscriptionError as e:
134
+ ... # empty input or audio that cannot be decoded
135
+ ```
136
+
137
+ ## Command line
138
+
139
+ ```bash
140
+ audio-transcriber recording.mp3 # prints JSON
141
+ audio-transcriber recording.mp3 --text # prints plain text
142
+ audio-transcriber recording.mp3 --model base --language pt
143
+ cat recording.mp3 | audio-transcriber - # read bytes from stdin
144
+ ```
145
+
146
+ ## Development
147
+
148
+ ```bash
149
+ python -m venv .venv && source .venv/bin/activate
150
+ pip install -e ".[dev]"
151
+ pytest # fast unit tests, no model needed
152
+ RUN_SLOW=1 pytest -m slow # end-to-end tests, downloads the tiny model
153
+ ```
154
+
155
+ ## Notes
156
+
157
+ - The default device is `cpu` because it works on every machine. Pass `device="cuda"`
158
+ if you have a working CUDA setup and want GPU speed.
159
+ - `av` is pinned below version 19 because faster-whisper still uses an argument that
160
+ newer releases removed. The pin can be lifted once faster-whisper fixes it.
161
+
162
+ ## License
163
+
164
+ MIT
@@ -0,0 +1,144 @@
1
+ # generic-audio-transcriber
2
+
3
+ Drop-in speech-to-text for any Python project. Audio bytes in, JSON-ready text out.
4
+
5
+ ```python
6
+ from audio_transcriber import transcribe
7
+
8
+ result = transcribe(audio_bytes) # mp3, wav, webm, ogg, m4a... detected automatically
9
+ print(result.text) # "Hello, this is a test."
10
+ print(result.to_json()) # {"text": "...", "language": "en", ...}
11
+ ```
12
+
13
+ No API keys, no servers, no cloud. It runs locally on the CPU, and the model is
14
+ downloaded automatically the first time you use it.
15
+
16
+ ## Why
17
+
18
+ Many apps need the same small feature: the user sends or records audio, and the app
19
+ needs the text. This package is that feature as one function call, so you can add it to
20
+ a project without learning anything about speech recognition.
21
+
22
+ Under the hood it uses [faster-whisper](https://github.com/SYSTRAN/faster-whisper)
23
+ (OpenAI's Whisper model), which is accurate, supports about 99 languages, and runs well
24
+ on a plain CPU. You never have to train or tune a model.
25
+
26
+ ## Install
27
+
28
+ ```bash
29
+ pip install git+https://github.com/EduardoMilani8/generic-audio-transcriber.git
30
+ ```
31
+
32
+ Requires Python 3.9+. You do **not** need to install ffmpeg: audio decoding is bundled.
33
+
34
+ ## Usage
35
+
36
+ ### From bytes (the main use case)
37
+
38
+ ```python
39
+ from audio_transcriber import transcribe
40
+
41
+ result = transcribe(audio_bytes)
42
+ ```
43
+
44
+ ### From a file path or file object
45
+
46
+ ```python
47
+ transcribe("recording.mp3")
48
+
49
+ with open("recording.wav", "rb") as f:
50
+ transcribe(f)
51
+ ```
52
+
53
+ ### Inside a web endpoint
54
+
55
+ ```python
56
+ from fastapi import FastAPI, UploadFile
57
+ from audio_transcriber import transcribe
58
+
59
+ app = FastAPI()
60
+
61
+ @app.post("/transcribe")
62
+ async def endpoint(file: UploadFile):
63
+ return transcribe(await file.read()).to_dict()
64
+ ```
65
+
66
+ ### The result
67
+
68
+ ```python
69
+ result.text # full transcript
70
+ result.language # detected language code, e.g. "pt"
71
+ result.language_probability # confidence of the language detection
72
+ result.duration # audio length in seconds
73
+ result.segments # timed pieces: .start, .end, .text
74
+ result.to_dict() # plain dict
75
+ result.to_json() # JSON string (non-ASCII kept readable)
76
+ ```
77
+
78
+ Silent audio gives an empty `text`, not an error.
79
+
80
+ ## Options
81
+
82
+ ```python
83
+ transcribe(
84
+ audio,
85
+ model="small", # tiny | base | small | medium | large-v3
86
+ language=None, # "pt", "en", ... None = auto-detect
87
+ device="cpu", # "cpu" | "cuda" | "auto"
88
+ compute_type="int8", # "int8" is light; "float16" suits GPUs
89
+ beam_size=5,
90
+ vad_filter=True, # skip silence, reduces made-up text
91
+ )
92
+ ```
93
+
94
+ Choosing a model is a trade-off between weight and accuracy:
95
+
96
+ | Model | Size on disk | Speed | Accuracy |
97
+ |---|---|---|---|
98
+ | `tiny` | smallest | fastest | lowest |
99
+ | `base` | small | fast | fair |
100
+ | `small` (default) | medium | good | good |
101
+ | `medium` / `large-v3` | large | slow on CPU | best |
102
+
103
+ The model is loaded once per process and reused, so only the first call is slow.
104
+ Setting `language` explicitly is faster and more accurate than auto-detection.
105
+
106
+ ## Errors
107
+
108
+ ```python
109
+ from audio_transcriber import transcribe, TranscriptionError
110
+
111
+ try:
112
+ transcribe(data)
113
+ except TranscriptionError as e:
114
+ ... # empty input or audio that cannot be decoded
115
+ ```
116
+
117
+ ## Command line
118
+
119
+ ```bash
120
+ audio-transcriber recording.mp3 # prints JSON
121
+ audio-transcriber recording.mp3 --text # prints plain text
122
+ audio-transcriber recording.mp3 --model base --language pt
123
+ cat recording.mp3 | audio-transcriber - # read bytes from stdin
124
+ ```
125
+
126
+ ## Development
127
+
128
+ ```bash
129
+ python -m venv .venv && source .venv/bin/activate
130
+ pip install -e ".[dev]"
131
+ pytest # fast unit tests, no model needed
132
+ RUN_SLOW=1 pytest -m slow # end-to-end tests, downloads the tiny model
133
+ ```
134
+
135
+ ## Notes
136
+
137
+ - The default device is `cpu` because it works on every machine. Pass `device="cuda"`
138
+ if you have a working CUDA setup and want GPU speed.
139
+ - `av` is pinned below version 19 because faster-whisper still uses an argument that
140
+ newer releases removed. The pin can be lifted once faster-whisper fixes it.
141
+
142
+ ## License
143
+
144
+ MIT
@@ -0,0 +1,39 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "generic-audio-transcriber"
7
+ version = "0.1.0"
8
+ description = "Drop-in speech-to-text for any Python project: audio bytes in, JSON-ready text out."
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ requires-python = ">=3.9"
12
+ authors = [{ name = "Eduardo Milani" }]
13
+ keywords = ["speech-to-text", "transcription", "whisper", "audio", "asr"]
14
+ classifiers = [
15
+ "License :: OSI Approved :: MIT License",
16
+ "Programming Language :: Python :: 3",
17
+ "Topic :: Multimedia :: Sound/Audio :: Speech",
18
+ ]
19
+ # av<19: faster-whisper still calls av.open(metadata_errors=...), which av 19 rejects.
20
+ dependencies = ["faster-whisper>=1.0", "av>=11,<19"]
21
+
22
+ [project.optional-dependencies]
23
+ dev = ["pytest>=8"]
24
+
25
+ [project.scripts]
26
+ audio-transcriber = "audio_transcriber.cli:main"
27
+
28
+ [project.urls]
29
+ Repository = "https://github.com/EduardoMilani8/generic-audio-transcriber"
30
+
31
+ [tool.setuptools.packages.find]
32
+ where = ["src"]
33
+
34
+ [tool.setuptools.package-data]
35
+ audio_transcriber = ["py.typed"]
36
+
37
+ [tool.pytest.ini_options]
38
+ testpaths = ["tests"]
39
+ markers = ["slow: downloads a real model and runs real inference (set RUN_SLOW=1)"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,6 @@
1
+ """Drop-in speech-to-text for any Python project."""
2
+
3
+ from .result import Segment, TranscriptionResult
4
+ from .transcriber import TranscriptionError, transcribe
5
+
6
+ __all__ = ["transcribe", "TranscriptionResult", "Segment", "TranscriptionError"]
@@ -0,0 +1,32 @@
1
+ """Lazy, cached loading of the speech recognition model."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import threading
6
+ from typing import Any, Dict, Tuple
7
+
8
+ _cache: Dict[Tuple[str, str, str], Any] = {}
9
+ _lock = threading.Lock()
10
+
11
+
12
+ def get_model(name: str, device: str, compute_type: str) -> Any:
13
+ """Return a loaded model, loading (and downloading) it only the first time.
14
+
15
+ ``faster_whisper`` is imported here rather than at module level so that
16
+ importing this package stays instant and tests can run without it.
17
+ """
18
+ key = (name, device, compute_type)
19
+ with _lock:
20
+ model = _cache.get(key)
21
+ if model is None:
22
+ from faster_whisper import WhisperModel
23
+
24
+ model = WhisperModel(name, device=device, compute_type=compute_type)
25
+ _cache[key] = model
26
+ return model
27
+
28
+
29
+ def clear_cache() -> None:
30
+ """Drop every loaded model so its memory can be reclaimed."""
31
+ with _lock:
32
+ _cache.clear()
@@ -0,0 +1,39 @@
1
+ """Command line entry point: ``audio-transcriber FILE`` prints JSON."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from typing import List, Optional
8
+
9
+ from .transcriber import DEFAULT_MODEL, TranscriptionError, transcribe
10
+
11
+
12
+ def build_parser() -> argparse.ArgumentParser:
13
+ parser = argparse.ArgumentParser(
14
+ prog="audio-transcriber",
15
+ description="Transcribe an audio file to text and print the result as JSON.",
16
+ )
17
+ parser.add_argument("audio", help="Path to an audio file, or '-' to read bytes from stdin.")
18
+ parser.add_argument("--model", default=DEFAULT_MODEL, help=f"Model size (default: {DEFAULT_MODEL}).")
19
+ parser.add_argument("--language", default=None, help="Language code such as 'pt'. Auto-detected if omitted.")
20
+ parser.add_argument("--text", action="store_true", help="Print only the plain text instead of JSON.")
21
+ return parser
22
+
23
+
24
+ def main(argv: Optional[List[str]] = None) -> int:
25
+ args = build_parser().parse_args(argv)
26
+ audio = sys.stdin.buffer.read() if args.audio == "-" else args.audio
27
+
28
+ try:
29
+ result = transcribe(audio, model=args.model, language=args.language)
30
+ except TranscriptionError as exc:
31
+ print(f"error: {exc}", file=sys.stderr)
32
+ return 1
33
+
34
+ print(result.text if args.text else result.to_json(indent=2))
35
+ return 0
36
+
37
+
38
+ if __name__ == "__main__":
39
+ sys.exit(main())
@@ -0,0 +1,36 @@
1
+ """Plain data types returned by :func:`audio_transcriber.transcribe`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from dataclasses import asdict, dataclass, field
7
+ from typing import Any, Dict, Tuple
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class Segment:
12
+ """A timed piece of the transcript. Times are in seconds."""
13
+
14
+ start: float
15
+ end: float
16
+ text: str
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class TranscriptionResult:
21
+ """The outcome of a transcription, easy to turn into JSON."""
22
+
23
+ text: str
24
+ language: str
25
+ language_probability: float
26
+ duration: float
27
+ segments: Tuple[Segment, ...] = field(default_factory=tuple)
28
+
29
+ def to_dict(self) -> Dict[str, Any]:
30
+ data = asdict(self)
31
+ data["segments"] = [asdict(s) for s in self.segments]
32
+ return data
33
+
34
+ def to_json(self, **kwargs: Any) -> str:
35
+ kwargs.setdefault("ensure_ascii", False)
36
+ return json.dumps(self.to_dict(), **kwargs)
@@ -0,0 +1,82 @@
1
+ """The public ``transcribe`` function."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import io
6
+ import os
7
+ from typing import BinaryIO, Optional, Union
8
+
9
+ from av.error import FFmpegError
10
+
11
+ from . import _model
12
+ from .result import Segment, TranscriptionResult
13
+
14
+ AudioInput = Union[bytes, bytearray, memoryview, str, "os.PathLike[str]", BinaryIO]
15
+
16
+ DEFAULT_MODEL = "small"
17
+
18
+
19
+ class TranscriptionError(Exception):
20
+ """Raised when the audio is empty or cannot be decoded."""
21
+
22
+
23
+ def _prepare_audio(audio: AudioInput) -> Union[str, BinaryIO]:
24
+ if isinstance(audio, (bytes, bytearray, memoryview)):
25
+ data = bytes(audio)
26
+ if not data:
27
+ raise TranscriptionError("Audio is empty.")
28
+ return io.BytesIO(data)
29
+ if isinstance(audio, (str, os.PathLike)):
30
+ return os.fspath(audio)
31
+ return audio
32
+
33
+
34
+ def transcribe(
35
+ audio: AudioInput,
36
+ *,
37
+ model: str = DEFAULT_MODEL,
38
+ language: Optional[str] = None,
39
+ device: str = "cpu",
40
+ compute_type: str = "int8",
41
+ beam_size: int = 5,
42
+ vad_filter: bool = True,
43
+ ) -> TranscriptionResult:
44
+ """Convert speech to text.
45
+
46
+ Args:
47
+ audio: Raw audio bytes (mp3, wav, webm, ogg, m4a, ...), a file path,
48
+ or a binary file-like object. The format is detected automatically.
49
+ model: Model size or name: ``tiny``, ``base``, ``small``, ``medium``,
50
+ ``large-v3``, ... The model is downloaded on first use and cached.
51
+ language: ISO code such as ``"pt"`` or ``"en"``. ``None`` detects it.
52
+ device: ``"cpu"`` (works everywhere), ``"cuda"`` or ``"auto"``.
53
+ compute_type: Quantization, e.g. ``"int8"`` (light) or ``"float16"``.
54
+ beam_size: Higher is slightly more accurate and slower.
55
+ vad_filter: Skip silent parts, which also reduces hallucinated text.
56
+
57
+ Raises:
58
+ TranscriptionError: If the audio is empty or cannot be decoded.
59
+ """
60
+ source = _prepare_audio(audio)
61
+ whisper = _model.get_model(model, device, compute_type)
62
+
63
+ try:
64
+ raw_segments, info = whisper.transcribe(
65
+ source,
66
+ language=language,
67
+ beam_size=beam_size,
68
+ vad_filter=vad_filter,
69
+ )
70
+ segments = tuple(
71
+ Segment(start=s.start, end=s.end, text=s.text.strip()) for s in raw_segments
72
+ )
73
+ except FFmpegError as exc:
74
+ raise TranscriptionError(f"Could not decode audio: {exc}") from exc
75
+
76
+ return TranscriptionResult(
77
+ text=" ".join(s.text for s in segments if s.text),
78
+ language=info.language,
79
+ language_probability=info.language_probability,
80
+ duration=info.duration,
81
+ segments=segments,
82
+ )
@@ -0,0 +1,164 @@
1
+ Metadata-Version: 2.4
2
+ Name: generic-audio-transcriber
3
+ Version: 0.1.0
4
+ Summary: Drop-in speech-to-text for any Python project: audio bytes in, JSON-ready text out.
5
+ Author: Eduardo Milani
6
+ License: MIT
7
+ Project-URL: Repository, https://github.com/EduardoMilani8/generic-audio-transcriber
8
+ Keywords: speech-to-text,transcription,whisper,audio,asr
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
12
+ Requires-Python: >=3.9
13
+ Description-Content-Type: text/markdown
14
+ License-File: LICENSE
15
+ Requires-Dist: faster-whisper>=1.0
16
+ Requires-Dist: av<19,>=11
17
+ Provides-Extra: dev
18
+ Requires-Dist: pytest>=8; extra == "dev"
19
+ Dynamic: license-file
20
+
21
+ # generic-audio-transcriber
22
+
23
+ Drop-in speech-to-text for any Python project. Audio bytes in, JSON-ready text out.
24
+
25
+ ```python
26
+ from audio_transcriber import transcribe
27
+
28
+ result = transcribe(audio_bytes) # mp3, wav, webm, ogg, m4a... detected automatically
29
+ print(result.text) # "Hello, this is a test."
30
+ print(result.to_json()) # {"text": "...", "language": "en", ...}
31
+ ```
32
+
33
+ No API keys, no servers, no cloud. It runs locally on the CPU, and the model is
34
+ downloaded automatically the first time you use it.
35
+
36
+ ## Why
37
+
38
+ Many apps need the same small feature: the user sends or records audio, and the app
39
+ needs the text. This package is that feature as one function call, so you can add it to
40
+ a project without learning anything about speech recognition.
41
+
42
+ Under the hood it uses [faster-whisper](https://github.com/SYSTRAN/faster-whisper)
43
+ (OpenAI's Whisper model), which is accurate, supports about 99 languages, and runs well
44
+ on a plain CPU. You never have to train or tune a model.
45
+
46
+ ## Install
47
+
48
+ ```bash
49
+ pip install git+https://github.com/EduardoMilani8/generic-audio-transcriber.git
50
+ ```
51
+
52
+ Requires Python 3.9+. You do **not** need to install ffmpeg: audio decoding is bundled.
53
+
54
+ ## Usage
55
+
56
+ ### From bytes (the main use case)
57
+
58
+ ```python
59
+ from audio_transcriber import transcribe
60
+
61
+ result = transcribe(audio_bytes)
62
+ ```
63
+
64
+ ### From a file path or file object
65
+
66
+ ```python
67
+ transcribe("recording.mp3")
68
+
69
+ with open("recording.wav", "rb") as f:
70
+ transcribe(f)
71
+ ```
72
+
73
+ ### Inside a web endpoint
74
+
75
+ ```python
76
+ from fastapi import FastAPI, UploadFile
77
+ from audio_transcriber import transcribe
78
+
79
+ app = FastAPI()
80
+
81
+ @app.post("/transcribe")
82
+ async def endpoint(file: UploadFile):
83
+ return transcribe(await file.read()).to_dict()
84
+ ```
85
+
86
+ ### The result
87
+
88
+ ```python
89
+ result.text # full transcript
90
+ result.language # detected language code, e.g. "pt"
91
+ result.language_probability # confidence of the language detection
92
+ result.duration # audio length in seconds
93
+ result.segments # timed pieces: .start, .end, .text
94
+ result.to_dict() # plain dict
95
+ result.to_json() # JSON string (non-ASCII kept readable)
96
+ ```
97
+
98
+ Silent audio gives an empty `text`, not an error.
99
+
100
+ ## Options
101
+
102
+ ```python
103
+ transcribe(
104
+ audio,
105
+ model="small", # tiny | base | small | medium | large-v3
106
+ language=None, # "pt", "en", ... None = auto-detect
107
+ device="cpu", # "cpu" | "cuda" | "auto"
108
+ compute_type="int8", # "int8" is light; "float16" suits GPUs
109
+ beam_size=5,
110
+ vad_filter=True, # skip silence, reduces made-up text
111
+ )
112
+ ```
113
+
114
+ Choosing a model is a trade-off between weight and accuracy:
115
+
116
+ | Model | Size on disk | Speed | Accuracy |
117
+ |---|---|---|---|
118
+ | `tiny` | smallest | fastest | lowest |
119
+ | `base` | small | fast | fair |
120
+ | `small` (default) | medium | good | good |
121
+ | `medium` / `large-v3` | large | slow on CPU | best |
122
+
123
+ The model is loaded once per process and reused, so only the first call is slow.
124
+ Setting `language` explicitly is faster and more accurate than auto-detection.
125
+
126
+ ## Errors
127
+
128
+ ```python
129
+ from audio_transcriber import transcribe, TranscriptionError
130
+
131
+ try:
132
+ transcribe(data)
133
+ except TranscriptionError as e:
134
+ ... # empty input or audio that cannot be decoded
135
+ ```
136
+
137
+ ## Command line
138
+
139
+ ```bash
140
+ audio-transcriber recording.mp3 # prints JSON
141
+ audio-transcriber recording.mp3 --text # prints plain text
142
+ audio-transcriber recording.mp3 --model base --language pt
143
+ cat recording.mp3 | audio-transcriber - # read bytes from stdin
144
+ ```
145
+
146
+ ## Development
147
+
148
+ ```bash
149
+ python -m venv .venv && source .venv/bin/activate
150
+ pip install -e ".[dev]"
151
+ pytest # fast unit tests, no model needed
152
+ RUN_SLOW=1 pytest -m slow # end-to-end tests, downloads the tiny model
153
+ ```
154
+
155
+ ## Notes
156
+
157
+ - The default device is `cpu` because it works on every machine. Pass `device="cuda"`
158
+ if you have a working CUDA setup and want GPU speed.
159
+ - `av` is pinned below version 19 because faster-whisper still uses an argument that
160
+ newer releases removed. The pin can be lifted once faster-whisper fixes it.
161
+
162
+ ## License
163
+
164
+ MIT
@@ -0,0 +1,20 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ src/audio_transcriber/__init__.py
5
+ src/audio_transcriber/_model.py
6
+ src/audio_transcriber/cli.py
7
+ src/audio_transcriber/py.typed
8
+ src/audio_transcriber/result.py
9
+ src/audio_transcriber/transcriber.py
10
+ src/generic_audio_transcriber.egg-info/PKG-INFO
11
+ src/generic_audio_transcriber.egg-info/SOURCES.txt
12
+ src/generic_audio_transcriber.egg-info/dependency_links.txt
13
+ src/generic_audio_transcriber.egg-info/entry_points.txt
14
+ src/generic_audio_transcriber.egg-info/requires.txt
15
+ src/generic_audio_transcriber.egg-info/top_level.txt
16
+ tests/test_cli.py
17
+ tests/test_integration.py
18
+ tests/test_model.py
19
+ tests/test_result.py
20
+ tests/test_transcribe.py
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ audio-transcriber = audio_transcriber.cli:main
@@ -0,0 +1,5 @@
1
+ faster-whisper>=1.0
2
+ av<19,>=11
3
+
4
+ [dev]
5
+ pytest>=8
@@ -0,0 +1,62 @@
1
+ import io
2
+ import json
3
+
4
+ from audio_transcriber import cli
5
+ from audio_transcriber.result import Segment, TranscriptionResult
6
+ from audio_transcriber.transcriber import TranscriptionError
7
+
8
+
9
+ def fake_result():
10
+ return TranscriptionResult(
11
+ text="hi there",
12
+ language="en",
13
+ language_probability=0.9,
14
+ duration=1.0,
15
+ segments=(Segment(0.0, 1.0, "hi there"),),
16
+ )
17
+
18
+
19
+ def test_prints_json_by_default(monkeypatch, capsys):
20
+ monkeypatch.setattr(cli, "transcribe", lambda audio, **kw: fake_result())
21
+ assert cli.main(["a.wav"]) == 0
22
+ assert json.loads(capsys.readouterr().out)["text"] == "hi there"
23
+
24
+
25
+ def test_text_flag_prints_plain_text(monkeypatch, capsys):
26
+ monkeypatch.setattr(cli, "transcribe", lambda audio, **kw: fake_result())
27
+ assert cli.main(["a.wav", "--text"]) == 0
28
+ assert capsys.readouterr().out.strip() == "hi there"
29
+
30
+
31
+ def test_dash_reads_bytes_from_stdin(monkeypatch):
32
+ seen = {}
33
+
34
+ def fake_transcribe(audio, **kw):
35
+ seen["audio"] = audio
36
+ return fake_result()
37
+
38
+ monkeypatch.setattr(cli, "transcribe", fake_transcribe)
39
+ monkeypatch.setattr("sys.stdin", type("S", (), {"buffer": io.BytesIO(b"raw")})())
40
+ cli.main(["-"])
41
+ assert seen["audio"] == b"raw"
42
+
43
+
44
+ def test_options_are_forwarded(monkeypatch):
45
+ seen = {}
46
+
47
+ def fake_transcribe(audio, **kw):
48
+ seen.update(kw)
49
+ return fake_result()
50
+
51
+ monkeypatch.setattr(cli, "transcribe", fake_transcribe)
52
+ cli.main(["a.wav", "--model", "base", "--language", "pt"])
53
+ assert seen == {"model": "base", "language": "pt"}
54
+
55
+
56
+ def test_errors_exit_with_status_1(monkeypatch, capsys):
57
+ def boom(audio, **kw):
58
+ raise TranscriptionError("bad audio")
59
+
60
+ monkeypatch.setattr(cli, "transcribe", boom)
61
+ assert cli.main(["a.wav"]) == 1
62
+ assert "bad audio" in capsys.readouterr().err
@@ -0,0 +1,57 @@
1
+ """End-to-end checks with the real model. Run with: RUN_SLOW=1 pytest -m slow"""
2
+
3
+ import io
4
+ import math
5
+ import os
6
+ import shutil
7
+ import struct
8
+ import subprocess
9
+ import wave
10
+
11
+ import pytest
12
+
13
+ from audio_transcriber import TranscriptionError, TranscriptionResult, transcribe
14
+
15
+ pytestmark = [
16
+ pytest.mark.slow,
17
+ pytest.mark.skipif(not os.environ.get("RUN_SLOW"), reason="set RUN_SLOW=1 to run"),
18
+ ]
19
+
20
+
21
+ def tone_wav(seconds=1.0, rate=16000, freq=440.0) -> bytes:
22
+ frames = b"".join(
23
+ struct.pack("<h", int(8000 * math.sin(2 * math.pi * freq * i / rate)))
24
+ for i in range(int(seconds * rate))
25
+ )
26
+ buffer = io.BytesIO()
27
+ with wave.open(buffer, "wb") as wav:
28
+ wav.setnchannels(1)
29
+ wav.setsampwidth(2)
30
+ wav.setframerate(rate)
31
+ wav.writeframes(frames)
32
+ return buffer.getvalue()
33
+
34
+
35
+ def test_wav_bytes_are_decoded_and_transcribed():
36
+ result = transcribe(tone_wav(), model="tiny", compute_type="int8")
37
+ assert isinstance(result, TranscriptionResult)
38
+ assert result.duration == pytest.approx(1.0, abs=0.1)
39
+ assert isinstance(result.language, str)
40
+
41
+
42
+ @pytest.mark.skipif(shutil.which("ffmpeg") is None, reason="ffmpeg needed to build an mp3")
43
+ def test_mp3_bytes_are_decoded_too(tmp_path):
44
+ wav_path = tmp_path / "tone.wav"
45
+ wav_path.write_bytes(tone_wav())
46
+ mp3 = subprocess.run(
47
+ ["ffmpeg", "-loglevel", "error", "-i", str(wav_path), "-f", "mp3", "-"],
48
+ check=True,
49
+ capture_output=True,
50
+ ).stdout
51
+ result = transcribe(mp3, model="tiny", compute_type="int8")
52
+ assert result.duration == pytest.approx(1.0, abs=0.2)
53
+
54
+
55
+ def test_garbage_bytes_raise_a_clear_error():
56
+ with pytest.raises(TranscriptionError):
57
+ transcribe(b"this is definitely not audio" * 100, model="tiny", compute_type="int8")
@@ -0,0 +1,48 @@
1
+ import sys
2
+ import types
3
+
4
+ import pytest
5
+
6
+ from audio_transcriber import _model
7
+
8
+
9
+ @pytest.fixture(autouse=True)
10
+ def fresh_cache():
11
+ _model.clear_cache()
12
+ yield
13
+ _model.clear_cache()
14
+
15
+
16
+ @pytest.fixture
17
+ def fake_faster_whisper(monkeypatch):
18
+ created = []
19
+
20
+ class FakeWhisperModel:
21
+ def __init__(self, name, device, compute_type):
22
+ created.append((name, device, compute_type))
23
+
24
+ module = types.ModuleType("faster_whisper")
25
+ module.WhisperModel = FakeWhisperModel
26
+ monkeypatch.setitem(sys.modules, "faster_whisper", module)
27
+ return created
28
+
29
+
30
+ def test_model_is_loaded_once_per_configuration(fake_faster_whisper):
31
+ first = _model.get_model("small", "auto", "int8")
32
+ second = _model.get_model("small", "auto", "int8")
33
+ assert first is second
34
+ assert fake_faster_whisper == [("small", "auto", "int8")]
35
+
36
+
37
+ def test_different_configurations_load_different_models(fake_faster_whisper):
38
+ small = _model.get_model("small", "auto", "int8")
39
+ base = _model.get_model("base", "auto", "int8")
40
+ assert small is not base
41
+ assert len(fake_faster_whisper) == 2
42
+
43
+
44
+ def test_clear_cache_forces_a_reload(fake_faster_whisper):
45
+ _model.get_model("small", "auto", "int8")
46
+ _model.clear_cache()
47
+ _model.get_model("small", "auto", "int8")
48
+ assert len(fake_faster_whisper) == 2
@@ -0,0 +1,26 @@
1
+ import json
2
+
3
+ from audio_transcriber.result import Segment, TranscriptionResult
4
+
5
+
6
+ def make_result():
7
+ return TranscriptionResult(
8
+ text="olá mundo",
9
+ language="pt",
10
+ language_probability=0.98,
11
+ duration=1.5,
12
+ segments=(Segment(start=0.0, end=1.5, text="olá mundo"),),
13
+ )
14
+
15
+
16
+ def test_to_dict_is_json_serializable_and_nested():
17
+ data = make_result().to_dict()
18
+ assert data["text"] == "olá mundo"
19
+ assert data["segments"] == [{"start": 0.0, "end": 1.5, "text": "olá mundo"}]
20
+ json.dumps(data)
21
+
22
+
23
+ def test_to_json_keeps_non_ascii_characters():
24
+ raw = make_result().to_json()
25
+ assert "olá mundo" in raw
26
+ assert json.loads(raw)["language"] == "pt"
@@ -0,0 +1,111 @@
1
+ import io
2
+ from types import SimpleNamespace
3
+
4
+ import pytest
5
+ from av.error import InvalidDataError
6
+
7
+ from audio_transcriber import _model, transcriber
8
+ from audio_transcriber.transcriber import TranscriptionError, transcribe
9
+
10
+
11
+ class FakeModel:
12
+ def __init__(self, segments=None, error=None):
13
+ self.calls = []
14
+ self._segments = segments if segments is not None else [
15
+ SimpleNamespace(start=0.0, end=1.0, text=" hello"),
16
+ SimpleNamespace(start=1.0, end=2.0, text=" world "),
17
+ ]
18
+ self._error = error
19
+
20
+ def transcribe(self, source, **kwargs):
21
+ self.calls.append((source, kwargs))
22
+ if self._error:
23
+ raise self._error
24
+ info = SimpleNamespace(language="en", language_probability=0.9, duration=2.0)
25
+ return iter(self._segments), info
26
+
27
+
28
+ @pytest.fixture
29
+ def fake(monkeypatch):
30
+ model = FakeModel()
31
+ monkeypatch.setattr(_model, "get_model", lambda name, device, ct: model)
32
+ return model
33
+
34
+
35
+ def test_bytes_are_wrapped_in_a_file_like_object(fake):
36
+ transcribe(b"fake-audio")
37
+ source, _ = fake.calls[0]
38
+ assert isinstance(source, io.BytesIO)
39
+ assert source.read() == b"fake-audio"
40
+
41
+
42
+ def test_path_is_passed_as_string(fake, tmp_path):
43
+ path = tmp_path / "a.wav"
44
+ transcribe(path)
45
+ assert fake.calls[0][0] == str(path)
46
+
47
+
48
+ def test_file_like_is_passed_through(fake):
49
+ buffer = io.BytesIO(b"x")
50
+ transcribe(buffer)
51
+ assert fake.calls[0][0] is buffer
52
+
53
+
54
+ def test_result_joins_and_strips_segment_text(fake):
55
+ result = transcribe(b"x")
56
+ assert result.text == "hello world"
57
+ assert [s.text for s in result.segments] == ["hello", "world"]
58
+ assert result.language == "en"
59
+ assert result.duration == 2.0
60
+
61
+
62
+ def test_silence_gives_empty_text(monkeypatch):
63
+ monkeypatch.setattr(_model, "get_model", lambda *a: FakeModel(segments=[]))
64
+ result = transcribe(b"x")
65
+ assert result.text == ""
66
+ assert result.segments == ()
67
+
68
+
69
+ def test_options_are_forwarded(fake):
70
+ transcribe(b"x", language="pt", beam_size=1, vad_filter=False)
71
+ assert fake.calls[0][1] == {"language": "pt", "beam_size": 1, "vad_filter": False}
72
+
73
+
74
+ def test_model_options_select_the_model(monkeypatch):
75
+ seen = []
76
+ monkeypatch.setattr(
77
+ _model, "get_model", lambda *args: seen.append(args) or FakeModel()
78
+ )
79
+ transcribe(b"x", model="base", device="cpu", compute_type="float32")
80
+ assert seen == [("base", "cpu", "float32")]
81
+
82
+
83
+ def test_empty_bytes_raise(fake):
84
+ with pytest.raises(TranscriptionError, match="empty"):
85
+ transcribe(b"")
86
+ assert fake.calls == []
87
+
88
+
89
+ def test_decoder_failures_become_transcription_errors(monkeypatch):
90
+ broken = FakeModel(error=InvalidDataError(1, "invalid data"))
91
+ monkeypatch.setattr(_model, "get_model", lambda *a: broken)
92
+ with pytest.raises(TranscriptionError, match="invalid data"):
93
+ transcribe(b"not audio")
94
+
95
+
96
+ def test_default_model_is_small():
97
+ assert transcriber.DEFAULT_MODEL == "small"
98
+
99
+
100
+ def test_non_decoding_failures_are_not_disguised(monkeypatch):
101
+ broken = FakeModel(error=RuntimeError("cuda library missing"))
102
+ monkeypatch.setattr(_model, "get_model", lambda *a: broken)
103
+ with pytest.raises(RuntimeError, match="cuda library missing"):
104
+ transcribe(b"x")
105
+
106
+
107
+ def test_device_defaults_to_cpu(monkeypatch):
108
+ seen = []
109
+ monkeypatch.setattr(_model, "get_model", lambda *args: seen.append(args) or FakeModel())
110
+ transcribe(b"x")
111
+ assert seen == [("small", "cpu", "int8")]