speech-quality 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- speech_quality-0.1.0/.gitignore +30 -0
- speech_quality-0.1.0/LICENSE +21 -0
- speech_quality-0.1.0/PKG-INFO +196 -0
- speech_quality-0.1.0/README.md +171 -0
- speech_quality-0.1.0/pyproject.toml +52 -0
- speech_quality-0.1.0/src/speech_quality/__init__.py +47 -0
- speech_quality-0.1.0/src/speech_quality/_frames.py +299 -0
- speech_quality-0.1.0/src/speech_quality/_scoring.py +104 -0
- speech_quality-0.1.0/src/speech_quality/audio.py +448 -0
- speech_quality-0.1.0/src/speech_quality/cli.py +203 -0
- speech_quality-0.1.0/src/speech_quality/core.py +284 -0
- speech_quality-0.1.0/src/speech_quality/measures.py +782 -0
- speech_quality-0.1.0/src/speech_quality/report.py +444 -0
- speech_quality-0.1.0/src/speech_quality/thresholds.py +222 -0
- speech_quality-0.1.0/tests/conftest.py +202 -0
- speech_quality-0.1.0/tests/test_api.py +293 -0
- speech_quality-0.1.0/tests/test_audio.py +301 -0
- speech_quality-0.1.0/tests/test_cli.py +187 -0
- speech_quality-0.1.0/tests/test_frames.py +122 -0
- speech_quality-0.1.0/tests/test_measures.py +351 -0
- speech_quality-0.1.0/tests/test_report.py +211 -0
- speech_quality-0.1.0/tests/test_verdict.py +230 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Virtual environments and scratch space used while building. Never committed:
|
|
2
|
+
# they are large, machine-specific, and rebuilt from pyproject.toml anyway.
|
|
3
|
+
_venvs/
|
|
4
|
+
_proof/
|
|
5
|
+
.venv*/
|
|
6
|
+
|
|
7
|
+
# Build output. Wheels are built by the release workflow from this source, so a
|
|
8
|
+
# wheel in git could silently differ from the code beside it.
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
src/*.egg-info/
|
|
13
|
+
|
|
14
|
+
# Python noise
|
|
15
|
+
__pycache__/
|
|
16
|
+
*.py[cod]
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.mypy_cache/
|
|
19
|
+
.ruff_cache/
|
|
20
|
+
|
|
21
|
+
# Local verification state, not source
|
|
22
|
+
verify_results.json
|
|
23
|
+
publish-log.txt
|
|
24
|
+
published.json
|
|
25
|
+
|
|
26
|
+
# Credentials. None of these belong here, and this line is the backstop.
|
|
27
|
+
.pypirc
|
|
28
|
+
.env
|
|
29
|
+
*.pem
|
|
30
|
+
*.key
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pranay Mahendrakar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: speech-quality
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Measure whether a voice recording is clean enough to transcribe or publish
|
|
5
|
+
Project-URL: Homepage, https://pypi.org/project/speech-quality/
|
|
6
|
+
Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
|
|
7
|
+
Author: Pranay Mahendrakar
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: audio-analysis,audio-quality,clipping,noise-floor,recording,signal-to-noise,speech-quality,transcription,voice,wav
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Analysis
|
|
18
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Requires-Python: >=3.9
|
|
21
|
+
Requires-Dist: numpy>=1.23
|
|
22
|
+
Provides-Extra: dev
|
|
23
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# speech-quality
|
|
27
|
+
|
|
28
|
+
Measure whether a voice recording is clean enough to transcribe or publish, and
|
|
29
|
+
say in plain words what is wrong with it when it is not.
|
|
30
|
+
|
|
31
|
+
## Install
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install speech-quality
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
The only dependency is numpy. WAV files are read with the standard library
|
|
38
|
+
`wave` module, so there is no audio library to install, no codec to find and
|
|
39
|
+
nothing is ever downloaded.
|
|
40
|
+
|
|
41
|
+
## Quickstart
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
import numpy as np
|
|
45
|
+
from speech_quality import assess
|
|
46
|
+
|
|
47
|
+
t = np.arange(3 * 16000) / 16000.0 # three seconds at 16 kHz
|
|
48
|
+
hiss = np.random.default_rng(0).standard_normal(t.size + 4)
|
|
49
|
+
voice = np.convolve(hiss, np.ones(5) / 5, "valid") # roll off the top
|
|
50
|
+
voice *= 0.35 * (0.5 + 0.5 * np.sin(2 * np.pi * 3.5 * t)) # syllable-rate envelope
|
|
51
|
+
print(assess(voice, sample_rate=16000).summary())
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
recording: usable (score 96.9 / 100, grade A)
|
|
56
|
+
3.00 s, 16000 Hz, mono
|
|
57
|
+
measures:
|
|
58
|
+
ok level 98.7 -20.27 dBFS level is healthy at -20.3 dBFS RMS, peaks at -3.6 dBFS, 3.6 dB of headroom left
|
|
59
|
+
ok clipping 100.0 0.0% no samples reach full scale, so nothing is clipped
|
|
60
|
+
ok noise 85.7 25.12 dB the voice stands 25.1 dB above a -41.7 dBFS noise floor, clean enough to transcribe
|
|
61
|
+
ok silence 99.5 3.2% 3% silence overall, 0.00 s at the start and 0.00 s at the end, which is normal for speech
|
|
62
|
+
ok speech 100.0 96.8% 97% of the recording carries speech, centred on 1300 Hz with the level swinging 9.0 dB between syllables
|
|
63
|
+
ok dynamics 100.0 16.52 dB a 16.5 dB crest factor over the parts that carry sound, the natural rise and fall of unprocessed speech
|
|
64
|
+
ok bandwidth 100.0 8000 Hz energy reaches 8000 Hz, 100% of the way to the 8000 Hz ceiling, wide enough for clear consonants
|
|
65
|
+
issues: none
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
On a real file it is one line:
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
report = assess("interview.wav")
|
|
72
|
+
if not report.usable:
|
|
73
|
+
print(report.issues[0])
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## What it checks
|
|
77
|
+
|
|
78
|
+
Seven measures, each reporting the raw number it found, a 0-100 score, whether
|
|
79
|
+
that passes, and a sentence you can act on.
|
|
80
|
+
|
|
81
|
+
- **level** - RMS and peak in dBFS, and how far the level sits from the -20 dBFS
|
|
82
|
+
a healthy speech recording holds. Catches material recorded too quiet to
|
|
83
|
+
survive noise reduction and material recorded too hot to survive anything.
|
|
84
|
+
- **clipping** - the share of samples pinned at or near full scale, how many
|
|
85
|
+
separate clipped runs there are and how long the worst one lasts. Clipping is
|
|
86
|
+
destroyed waveform; no processing brings it back.
|
|
87
|
+
- **noise** - the noise floor taken from the quietest decile of frames, and the
|
|
88
|
+
signal-to-noise ratio between that and the loudest decile. No voice-activity
|
|
89
|
+
decision is involved, so none can go wrong.
|
|
90
|
+
- **silence** - leading silence, trailing silence and the total silent share, so
|
|
91
|
+
dead air is trimmed before anyone pays to transcribe it.
|
|
92
|
+
- **speech** - the share of the recording carrying speech-like energy, judged on
|
|
93
|
+
band energy, spectral centroid and whether the level actually swings at
|
|
94
|
+
syllable rate. A steady tone passes every spectral test and is still not
|
|
95
|
+
speech.
|
|
96
|
+
- **dynamics** - crest factor, and whether the life has been compressed or
|
|
97
|
+
limited out of the recording.
|
|
98
|
+
- **bandwidth** - the highest frequency still carrying real energy. This is what
|
|
99
|
+
exposes telephone-band audio, and audio upsampled from a narrower original:
|
|
100
|
+
the file says 48 kHz, the sound stops at 3.4 kHz, and every consonant that
|
|
101
|
+
lived above that is already gone.
|
|
102
|
+
|
|
103
|
+
Four of those faults are damage rather than a chore. Clipped samples are a
|
|
104
|
+
destroyed waveform, noise is already mixed in under the voice, a recording that
|
|
105
|
+
does not behave like speech has nothing in it to transcribe, and a band that
|
|
106
|
+
stops at 3.4 kHz has lost its consonants. Nothing you do later brings any of
|
|
107
|
+
them back, so **any one of clipping, noise, speech or bandwidth failing makes
|
|
108
|
+
the recording unusable**, however well the other measures average out;
|
|
109
|
+
`report.blocking` names the ones that did and the CLI exits `1`. A level 6 dB
|
|
110
|
+
off, dead air at the ends and heavy compression are all things a later pass
|
|
111
|
+
fixes, so they pull the score down without condemning the take.
|
|
112
|
+
|
|
113
|
+
Input can be a path to a `.wav` file (PCM 8, 16, 24 or 32 bit, 8 kHz to 48 kHz
|
|
114
|
+
and beyond), a numpy array of samples, or a `(samples, sample_rate)` pair.
|
|
115
|
+
Multi-channel audio is mixed down to mono and the mixdown is recorded in
|
|
116
|
+
`report.notes`. Your array is never written to. The same input always gives the
|
|
117
|
+
same answer.
|
|
118
|
+
|
|
119
|
+
**These are signal measurements, not a perceptual model.** speech-quality is not
|
|
120
|
+
PESQ, POLQA or ViSQOL and does not estimate a mean opinion score. It tells you
|
|
121
|
+
that a recording clips, hisses, sits 8 dB too quiet or stops dead at 3.4 kHz. It
|
|
122
|
+
does not tell you how a listener would rate it, and it has no opinion at all
|
|
123
|
+
about accent, delivery or content.
|
|
124
|
+
|
|
125
|
+
## API
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from speech_quality import assess, assess_batch, signal_to_noise, estimate_noise_floor
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
`assess(audio, *, sample_rate=None, thresholds=None, source=None) -> AudioReport`
|
|
132
|
+
Assess one recording.
|
|
133
|
+
|
|
134
|
+
`assess_batch(items, *, sample_rate=None, thresholds=None) -> BatchReport`
|
|
135
|
+
Assess many, keeping going when one cannot be read. An item may be a
|
|
136
|
+
`(label, recording)` pair, which is how arrays get names.
|
|
137
|
+
|
|
138
|
+
`signal_to_noise(audio, **kw) -> float` the ratio in dB.
|
|
139
|
+
`estimate_noise_floor(audio, **kw) -> float` the floor in dBFS.
|
|
140
|
+
|
|
141
|
+
**AudioReport**
|
|
142
|
+
|
|
143
|
+
| attribute | meaning |
|
|
144
|
+
| --- | --- |
|
|
145
|
+
| `.score` | 0-100 overall, the weighted average of the measures |
|
|
146
|
+
| `.grade` | `"A"` (best) through `"F"` |
|
|
147
|
+
| `.usable` | good enough to transcribe or publish: the score clears `usable_score`, enough of the measures could be taken, and none of the four decisive ones failed |
|
|
148
|
+
| `.blocking` | the decisive measures that failed, worst fault first; empty when none did |
|
|
149
|
+
| `.metrics` | `dict[name -> Metric]`, one per measure |
|
|
150
|
+
| `.issues` | what is wrong, worst first, one sentence each |
|
|
151
|
+
| `.notes` | decisions taken while loading, such as a stereo mixdown |
|
|
152
|
+
| `.coverage` | share of the measures that had anything to work with |
|
|
153
|
+
| `.duration`, `.sample_rate`, `.channels`, `.digital_silence` | what was read |
|
|
154
|
+
| `.summary()` | the whole report as plain text |
|
|
155
|
+
| `.to_dict()` | the whole report as JSON-safe data |
|
|
156
|
+
|
|
157
|
+
**Metric** carries `.value` (the raw number), `.unit`, `.score` (0-100), `.ok`,
|
|
158
|
+
`.measured` and `.message`. `.measured` is `False` when the recording could not
|
|
159
|
+
support that measurement - a spectrum of four samples, a noise floor of a signal
|
|
160
|
+
whose level never drops - in which case the score is a neutral placeholder and
|
|
161
|
+
not a judgement, and `.ok` is `None` rather than `True`, because nothing was
|
|
162
|
+
measured and nothing passed. A recording most of the measures could not touch is
|
|
163
|
+
never called usable, however well the placeholders average out.
|
|
164
|
+
|
|
165
|
+
**BatchReport** carries `.results`, `.failures`, `.usable`, `.unusable`,
|
|
166
|
+
`.mean_score`, `.worst(n=5)`, `.summary()` and `.to_dict()`.
|
|
167
|
+
|
|
168
|
+
**Thresholds** holds every limit the verdict uses, and any of them can be moved:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from speech_quality import assess
|
|
172
|
+
|
|
173
|
+
report = assess("podcast.wav", thresholds={"min_snr_db": 25, "usable_score": 70})
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
## CLI
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
speech-quality interview.wav # the report as text
|
|
180
|
+
speech-quality takes/*.wav --worst 3 # a batch, worst offenders first
|
|
181
|
+
speech-quality interview.wav --json # the report as JSON
|
|
182
|
+
speech-quality interview.wav --output report.txt # written as UTF-8
|
|
183
|
+
speech-quality interview.wav --min-snr 20 --usable-score 70
|
|
184
|
+
speech-quality --help
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
Exit codes: `0` every recording is usable, `1` at least one is not, `2` nothing
|
|
188
|
+
could be read. That makes it a filter:
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
for f in takes/*.wav; do speech-quality "$f" --quiet || echo "redo: $f"; done
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
## License
|
|
195
|
+
|
|
196
|
+
MIT. Copyright (c) 2026 Pranay Mahendrakar.
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# speech-quality
|
|
2
|
+
|
|
3
|
+
Measure whether a voice recording is clean enough to transcribe or publish, and
|
|
4
|
+
say in plain words what is wrong with it when it is not.
|
|
5
|
+
|
|
6
|
+
## Install
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
pip install speech-quality
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
The only dependency is numpy. WAV files are read with the standard library
|
|
13
|
+
`wave` module, so there is no audio library to install, no codec to find and
|
|
14
|
+
nothing is ever downloaded.
|
|
15
|
+
|
|
16
|
+
## Quickstart
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
import numpy as np
|
|
20
|
+
from speech_quality import assess
|
|
21
|
+
|
|
22
|
+
t = np.arange(3 * 16000) / 16000.0 # three seconds at 16 kHz
|
|
23
|
+
hiss = np.random.default_rng(0).standard_normal(t.size + 4)
|
|
24
|
+
voice = np.convolve(hiss, np.ones(5) / 5, "valid") # roll off the top
|
|
25
|
+
voice *= 0.35 * (0.5 + 0.5 * np.sin(2 * np.pi * 3.5 * t)) # syllable-rate envelope
|
|
26
|
+
print(assess(voice, sample_rate=16000).summary())
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
recording: usable (score 96.9 / 100, grade A)
|
|
31
|
+
3.00 s, 16000 Hz, mono
|
|
32
|
+
measures:
|
|
33
|
+
ok level 98.7 -20.27 dBFS level is healthy at -20.3 dBFS RMS, peaks at -3.6 dBFS, 3.6 dB of headroom left
|
|
34
|
+
ok clipping 100.0 0.0% no samples reach full scale, so nothing is clipped
|
|
35
|
+
ok noise 85.7 25.12 dB the voice stands 25.1 dB above a -41.7 dBFS noise floor, clean enough to transcribe
|
|
36
|
+
ok silence 99.5 3.2% 3% silence overall, 0.00 s at the start and 0.00 s at the end, which is normal for speech
|
|
37
|
+
ok speech 100.0 96.8% 97% of the recording carries speech, centred on 1300 Hz with the level swinging 9.0 dB between syllables
|
|
38
|
+
ok dynamics 100.0 16.52 dB a 16.5 dB crest factor over the parts that carry sound, the natural rise and fall of unprocessed speech
|
|
39
|
+
ok bandwidth 100.0 8000 Hz energy reaches 8000 Hz, 100% of the way to the 8000 Hz ceiling, wide enough for clear consonants
|
|
40
|
+
issues: none
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
On a real file it is one line:
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
report = assess("interview.wav")
|
|
47
|
+
if not report.usable:
|
|
48
|
+
print(report.issues[0])
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## What it checks
|
|
52
|
+
|
|
53
|
+
Seven measures, each reporting the raw number it found, a 0-100 score, whether
|
|
54
|
+
that passes, and a sentence you can act on.
|
|
55
|
+
|
|
56
|
+
- **level** - RMS and peak in dBFS, and how far the level sits from the -20 dBFS
|
|
57
|
+
a healthy speech recording holds. Catches material recorded too quiet to
|
|
58
|
+
survive noise reduction and material recorded too hot to survive anything.
|
|
59
|
+
- **clipping** - the share of samples pinned at or near full scale, how many
|
|
60
|
+
separate clipped runs there are and how long the worst one lasts. Clipping is
|
|
61
|
+
destroyed waveform; no processing brings it back.
|
|
62
|
+
- **noise** - the noise floor taken from the quietest decile of frames, and the
|
|
63
|
+
signal-to-noise ratio between that and the loudest decile. No voice-activity
|
|
64
|
+
decision is involved, so none can go wrong.
|
|
65
|
+
- **silence** - leading silence, trailing silence and the total silent share, so
|
|
66
|
+
dead air is trimmed before anyone pays to transcribe it.
|
|
67
|
+
- **speech** - the share of the recording carrying speech-like energy, judged on
|
|
68
|
+
band energy, spectral centroid and whether the level actually swings at
|
|
69
|
+
syllable rate. A steady tone passes every spectral test and is still not
|
|
70
|
+
speech.
|
|
71
|
+
- **dynamics** - crest factor, and whether the life has been compressed or
|
|
72
|
+
limited out of the recording.
|
|
73
|
+
- **bandwidth** - the highest frequency still carrying real energy. This is what
|
|
74
|
+
exposes telephone-band audio, and audio upsampled from a narrower original:
|
|
75
|
+
the file says 48 kHz, the sound stops at 3.4 kHz, and every consonant that
|
|
76
|
+
lived above that is already gone.
|
|
77
|
+
|
|
78
|
+
Four of those faults are damage rather than a chore. Clipped samples are a
|
|
79
|
+
destroyed waveform, noise is already mixed in under the voice, a recording that
|
|
80
|
+
does not behave like speech has nothing in it to transcribe, and a band that
|
|
81
|
+
stops at 3.4 kHz has lost its consonants. Nothing you do later brings any of
|
|
82
|
+
them back, so **any one of clipping, noise, speech or bandwidth failing makes
|
|
83
|
+
the recording unusable**, however well the other measures average out;
|
|
84
|
+
`report.blocking` names the ones that did and the CLI exits `1`. A level 6 dB
|
|
85
|
+
off, dead air at the ends and heavy compression are all things a later pass
|
|
86
|
+
fixes, so they pull the score down without condemning the take.
|
|
87
|
+
|
|
88
|
+
Input can be a path to a `.wav` file (PCM 8, 16, 24 or 32 bit, 8 kHz to 48 kHz
|
|
89
|
+
and beyond), a numpy array of samples, or a `(samples, sample_rate)` pair.
|
|
90
|
+
Multi-channel audio is mixed down to mono and the mixdown is recorded in
|
|
91
|
+
`report.notes`. Your array is never written to. The same input always gives the
|
|
92
|
+
same answer.
|
|
93
|
+
|
|
94
|
+
**These are signal measurements, not a perceptual model.** speech-quality is not
|
|
95
|
+
PESQ, POLQA or ViSQOL and does not estimate a mean opinion score. It tells you
|
|
96
|
+
that a recording clips, hisses, sits 8 dB too quiet or stops dead at 3.4 kHz. It
|
|
97
|
+
does not tell you how a listener would rate it, and it has no opinion at all
|
|
98
|
+
about accent, delivery or content.
|
|
99
|
+
|
|
100
|
+
## API
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from speech_quality import assess, assess_batch, signal_to_noise, estimate_noise_floor
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
`assess(audio, *, sample_rate=None, thresholds=None, source=None) -> AudioReport`
|
|
107
|
+
Assess one recording.
|
|
108
|
+
|
|
109
|
+
`assess_batch(items, *, sample_rate=None, thresholds=None) -> BatchReport`
|
|
110
|
+
Assess many, keeping going when one cannot be read. An item may be a
|
|
111
|
+
`(label, recording)` pair, which is how arrays get names.
|
|
112
|
+
|
|
113
|
+
`signal_to_noise(audio, **kw) -> float` the ratio in dB.
|
|
114
|
+
`estimate_noise_floor(audio, **kw) -> float` the floor in dBFS.
|
|
115
|
+
|
|
116
|
+
**AudioReport**
|
|
117
|
+
|
|
118
|
+
| attribute | meaning |
|
|
119
|
+
| --- | --- |
|
|
120
|
+
| `.score` | 0-100 overall, the weighted average of the measures |
|
|
121
|
+
| `.grade` | `"A"` (best) through `"F"` |
|
|
122
|
+
| `.usable` | good enough to transcribe or publish: the score clears `usable_score`, enough of the measures could be taken, and none of the four decisive ones failed |
|
|
123
|
+
| `.blocking` | the decisive measures that failed, worst fault first; empty when none did |
|
|
124
|
+
| `.metrics` | `dict[name -> Metric]`, one per measure |
|
|
125
|
+
| `.issues` | what is wrong, worst first, one sentence each |
|
|
126
|
+
| `.notes` | decisions taken while loading, such as a stereo mixdown |
|
|
127
|
+
| `.coverage` | share of the measures that had anything to work with |
|
|
128
|
+
| `.duration`, `.sample_rate`, `.channels`, `.digital_silence` | what was read |
|
|
129
|
+
| `.summary()` | the whole report as plain text |
|
|
130
|
+
| `.to_dict()` | the whole report as JSON-safe data |
|
|
131
|
+
|
|
132
|
+
**Metric** carries `.value` (the raw number), `.unit`, `.score` (0-100), `.ok`,
|
|
133
|
+
`.measured` and `.message`. `.measured` is `False` when the recording could not
|
|
134
|
+
support that measurement - a spectrum of four samples, a noise floor of a signal
|
|
135
|
+
whose level never drops - in which case the score is a neutral placeholder and
|
|
136
|
+
not a judgement, and `.ok` is `None` rather than `True`, because nothing was
|
|
137
|
+
measured and nothing passed. A recording most of the measures could not touch is
|
|
138
|
+
never called usable, however well the placeholders average out.
|
|
139
|
+
|
|
140
|
+
**BatchReport** carries `.results`, `.failures`, `.usable`, `.unusable`,
|
|
141
|
+
`.mean_score`, `.worst(n=5)`, `.summary()` and `.to_dict()`.
|
|
142
|
+
|
|
143
|
+
**Thresholds** holds every limit the verdict uses, and any of them can be moved:
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
from speech_quality import assess
|
|
147
|
+
|
|
148
|
+
report = assess("podcast.wav", thresholds={"min_snr_db": 25, "usable_score": 70})
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
## CLI
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
speech-quality interview.wav # the report as text
|
|
155
|
+
speech-quality takes/*.wav --worst 3 # a batch, worst offenders first
|
|
156
|
+
speech-quality interview.wav --json # the report as JSON
|
|
157
|
+
speech-quality interview.wav --output report.txt # written as UTF-8
|
|
158
|
+
speech-quality interview.wav --min-snr 20 --usable-score 70
|
|
159
|
+
speech-quality --help
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
Exit codes: `0` every recording is usable, `1` at least one is not, `2` nothing
|
|
163
|
+
could be read. That makes it a filter:
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
for f in takes/*.wav; do speech-quality "$f" --quiet || echo "redo: $f"; done
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
## License
|
|
170
|
+
|
|
171
|
+
MIT. Copyright (c) 2026 Pranay Mahendrakar.
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "speech-quality"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Measure whether a voice recording is clean enough to transcribe or publish"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Pranay Mahendrakar" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"speech-quality",
|
|
16
|
+
"audio-quality",
|
|
17
|
+
"voice",
|
|
18
|
+
"recording",
|
|
19
|
+
"transcription",
|
|
20
|
+
"clipping",
|
|
21
|
+
"signal-to-noise",
|
|
22
|
+
"noise-floor",
|
|
23
|
+
"wav",
|
|
24
|
+
"audio-analysis",
|
|
25
|
+
]
|
|
26
|
+
classifiers = [
|
|
27
|
+
"Development Status :: 4 - Beta",
|
|
28
|
+
"Intended Audience :: Developers",
|
|
29
|
+
"Intended Audience :: Science/Research",
|
|
30
|
+
"Programming Language :: Python :: 3",
|
|
31
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
32
|
+
"Operating System :: OS Independent",
|
|
33
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
34
|
+
"Topic :: Multimedia :: Sound/Audio :: Analysis",
|
|
35
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
36
|
+
]
|
|
37
|
+
dependencies = [
|
|
38
|
+
"numpy>=1.23",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[project.optional-dependencies]
|
|
42
|
+
dev = ["pytest>=7"]
|
|
43
|
+
|
|
44
|
+
[project.scripts]
|
|
45
|
+
speech-quality = "speech_quality.cli:main"
|
|
46
|
+
|
|
47
|
+
[project.urls]
|
|
48
|
+
Homepage = "https://pypi.org/project/speech-quality/"
|
|
49
|
+
Author = "https://pypi.org/user/pranaymahendrakar/"
|
|
50
|
+
|
|
51
|
+
[tool.hatch.build.targets.wheel]
|
|
52
|
+
packages = ["src/speech_quality"]
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""speech-quality: is this voice recording clean enough to transcribe or publish?
|
|
2
|
+
|
|
3
|
+
from speech_quality import assess
|
|
4
|
+
report = assess("interview.wav")
|
|
5
|
+
print(report.summary())
|
|
6
|
+
|
|
7
|
+
Seven measures - level, clipping, noise, silence, speech, dynamics and
|
|
8
|
+
bandwidth - each give a raw number, a 0-100 score and a sentence you can act
|
|
9
|
+
on. WAV files are read with the standard library, so the only dependency is
|
|
10
|
+
numpy and nothing is ever downloaded.
|
|
11
|
+
|
|
12
|
+
These are signal measurements, not a perceptual model: the package tells you
|
|
13
|
+
that a recording clips, hisses or stops at 3.4 kHz, not what a listener would
|
|
14
|
+
score it out of five.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from ._frames import Frames, Spectra, frame_signal, take_spectra
|
|
20
|
+
from .audio import Audio, load_audio, read_wav
|
|
21
|
+
from .core import assess, assess_batch, estimate_noise_floor, signal_to_noise
|
|
22
|
+
from .report import AudioReport, BatchReport, Metric
|
|
23
|
+
from .thresholds import DECISIVE_MEASURES, DEFAULT_THRESHOLDS, MEASURES, Thresholds
|
|
24
|
+
|
|
25
|
+
__version__ = "0.1.0"
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"assess",
|
|
29
|
+
"assess_batch",
|
|
30
|
+
"signal_to_noise",
|
|
31
|
+
"estimate_noise_floor",
|
|
32
|
+
"AudioReport",
|
|
33
|
+
"BatchReport",
|
|
34
|
+
"Metric",
|
|
35
|
+
"Audio",
|
|
36
|
+
"Thresholds",
|
|
37
|
+
"DEFAULT_THRESHOLDS",
|
|
38
|
+
"MEASURES",
|
|
39
|
+
"DECISIVE_MEASURES",
|
|
40
|
+
"Frames",
|
|
41
|
+
"Spectra",
|
|
42
|
+
"load_audio",
|
|
43
|
+
"read_wav",
|
|
44
|
+
"frame_signal",
|
|
45
|
+
"take_spectra",
|
|
46
|
+
"__version__",
|
|
47
|
+
]
|