speech-quality 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,47 @@
1
+ """speech-quality: is this voice recording clean enough to transcribe or publish?
2
+
3
+ from speech_quality import assess
4
+ report = assess("interview.wav")
5
+ print(report.summary())
6
+
7
+ Seven measures - level, clipping, noise, silence, speech, dynamics and
8
+ bandwidth - each give a raw number, a 0-100 score and a sentence you can act
9
+ on. WAV files are read with the standard library, so the only dependency is
10
+ numpy and nothing is ever downloaded.
11
+
12
+ These are signal measurements, not a perceptual model: the package tells you
13
+ that a recording clips, hisses or stops at 3.4 kHz, not what a listener would
14
+ score it out of five.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from ._frames import Frames, Spectra, frame_signal, take_spectra
20
+ from .audio import Audio, load_audio, read_wav
21
+ from .core import assess, assess_batch, estimate_noise_floor, signal_to_noise
22
+ from .report import AudioReport, BatchReport, Metric
23
+ from .thresholds import DECISIVE_MEASURES, DEFAULT_THRESHOLDS, MEASURES, Thresholds
24
+
25
+ __version__ = "0.1.0"
26
+
27
+ __all__ = [
28
+ "assess",
29
+ "assess_batch",
30
+ "signal_to_noise",
31
+ "estimate_noise_floor",
32
+ "AudioReport",
33
+ "BatchReport",
34
+ "Metric",
35
+ "Audio",
36
+ "Thresholds",
37
+ "DEFAULT_THRESHOLDS",
38
+ "MEASURES",
39
+ "DECISIVE_MEASURES",
40
+ "Frames",
41
+ "Spectra",
42
+ "load_audio",
43
+ "read_wav",
44
+ "frame_signal",
45
+ "take_spectra",
46
+ "__version__",
47
+ ]
@@ -0,0 +1,299 @@
1
+ """Cutting a recording into short frames, and taking spectra of some of them.
2
+
3
+ Every measure that is about *when* something happens - silence at the edges,
4
+ speech in the middle, a noise floor between words - works on short overlapping
5
+ frames rather than on the whole signal, because a single number over a whole
6
+ recording hides exactly what the caller wants to know.
7
+
8
+ Frame levels come from a running sum of squares, so the cost is linear in the
9
+ number of samples and a minute of 48 kHz audio is framed in milliseconds.
10
+ Spectra cost far more, so at most ``max_spectral_frames`` of them are taken,
11
+ spread evenly across the recording; evenly, not randomly, so two runs on the
12
+ same input always agree.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass
18
+ from typing import Optional, Tuple
19
+
20
+ import numpy as np
21
+
22
+ from ._scoring import dbfs_array
23
+ from .thresholds import Thresholds
24
+
25
+ __all__ = ["Frames", "Spectra", "frame_peaks", "frame_signal", "take_spectra"]
26
+
27
+ MIN_FRAME_LENGTH = 16
28
+ """No frame is shorter than this, whatever the sample rate works out to."""
29
+
30
+ PEAK_BLOCK_SAMPLES = 1 << 20
31
+ """Most samples one peak-per-frame block may hold: 8 MB of float64, whatever the
32
+ recording is. Frames overlap, so materialising them all at once costs frames x
33
+ frame_length rather than samples - twice the audio at the default 2:1 overlap,
34
+ 2.8 GB for an hour of 48 kHz interview, which is a MemoryError on a small
35
+ laptop. The peak is taken a block at a time instead, and this is the ceiling on
36
+ that block."""
37
+
38
+
39
+ @dataclass
40
+ class Frames:
41
+ """Short overlapping windows of the recording, with a level for each.
42
+
43
+ Attributes:
44
+ starts: first sample index of each frame.
45
+ length: frame length in samples.
46
+ hop: step between frame starts, in samples.
47
+ rms: per-frame RMS amplitude, full scale at 1.0.
48
+ dbfs: per-frame level in dBFS, floored rather than negative infinity.
49
+ peak: per-frame peak absolute amplitude.
50
+ sample_rate: samples per second.
51
+ n_samples: length of the recording the frames came from.
52
+ """
53
+
54
+ starts: np.ndarray
55
+ length: int
56
+ hop: int
57
+ rms: np.ndarray
58
+ dbfs: np.ndarray
59
+ peak: np.ndarray
60
+ sample_rate: int
61
+ n_samples: int
62
+
63
+ @property
64
+ def count(self) -> int:
65
+ """How many frames there are; always at least one."""
66
+ return int(self.starts.size)
67
+
68
+ @property
69
+ def seconds_per_frame(self) -> float:
70
+ """How much time one frame step stands for."""
71
+ return float(self.hop) / float(self.sample_rate)
72
+
73
+ def active_mask(self, thresholds: Thresholds) -> np.ndarray:
74
+ """Which frames carry something rather than silence.
75
+
76
+ A frame is active when it is both within ``silence_drop_db`` of the
77
+ loudest frame and above the absolute ``silence_floor_dbfs``. The
78
+ relative half keeps a quiet but clean recording from reading as all
79
+ silence; the absolute half keeps a recording of nothing but hiss from
80
+ reading as all speech.
81
+
82
+ Args:
83
+ thresholds: the limits to apply.
84
+
85
+ Returns:
86
+ A boolean array, one entry per frame.
87
+ """
88
+ loudest = float(np.max(self.dbfs))
89
+ relative = loudest - float(thresholds.silence_drop_db)
90
+ cutoff = max(relative, float(thresholds.silence_floor_dbfs))
91
+ return self.dbfs > cutoff
92
+
93
+
94
+ @dataclass
95
+ class Spectra:
96
+ """Power spectra of a sample of frames, and what they say per frame.
97
+
98
+ Attributes:
99
+ indices: which frames were analysed, ascending.
100
+ freqs: bin centre frequencies in Hz.
101
+ power: ``(len(indices), len(freqs))`` power in each bin.
102
+ centroid: per-analysed-frame spectral centroid in Hz.
103
+ band_share: per-analysed-frame share of energy inside the speech band.
104
+ total: per-analysed-frame total power, for weighting.
105
+ """
106
+
107
+ indices: np.ndarray
108
+ freqs: np.ndarray
109
+ power: np.ndarray
110
+ centroid: np.ndarray
111
+ band_share: np.ndarray
112
+ total: np.ndarray
113
+
114
+ @property
115
+ def count(self) -> int:
116
+ """How many frames were analysed."""
117
+ return int(self.indices.size)
118
+
119
+ def mean_power(self, mask: Optional[np.ndarray] = None) -> np.ndarray:
120
+ """Average spectrum over the analysed frames, or a subset of them.
121
+
122
+ Args:
123
+ mask: boolean array over the analysed frames. A mask that selects
124
+ nothing is ignored, so the caller always gets a usable spectrum.
125
+
126
+ Returns:
127
+ One power value per frequency bin.
128
+ """
129
+ if self.power.size == 0:
130
+ return np.zeros(self.freqs.size, dtype=np.float64)
131
+ if mask is not None and bool(np.any(mask)):
132
+ return np.asarray(self.power[mask].mean(axis=0), dtype=np.float64)
133
+ return np.asarray(self.power.mean(axis=0), dtype=np.float64)
134
+
135
+
136
+ def _frame_geometry(n_samples: int, sample_rate: int, thresholds: Thresholds) -> Tuple[int, int]:
137
+ """Frame length and hop in samples, kept sane for very short recordings."""
138
+ length = int(round(float(thresholds.frame_seconds) * float(sample_rate)))
139
+ length = max(MIN_FRAME_LENGTH, length)
140
+ length = min(length, max(1, n_samples))
141
+ hop = int(round(float(thresholds.hop_seconds) * float(sample_rate)))
142
+ hop = max(1, min(hop, length))
143
+ return length, hop
144
+
145
+
146
+ def _peak_block_frames(length: int) -> int:
147
+ """How many frames to take the peak of at once, so the temporary stays bounded."""
148
+ return max(1, int(PEAK_BLOCK_SAMPLES // max(1, int(length))))
149
+
150
+
151
+ def frame_peaks(signal: np.ndarray, starts: np.ndarray, length: int) -> np.ndarray:
152
+ """The largest absolute sample in each frame, in blocks.
153
+
154
+ ``sliding_window_view`` hands back every frame as a view costing nothing,
155
+ but indexing that view with the frame starts copies: frames x frame_length
156
+ floats, which for overlapping frames is a multiple of the recording itself.
157
+ Copying one bounded block at a time gives the same answer for a temporary
158
+ that never exceeds :data:`PEAK_BLOCK_SAMPLES` however long the recording is.
159
+
160
+ The block's largest and smallest values answer the same question as the
161
+ largest absolute value, which saves taking ``abs`` of the whole signal.
162
+
163
+ Args:
164
+ signal: one-dimensional float samples.
165
+ starts: first sample index of each frame; every frame must fit.
166
+ length: frame length in samples.
167
+
168
+ Returns:
169
+ One peak amplitude per frame, as float64.
170
+ """
171
+ length = int(length)
172
+ windows = np.lib.stride_tricks.sliding_window_view(signal, length)
173
+ peak = np.empty(int(starts.size), dtype=np.float64)
174
+ step = _peak_block_frames(length)
175
+ for begin in range(0, int(starts.size), step):
176
+ end = min(begin + step, int(starts.size))
177
+ block = windows[starts[begin:end]]
178
+ np.maximum(block.max(axis=1), -block.min(axis=1), out=peak[begin:end])
179
+ return peak
180
+
181
+
182
+ def frame_signal(samples: np.ndarray, sample_rate: int, thresholds: Thresholds) -> Frames:
183
+ """Cut the signal into overlapping frames and measure the level of each.
184
+
185
+ Args:
186
+ samples: one-dimensional float samples.
187
+ sample_rate: samples per second.
188
+ thresholds: supplies the frame and hop length.
189
+
190
+ Returns:
191
+ The frames and their levels. A recording shorter than one frame becomes
192
+ a single frame holding all of it. The working memory is one array the
193
+ size of the signal plus one bounded block, never frames x frame_length,
194
+ so an hour-long interview frames on an ordinary laptop.
195
+ """
196
+ signal = np.asarray(samples, dtype=np.float64)
197
+ n_samples = int(signal.size)
198
+ length, hop = _frame_geometry(n_samples, sample_rate, thresholds)
199
+
200
+ if n_samples <= length:
201
+ starts = np.zeros(1, dtype=np.int64)
202
+ divisor = np.array([float(max(n_samples, 1))], dtype=np.float64)
203
+ energy = np.array([float(np.dot(signal, signal))], dtype=np.float64)
204
+ peak = np.array([float(np.max(np.abs(signal)))], dtype=np.float64)
205
+ else:
206
+ last = n_samples - length
207
+ starts = np.arange(0, last + 1, hop, dtype=np.int64)
208
+ if int(starts[-1]) != last:
209
+ starts = np.append(starts, last)
210
+ # A running sum of squares makes every frame energy one subtraction. The
211
+ # squares are written straight into the running total and summed in
212
+ # place, so framing an hour-long interview holds one array beside the
213
+ # audio rather than three.
214
+ cumulative = np.empty(n_samples + 1, dtype=np.float64)
215
+ cumulative[0] = 0.0
216
+ np.multiply(signal, signal, out=cumulative[1:])
217
+ np.cumsum(cumulative[1:], out=cumulative[1:])
218
+ energy = np.maximum(cumulative[starts + length] - cumulative[starts], 0.0)
219
+ divisor = np.full(starts.size, float(length), dtype=np.float64)
220
+ peak = frame_peaks(signal, starts, length)
221
+
222
+ rms = np.sqrt(energy / divisor)
223
+ return Frames(
224
+ starts=starts,
225
+ length=int(length),
226
+ hop=int(hop),
227
+ rms=rms,
228
+ dbfs=dbfs_array(rms),
229
+ peak=peak,
230
+ sample_rate=int(sample_rate),
231
+ n_samples=n_samples,
232
+ )
233
+
234
+
235
+ def _pick_frames(frames: Frames, limit: int) -> np.ndarray:
236
+ """Up to ``limit`` frame indices, spread evenly and in order."""
237
+ count = frames.count
238
+ if limit <= 0 or count <= limit:
239
+ return np.arange(count, dtype=np.int64)
240
+ positions = np.linspace(0.0, float(count - 1), num=int(limit))
241
+ return np.unique(np.round(positions).astype(np.int64))
242
+
243
+
244
+ def take_spectra(samples: np.ndarray, frames: Frames, thresholds: Thresholds) -> Spectra:
245
+ """Take Hann-windowed power spectra of an evenly spread sample of frames.
246
+
247
+ Args:
248
+ samples: the signal the frames came from.
249
+ frames: the framing to follow.
250
+ thresholds: supplies the speech band and the frame budget.
251
+
252
+ Returns:
253
+ The spectra, plus the centroid and speech-band share of each analysed
254
+ frame. A recording too short for a usable transform comes back with
255
+ empty arrays, which every caller checks for.
256
+ """
257
+ signal = np.asarray(samples, dtype=np.float64)
258
+ length = frames.length
259
+ indices = _pick_frames(frames, int(thresholds.max_spectral_frames))
260
+ if length < 8 or indices.size == 0:
261
+ return Spectra(
262
+ indices=np.zeros(0, dtype=np.int64),
263
+ freqs=np.zeros(0, dtype=np.float64),
264
+ power=np.zeros((0, 0), dtype=np.float64),
265
+ centroid=np.zeros(0, dtype=np.float64),
266
+ band_share=np.zeros(0, dtype=np.float64),
267
+ total=np.zeros(0, dtype=np.float64),
268
+ )
269
+
270
+ starts = frames.starts[indices]
271
+ offsets = np.arange(length, dtype=np.int64)
272
+ block = signal[starts[:, None] + offsets[None, :]]
273
+ window = np.hanning(length)
274
+ spectrum = np.fft.rfft(block * window[None, :], axis=1)
275
+ power = (spectrum.real * spectrum.real) + (spectrum.imag * spectrum.imag)
276
+ freqs = np.fft.rfftfreq(length, d=1.0 / float(frames.sample_rate))
277
+
278
+ # The DC bin is a recording offset, not a frequency anyone hears.
279
+ if power.shape[1] > 1:
280
+ power[:, 0] = 0.0
281
+
282
+ total = power.sum(axis=1)
283
+ positive = total > 0.0
284
+ safe_total = np.where(positive, total, 1.0)
285
+ centroid = np.where(positive, (power * freqs[None, :]).sum(axis=1) / safe_total, 0.0)
286
+
287
+ in_band = (freqs >= float(thresholds.speech_band_low_hz)) & (
288
+ freqs <= float(thresholds.speech_band_high_hz)
289
+ )
290
+ band_share = np.where(positive, power[:, in_band].sum(axis=1) / safe_total, 0.0)
291
+
292
+ return Spectra(
293
+ indices=indices,
294
+ freqs=freqs,
295
+ power=power,
296
+ centroid=centroid,
297
+ band_share=band_share,
298
+ total=total,
299
+ )
@@ -0,0 +1,104 @@
1
+ """Small numeric helpers shared by every measure.
2
+
3
+ Two ideas live here. ``dbfs`` never returns ``-inf``, so digital silence is a
4
+ large negative number rather than something that poisons every later average.
5
+ ``piecewise`` turns a measured quantity into a 0-100 score by naming the points
6
+ the scale passes through, which keeps every scoring curve readable as data
7
+ instead of hidden in arithmetic.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import math
13
+ from typing import Sequence, Tuple
14
+
15
+ import numpy as np
16
+
17
+ from .thresholds import DB_FLOOR
18
+
19
+ __all__ = [
20
+ "amplitude_from_dbfs",
21
+ "clamp",
22
+ "dbfs",
23
+ "dbfs_array",
24
+ "piecewise",
25
+ "round_or_none",
26
+ ]
27
+
28
+
29
+ def clamp(value: float, low: float = 0.0, high: float = 100.0) -> float:
30
+ """``value`` pulled inside ``[low, high]``."""
31
+ return float(min(max(float(value), low), high))
32
+
33
+
34
+ def dbfs(amplitude: float) -> float:
35
+ """Amplitude relative to full scale, in dB, with a floor instead of ``-inf``.
36
+
37
+ Args:
38
+ amplitude: a non-negative linear amplitude where 1.0 is full scale.
39
+
40
+ Returns:
41
+ ``20 * log10(amplitude)``, never below :data:`DB_FLOOR`.
42
+ """
43
+ value = float(amplitude)
44
+ if not math.isfinite(value) or value <= 0.0:
45
+ return DB_FLOOR
46
+ return max(DB_FLOOR, 20.0 * math.log10(value))
47
+
48
+
49
+ def dbfs_array(amplitudes: np.ndarray) -> np.ndarray:
50
+ """:func:`dbfs` over an array, vectorised and still free of ``-inf``."""
51
+ values = np.asarray(amplitudes, dtype=np.float64)
52
+ safe = np.where(np.isfinite(values) & (values > 0.0), values, 0.0)
53
+ with np.errstate(divide="ignore", invalid="ignore"):
54
+ out = 20.0 * np.log10(safe)
55
+ return np.maximum(np.where(np.isfinite(out), out, DB_FLOOR), DB_FLOOR)
56
+
57
+
58
+ def amplitude_from_dbfs(level_db: float) -> float:
59
+ """The linear amplitude a dBFS level stands for."""
60
+ return float(10.0 ** (float(level_db) / 20.0))
61
+
62
+
63
+ def piecewise(value: float, points: Sequence[Tuple[float, float]]) -> float:
64
+ """Score ``value`` on a linear scale through named ``(value, score)`` points.
65
+
66
+ The points must be sorted by value. Anything below the first point takes the
67
+ first score, anything above the last takes the last, and everything between
68
+ is interpolated. Writing a curve this way means the thresholds that matter
69
+ are visible at the call site.
70
+
71
+ Args:
72
+ value: the measurement to score.
73
+ points: at least two ``(value, score)`` pairs, ascending by value.
74
+
75
+ Returns:
76
+ A score clamped to 0-100.
77
+
78
+ Raises:
79
+ ValueError: fewer than two points were given.
80
+ """
81
+ if len(points) < 2:
82
+ raise ValueError("piecewise needs at least two points")
83
+ number = float(value)
84
+ if not math.isfinite(number):
85
+ return clamp(points[0][1])
86
+ if number <= points[0][0]:
87
+ return clamp(points[0][1])
88
+ for (low_x, low_y), (high_x, high_y) in zip(points, points[1:]):
89
+ if number <= high_x:
90
+ if high_x == low_x:
91
+ return clamp(high_y)
92
+ fraction = (number - low_x) / (high_x - low_x)
93
+ return clamp(low_y + fraction * (high_y - low_y))
94
+ return clamp(points[-1][1])
95
+
96
+
97
+ def round_or_none(value, digits: int = 4):
98
+ """Round a float for JSON, passing ``None`` and non-finite values through."""
99
+ if value is None:
100
+ return None
101
+ number = float(value)
102
+ if not math.isfinite(number):
103
+ return None
104
+ return round(number, digits)