speechdsp 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- speechdsp/__init__.py +89 -0
- speechdsp/enhance.py +223 -0
- speechdsp/features.py +473 -0
- speechdsp/framing.py +268 -0
- speechdsp/io.py +302 -0
- speechdsp/metrics.py +341 -0
- speechdsp/py.typed +0 -0
- speechdsp/spectral.py +283 -0
- speechdsp/vad.py +252 -0
- speechdsp-0.1.0.dist-info/METADATA +550 -0
- speechdsp-0.1.0.dist-info/RECORD +14 -0
- speechdsp-0.1.0.dist-info/WHEEL +5 -0
- speechdsp-0.1.0.dist-info/licenses/LICENSE +21 -0
- speechdsp-0.1.0.dist-info/top_level.txt +1 -0
speechdsp/__init__.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""speechdsp -- signal processing, speech features and evaluation utilities.
|
|
2
|
+
|
|
3
|
+
The package is organised as seven small modules that build on one another:
|
|
4
|
+
|
|
5
|
+
========================== ======================================================
|
|
6
|
+
:mod:`speechdsp.io` WAV and HTK parameter file readers/writers
|
|
7
|
+
:mod:`speechdsp.framing` frame blocking, overlap-add, frame/sample conversion
|
|
8
|
+
:mod:`speechdsp.spectral` STFT, inverse STFT, spectrograms, CNN-ready images
|
|
9
|
+
:mod:`speechdsp.features` pre-emphasis, mel filterbank, MFCC, deltas, CMVN
|
|
10
|
+
:mod:`speechdsp.vad` short-time energy, zero-crossing rate, endpointing
|
|
11
|
+
:mod:`speechdsp.enhance` log-MMSE and spectral-subtraction noise reduction
|
|
12
|
+
:mod:`speechdsp.metrics` UAR, sensitivity/specificity, cross-validation reports
|
|
13
|
+
========================== ======================================================
|
|
14
|
+
|
|
15
|
+
Everything is implemented on top of NumPy, SciPy and scikit-learn only; deep
|
|
16
|
+
learning back ends are optional extras and are never imported at package import
|
|
17
|
+
time.
|
|
18
|
+
|
|
19
|
+
Examples
|
|
20
|
+
--------
|
|
21
|
+
>>> import numpy as np
|
|
22
|
+
>>> import speechdsp
|
|
23
|
+
>>> sr = 16000
|
|
24
|
+
>>> x = np.sin(2 * np.pi * 440 * np.arange(sr) / sr)
|
|
25
|
+
>>> speechdsp.mfcc(x, sr).shape[1]
|
|
26
|
+
13
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
from .enhance import log_mmse, spectral_subtraction
|
|
32
|
+
from .features import (
|
|
33
|
+
cmn,
|
|
34
|
+
cmvn,
|
|
35
|
+
cvn,
|
|
36
|
+
delta,
|
|
37
|
+
hz_to_mel,
|
|
38
|
+
mel_filterbank,
|
|
39
|
+
mel_to_hz,
|
|
40
|
+
mfcc,
|
|
41
|
+
mfcc_with_deltas,
|
|
42
|
+
preemphasis,
|
|
43
|
+
)
|
|
44
|
+
from .framing import enframe, frame_time, frame_to_sample, num_frames, overlap_add
|
|
45
|
+
from .io import read_htk, read_wav, write_htk, write_wav
|
|
46
|
+
from .metrics import confusion_report, cross_val_report, sensitivity_specificity, uar
|
|
47
|
+
from .spectral import istft, power_spectrum, spectrogram_db, spectrogram_image, stft
|
|
48
|
+
from .vad import endpoint_detect, frame_energy, trim_silence, zero_crossing_rate
|
|
49
|
+
|
|
50
|
+
__version__ = "0.1.0"
|
|
51
|
+
__author__ = "RL"
|
|
52
|
+
|
|
53
|
+
__all__ = [
|
|
54
|
+
"__version__",
|
|
55
|
+
"cmn",
|
|
56
|
+
"cmvn",
|
|
57
|
+
"confusion_report",
|
|
58
|
+
"cross_val_report",
|
|
59
|
+
"cvn",
|
|
60
|
+
"delta",
|
|
61
|
+
"endpoint_detect",
|
|
62
|
+
"enframe",
|
|
63
|
+
"frame_energy",
|
|
64
|
+
"frame_time",
|
|
65
|
+
"frame_to_sample",
|
|
66
|
+
"hz_to_mel",
|
|
67
|
+
"istft",
|
|
68
|
+
"log_mmse",
|
|
69
|
+
"mel_filterbank",
|
|
70
|
+
"mel_to_hz",
|
|
71
|
+
"mfcc",
|
|
72
|
+
"mfcc_with_deltas",
|
|
73
|
+
"num_frames",
|
|
74
|
+
"overlap_add",
|
|
75
|
+
"power_spectrum",
|
|
76
|
+
"preemphasis",
|
|
77
|
+
"read_htk",
|
|
78
|
+
"read_wav",
|
|
79
|
+
"sensitivity_specificity",
|
|
80
|
+
"spectral_subtraction",
|
|
81
|
+
"spectrogram_db",
|
|
82
|
+
"spectrogram_image",
|
|
83
|
+
"stft",
|
|
84
|
+
"trim_silence",
|
|
85
|
+
"uar",
|
|
86
|
+
"write_htk",
|
|
87
|
+
"write_wav",
|
|
88
|
+
"zero_crossing_rate",
|
|
89
|
+
]
|
speechdsp/enhance.py
ADDED
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
"""Single-channel speech enhancement in the short-time spectral domain.
|
|
2
|
+
|
|
3
|
+
Both estimators share the same skeleton: analyse the noisy signal with the STFT,
|
|
4
|
+
estimate the noise power from the leading frames (which are assumed to contain
|
|
5
|
+
background noise only), apply a real-valued gain to each spectral magnitude,
|
|
6
|
+
keep the noisy phase, and resynthesise by weighted overlap-add.
|
|
7
|
+
|
|
8
|
+
References
|
|
9
|
+
----------
|
|
10
|
+
.. [1] Y. Ephraim and D. Malah, "Speech enhancement using a minimum mean-square
|
|
11
|
+
error log-spectral amplitude estimator", *IEEE Trans. ASSP*,
|
|
12
|
+
33(2):443-445, 1985.
|
|
13
|
+
.. [2] Y. Ephraim and D. Malah, "Speech enhancement using a minimum mean-square
|
|
14
|
+
error short-time spectral amplitude estimator", *IEEE Trans. ASSP*,
|
|
15
|
+
32(6):1109-1121, 1984.
|
|
16
|
+
.. [3] S. F. Boll, "Suppression of acoustic noise in speech using spectral
|
|
17
|
+
subtraction", *IEEE Trans. ASSP*, 27(2):113-120, 1979.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import logging
|
|
23
|
+
|
|
24
|
+
import numpy as np
|
|
25
|
+
from scipy.special import exp1
|
|
26
|
+
|
|
27
|
+
from .spectral import istft, stft
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"log_mmse",
|
|
31
|
+
"spectral_subtraction",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
_LOGGER = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
#: Target analysis window length in seconds.
|
|
37
|
+
_WINDOW_SECONDS = 0.032
|
|
38
|
+
|
|
39
|
+
#: Smallest positive power used as a denominator floor.
|
|
40
|
+
_POWER_FLOOR = 1e-12
|
|
41
|
+
|
|
42
|
+
#: Clipping range for the a priori SNR, in linear power units (about -25..+40 dB).
|
|
43
|
+
_XI_MIN = 10.0**-2.5
|
|
44
|
+
_XI_MAX = 1.0e4
|
|
45
|
+
|
|
46
|
+
#: ``exp1`` underflows to zero well before this; clipping keeps it warning-free.
|
|
47
|
+
_VK_MAX = 500.0
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _analysis_sizes(sr: int) -> tuple[int, int]:
|
|
51
|
+
"""Pick a power-of-two FFT size near 32 ms and a 75 %-overlap hop."""
|
|
52
|
+
n_fft = int(2 ** max(6, round(np.log2(_WINDOW_SECONDS * sr))))
|
|
53
|
+
return n_fft, n_fft // 4
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _noise_power(mag: np.ndarray, noise_frames: int, tag: str) -> np.ndarray:
|
|
57
|
+
"""Average power spectrum of the leading frames, used as the noise estimate."""
|
|
58
|
+
usable = int(min(max(noise_frames, 1), mag.shape[0]))
|
|
59
|
+
if usable < noise_frames:
|
|
60
|
+
_LOGGER.warning(
|
|
61
|
+
"%s: only %d frames available for the noise estimate (requested %d)",
|
|
62
|
+
tag,
|
|
63
|
+
usable,
|
|
64
|
+
noise_frames,
|
|
65
|
+
)
|
|
66
|
+
return np.maximum(np.mean(mag[:usable] ** 2, axis=0), _POWER_FLOOR)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def log_mmse(
|
|
70
|
+
x: np.ndarray,
|
|
71
|
+
sr: int,
|
|
72
|
+
noise_frames: int = 6,
|
|
73
|
+
alpha: float = 0.98,
|
|
74
|
+
) -> np.ndarray:
|
|
75
|
+
"""Log-MMSE short-time spectral amplitude estimator.
|
|
76
|
+
|
|
77
|
+
The gain applied to bin ``k`` of frame ``t`` is
|
|
78
|
+
|
|
79
|
+
``G = xi / (1 + xi) * exp(0.5 * E1(v))``, ``v = xi / (1 + xi) * gamma``,
|
|
80
|
+
|
|
81
|
+
where ``gamma`` is the a posteriori SNR, ``xi`` the a priori SNR tracked by
|
|
82
|
+
the decision-directed rule, and ``E1`` the exponential integral. Compared
|
|
83
|
+
with the plain MMSE-STSA estimator this minimises the error of the *log*
|
|
84
|
+
spectral amplitude, which matches perceptual loudness far better and leaves
|
|
85
|
+
much less musical noise.
|
|
86
|
+
|
|
87
|
+
Parameters
|
|
88
|
+
----------
|
|
89
|
+
x : numpy.ndarray
|
|
90
|
+
Noisy waveform, flattened to 1-D.
|
|
91
|
+
sr : int
|
|
92
|
+
Sampling rate in Hz.
|
|
93
|
+
noise_frames : int, optional
|
|
94
|
+
Number of leading frames used to estimate the noise power spectrum,
|
|
95
|
+
default 6. They must contain background noise only.
|
|
96
|
+
alpha : float, optional
|
|
97
|
+
Smoothing factor of the decision-directed a priori SNR estimate,
|
|
98
|
+
default 0.98. Must lie in ``[0, 1)``.
|
|
99
|
+
|
|
100
|
+
Returns
|
|
101
|
+
-------
|
|
102
|
+
numpy.ndarray
|
|
103
|
+
Enhanced waveform with the same length as ``x``.
|
|
104
|
+
|
|
105
|
+
Raises
|
|
106
|
+
------
|
|
107
|
+
ValueError
|
|
108
|
+
If ``alpha`` is outside ``[0, 1)`` or ``sr`` is not positive.
|
|
109
|
+
|
|
110
|
+
Notes
|
|
111
|
+
-----
|
|
112
|
+
The recursion over frames is inherently sequential, but every frame is
|
|
113
|
+
processed as a whole vector over frequency, so the Python loop runs once per
|
|
114
|
+
frame rather than once per bin.
|
|
115
|
+
|
|
116
|
+
References
|
|
117
|
+
----------
|
|
118
|
+
.. [1] Y. Ephraim and D. Malah, "Speech enhancement using a minimum
|
|
119
|
+
mean-square error log-spectral amplitude estimator", *IEEE Trans.
|
|
120
|
+
ASSP*, 33(2):443-445, 1985.
|
|
121
|
+
"""
|
|
122
|
+
if sr <= 0:
|
|
123
|
+
raise ValueError("sr must be positive")
|
|
124
|
+
if not 0.0 <= alpha < 1.0:
|
|
125
|
+
raise ValueError(f"alpha must lie in [0, 1), got {alpha}")
|
|
126
|
+
x = np.asarray(x, dtype=np.float64).ravel()
|
|
127
|
+
n_fft, hop = _analysis_sizes(sr)
|
|
128
|
+
if x.size < n_fft:
|
|
129
|
+
_LOGGER.warning("signal shorter than one analysis window; returned unchanged")
|
|
130
|
+
return x.copy()
|
|
131
|
+
|
|
132
|
+
spec = stft(x, n_fft, hop, window="hann", center=True)
|
|
133
|
+
mag = np.abs(spec)
|
|
134
|
+
phase = np.angle(spec)
|
|
135
|
+
noise_pow = _noise_power(mag, noise_frames, "log_mmse")
|
|
136
|
+
|
|
137
|
+
gains = np.empty_like(mag)
|
|
138
|
+
prev_clean_pow = noise_pow.copy()
|
|
139
|
+
for t in range(mag.shape[0]):
|
|
140
|
+
obs_pow = mag[t] ** 2
|
|
141
|
+
gamma = np.minimum(obs_pow / noise_pow, _XI_MAX)
|
|
142
|
+
xi = alpha * (prev_clean_pow / noise_pow) + (1.0 - alpha) * np.maximum(gamma - 1.0, 0.0)
|
|
143
|
+
xi = np.clip(xi, _XI_MIN, _XI_MAX)
|
|
144
|
+
|
|
145
|
+
ratio = xi / (1.0 + xi)
|
|
146
|
+
vk = np.clip(ratio * gamma, _POWER_FLOOR, _VK_MAX)
|
|
147
|
+
gain = ratio * np.exp(0.5 * exp1(vk))
|
|
148
|
+
gain = np.clip(gain, 0.0, 1.0)
|
|
149
|
+
|
|
150
|
+
gains[t] = gain
|
|
151
|
+
prev_clean_pow = np.maximum((gain * mag[t]) ** 2, _POWER_FLOOR)
|
|
152
|
+
|
|
153
|
+
enhanced = (gains * mag) * np.exp(1j * phase)
|
|
154
|
+
y = istft(enhanced, n_fft, hop, window="hann", center=True)
|
|
155
|
+
return _match_length(y, x.size)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def spectral_subtraction(
|
|
159
|
+
x: np.ndarray,
|
|
160
|
+
sr: int,
|
|
161
|
+
noise_frames: int = 6,
|
|
162
|
+
over_sub: float = 2.0,
|
|
163
|
+
floor: float = 0.002,
|
|
164
|
+
) -> np.ndarray:
|
|
165
|
+
"""Power spectral subtraction with over-subtraction and a spectral floor.
|
|
166
|
+
|
|
167
|
+
Parameters
|
|
168
|
+
----------
|
|
169
|
+
x : numpy.ndarray
|
|
170
|
+
Noisy waveform, flattened to 1-D.
|
|
171
|
+
sr : int
|
|
172
|
+
Sampling rate in Hz.
|
|
173
|
+
noise_frames : int, optional
|
|
174
|
+
Number of leading noise-only frames used for the noise estimate,
|
|
175
|
+
default 6.
|
|
176
|
+
over_sub : float, optional
|
|
177
|
+
Over-subtraction factor, default 2.0. Values above 1 remove more noise
|
|
178
|
+
at the cost of more speech distortion.
|
|
179
|
+
floor : float, optional
|
|
180
|
+
Spectral floor as a fraction of the noisy power, default 0.002. It
|
|
181
|
+
keeps the residual from collapsing to zero, which is what produces
|
|
182
|
+
"musical noise".
|
|
183
|
+
|
|
184
|
+
Returns
|
|
185
|
+
-------
|
|
186
|
+
numpy.ndarray
|
|
187
|
+
Enhanced waveform with the same length as ``x``.
|
|
188
|
+
|
|
189
|
+
References
|
|
190
|
+
----------
|
|
191
|
+
.. [1] S. F. Boll, "Suppression of acoustic noise in speech using spectral
|
|
192
|
+
subtraction", *IEEE Trans. ASSP*, 27(2):113-120, 1979.
|
|
193
|
+
.. [2] M. Berouti, R. Schwartz and J. Makhoul, "Enhancement of speech
|
|
194
|
+
corrupted by acoustic noise", *ICASSP*, 4:208-211, 1979.
|
|
195
|
+
"""
|
|
196
|
+
if sr <= 0:
|
|
197
|
+
raise ValueError("sr must be positive")
|
|
198
|
+
if over_sub < 0.0:
|
|
199
|
+
raise ValueError("over_sub must be non-negative")
|
|
200
|
+
if not 0.0 <= floor < 1.0:
|
|
201
|
+
raise ValueError(f"floor must lie in [0, 1), got {floor}")
|
|
202
|
+
x = np.asarray(x, dtype=np.float64).ravel()
|
|
203
|
+
n_fft, hop = _analysis_sizes(sr)
|
|
204
|
+
if x.size < n_fft:
|
|
205
|
+
_LOGGER.warning("signal shorter than one analysis window; returned unchanged")
|
|
206
|
+
return x.copy()
|
|
207
|
+
|
|
208
|
+
spec = stft(x, n_fft, hop, window="hann", center=True)
|
|
209
|
+
mag = np.abs(spec)
|
|
210
|
+
noise_pow = _noise_power(mag, noise_frames, "spectral_subtraction")
|
|
211
|
+
|
|
212
|
+
obs_pow = mag**2
|
|
213
|
+
clean_pow = np.maximum(obs_pow - over_sub * noise_pow[None, :], floor * obs_pow)
|
|
214
|
+
enhanced = np.sqrt(clean_pow) * np.exp(1j * np.angle(spec))
|
|
215
|
+
y = istft(enhanced, n_fft, hop, window="hann", center=True)
|
|
216
|
+
return _match_length(y, x.size)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _match_length(y: np.ndarray, n: int) -> np.ndarray:
|
|
220
|
+
"""Trim or zero-pad a resynthesised signal to exactly ``n`` samples."""
|
|
221
|
+
if y.size >= n:
|
|
222
|
+
return np.ascontiguousarray(y[:n])
|
|
223
|
+
return np.pad(y, (0, n - y.size))
|