speechdsp 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
speechdsp/__init__.py ADDED
@@ -0,0 +1,89 @@
1
+ """speechdsp -- signal processing, speech features and evaluation utilities.
2
+
3
+ The package is organised as seven small modules that build on one another:
4
+
5
+ ========================== ======================================================
6
+ :mod:`speechdsp.io` WAV and HTK parameter file readers/writers
7
+ :mod:`speechdsp.framing` frame blocking, overlap-add, frame/sample conversion
8
+ :mod:`speechdsp.spectral` STFT, inverse STFT, spectrograms, CNN-ready images
9
+ :mod:`speechdsp.features` pre-emphasis, mel filterbank, MFCC, deltas, CMVN
10
+ :mod:`speechdsp.vad` short-time energy, zero-crossing rate, endpointing
11
+ :mod:`speechdsp.enhance` log-MMSE and spectral-subtraction noise reduction
12
+ :mod:`speechdsp.metrics` UAR, sensitivity/specificity, cross-validation reports
13
+ ========================== ======================================================
14
+
15
+ Everything is implemented on top of NumPy, SciPy and scikit-learn only; deep
16
+ learning back ends are optional extras and are never imported at package import
17
+ time.
18
+
19
+ Examples
20
+ --------
21
+ >>> import numpy as np
22
+ >>> import speechdsp
23
+ >>> sr = 16000
24
+ >>> x = np.sin(2 * np.pi * 440 * np.arange(sr) / sr)
25
+ >>> speechdsp.mfcc(x, sr).shape[1]
26
+ 13
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ from .enhance import log_mmse, spectral_subtraction
32
+ from .features import (
33
+ cmn,
34
+ cmvn,
35
+ cvn,
36
+ delta,
37
+ hz_to_mel,
38
+ mel_filterbank,
39
+ mel_to_hz,
40
+ mfcc,
41
+ mfcc_with_deltas,
42
+ preemphasis,
43
+ )
44
+ from .framing import enframe, frame_time, frame_to_sample, num_frames, overlap_add
45
+ from .io import read_htk, read_wav, write_htk, write_wav
46
+ from .metrics import confusion_report, cross_val_report, sensitivity_specificity, uar
47
+ from .spectral import istft, power_spectrum, spectrogram_db, spectrogram_image, stft
48
+ from .vad import endpoint_detect, frame_energy, trim_silence, zero_crossing_rate
49
+
50
+ __version__ = "0.1.0"
51
+ __author__ = "RL"
52
+
53
+ __all__ = [
54
+ "__version__",
55
+ "cmn",
56
+ "cmvn",
57
+ "confusion_report",
58
+ "cross_val_report",
59
+ "cvn",
60
+ "delta",
61
+ "endpoint_detect",
62
+ "enframe",
63
+ "frame_energy",
64
+ "frame_time",
65
+ "frame_to_sample",
66
+ "hz_to_mel",
67
+ "istft",
68
+ "log_mmse",
69
+ "mel_filterbank",
70
+ "mel_to_hz",
71
+ "mfcc",
72
+ "mfcc_with_deltas",
73
+ "num_frames",
74
+ "overlap_add",
75
+ "power_spectrum",
76
+ "preemphasis",
77
+ "read_htk",
78
+ "read_wav",
79
+ "sensitivity_specificity",
80
+ "spectral_subtraction",
81
+ "spectrogram_db",
82
+ "spectrogram_image",
83
+ "stft",
84
+ "trim_silence",
85
+ "uar",
86
+ "write_htk",
87
+ "write_wav",
88
+ "zero_crossing_rate",
89
+ ]
speechdsp/enhance.py ADDED
@@ -0,0 +1,223 @@
1
+ """Single-channel speech enhancement in the short-time spectral domain.
2
+
3
+ Both estimators share the same skeleton: analyse the noisy signal with the STFT,
4
+ estimate the noise power from the leading frames (which are assumed to contain
5
+ background noise only), apply a real-valued gain to each spectral magnitude,
6
+ keep the noisy phase, and resynthesise by weighted overlap-add.
7
+
8
+ References
9
+ ----------
10
+ .. [1] Y. Ephraim and D. Malah, "Speech enhancement using a minimum mean-square
11
+ error log-spectral amplitude estimator", *IEEE Trans. ASSP*,
12
+ 33(2):443-445, 1985.
13
+ .. [2] Y. Ephraim and D. Malah, "Speech enhancement using a minimum mean-square
14
+ error short-time spectral amplitude estimator", *IEEE Trans. ASSP*,
15
+ 32(6):1109-1121, 1984.
16
+ .. [3] S. F. Boll, "Suppression of acoustic noise in speech using spectral
17
+ subtraction", *IEEE Trans. ASSP*, 27(2):113-120, 1979.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import logging
23
+
24
+ import numpy as np
25
+ from scipy.special import exp1
26
+
27
+ from .spectral import istft, stft
28
+
29
+ __all__ = [
30
+ "log_mmse",
31
+ "spectral_subtraction",
32
+ ]
33
+
34
+ _LOGGER = logging.getLogger(__name__)
35
+
36
+ #: Target analysis window length in seconds.
37
+ _WINDOW_SECONDS = 0.032
38
+
39
+ #: Smallest positive power used as a denominator floor.
40
+ _POWER_FLOOR = 1e-12
41
+
42
+ #: Clipping range for the a priori SNR, in linear power units (about -25..+40 dB).
43
+ _XI_MIN = 10.0**-2.5
44
+ _XI_MAX = 1.0e4
45
+
46
+ #: ``exp1`` underflows to zero well before this; clipping keeps it warning-free.
47
+ _VK_MAX = 500.0
48
+
49
+
50
+ def _analysis_sizes(sr: int) -> tuple[int, int]:
51
+ """Pick a power-of-two FFT size near 32 ms and a 75 %-overlap hop."""
52
+ n_fft = int(2 ** max(6, round(np.log2(_WINDOW_SECONDS * sr))))
53
+ return n_fft, n_fft // 4
54
+
55
+
56
+ def _noise_power(mag: np.ndarray, noise_frames: int, tag: str) -> np.ndarray:
57
+ """Average power spectrum of the leading frames, used as the noise estimate."""
58
+ usable = int(min(max(noise_frames, 1), mag.shape[0]))
59
+ if usable < noise_frames:
60
+ _LOGGER.warning(
61
+ "%s: only %d frames available for the noise estimate (requested %d)",
62
+ tag,
63
+ usable,
64
+ noise_frames,
65
+ )
66
+ return np.maximum(np.mean(mag[:usable] ** 2, axis=0), _POWER_FLOOR)
67
+
68
+
69
+ def log_mmse(
70
+ x: np.ndarray,
71
+ sr: int,
72
+ noise_frames: int = 6,
73
+ alpha: float = 0.98,
74
+ ) -> np.ndarray:
75
+ """Log-MMSE short-time spectral amplitude estimator.
76
+
77
+ The gain applied to bin ``k`` of frame ``t`` is
78
+
79
+ ``G = xi / (1 + xi) * exp(0.5 * E1(v))``, ``v = xi / (1 + xi) * gamma``,
80
+
81
+ where ``gamma`` is the a posteriori SNR, ``xi`` the a priori SNR tracked by
82
+ the decision-directed rule, and ``E1`` the exponential integral. Compared
83
+ with the plain MMSE-STSA estimator this minimises the error of the *log*
84
+ spectral amplitude, which matches perceptual loudness far better and leaves
85
+ much less musical noise.
86
+
87
+ Parameters
88
+ ----------
89
+ x : numpy.ndarray
90
+ Noisy waveform, flattened to 1-D.
91
+ sr : int
92
+ Sampling rate in Hz.
93
+ noise_frames : int, optional
94
+ Number of leading frames used to estimate the noise power spectrum,
95
+ default 6. They must contain background noise only.
96
+ alpha : float, optional
97
+ Smoothing factor of the decision-directed a priori SNR estimate,
98
+ default 0.98. Must lie in ``[0, 1)``.
99
+
100
+ Returns
101
+ -------
102
+ numpy.ndarray
103
+ Enhanced waveform with the same length as ``x``.
104
+
105
+ Raises
106
+ ------
107
+ ValueError
108
+ If ``alpha`` is outside ``[0, 1)`` or ``sr`` is not positive.
109
+
110
+ Notes
111
+ -----
112
+ The recursion over frames is inherently sequential, but every frame is
113
+ processed as a whole vector over frequency, so the Python loop runs once per
114
+ frame rather than once per bin.
115
+
116
+ References
117
+ ----------
118
+ .. [1] Y. Ephraim and D. Malah, "Speech enhancement using a minimum
119
+ mean-square error log-spectral amplitude estimator", *IEEE Trans.
120
+ ASSP*, 33(2):443-445, 1985.
121
+ """
122
+ if sr <= 0:
123
+ raise ValueError("sr must be positive")
124
+ if not 0.0 <= alpha < 1.0:
125
+ raise ValueError(f"alpha must lie in [0, 1), got {alpha}")
126
+ x = np.asarray(x, dtype=np.float64).ravel()
127
+ n_fft, hop = _analysis_sizes(sr)
128
+ if x.size < n_fft:
129
+ _LOGGER.warning("signal shorter than one analysis window; returned unchanged")
130
+ return x.copy()
131
+
132
+ spec = stft(x, n_fft, hop, window="hann", center=True)
133
+ mag = np.abs(spec)
134
+ phase = np.angle(spec)
135
+ noise_pow = _noise_power(mag, noise_frames, "log_mmse")
136
+
137
+ gains = np.empty_like(mag)
138
+ prev_clean_pow = noise_pow.copy()
139
+ for t in range(mag.shape[0]):
140
+ obs_pow = mag[t] ** 2
141
+ gamma = np.minimum(obs_pow / noise_pow, _XI_MAX)
142
+ xi = alpha * (prev_clean_pow / noise_pow) + (1.0 - alpha) * np.maximum(gamma - 1.0, 0.0)
143
+ xi = np.clip(xi, _XI_MIN, _XI_MAX)
144
+
145
+ ratio = xi / (1.0 + xi)
146
+ vk = np.clip(ratio * gamma, _POWER_FLOOR, _VK_MAX)
147
+ gain = ratio * np.exp(0.5 * exp1(vk))
148
+ gain = np.clip(gain, 0.0, 1.0)
149
+
150
+ gains[t] = gain
151
+ prev_clean_pow = np.maximum((gain * mag[t]) ** 2, _POWER_FLOOR)
152
+
153
+ enhanced = (gains * mag) * np.exp(1j * phase)
154
+ y = istft(enhanced, n_fft, hop, window="hann", center=True)
155
+ return _match_length(y, x.size)
156
+
157
+
158
+ def spectral_subtraction(
159
+ x: np.ndarray,
160
+ sr: int,
161
+ noise_frames: int = 6,
162
+ over_sub: float = 2.0,
163
+ floor: float = 0.002,
164
+ ) -> np.ndarray:
165
+ """Power spectral subtraction with over-subtraction and a spectral floor.
166
+
167
+ Parameters
168
+ ----------
169
+ x : numpy.ndarray
170
+ Noisy waveform, flattened to 1-D.
171
+ sr : int
172
+ Sampling rate in Hz.
173
+ noise_frames : int, optional
174
+ Number of leading noise-only frames used for the noise estimate,
175
+ default 6.
176
+ over_sub : float, optional
177
+ Over-subtraction factor, default 2.0. Values above 1 remove more noise
178
+ at the cost of more speech distortion.
179
+ floor : float, optional
180
+ Spectral floor as a fraction of the noisy power, default 0.002. It
181
+ keeps the residual from collapsing to zero, which is what produces
182
+ "musical noise".
183
+
184
+ Returns
185
+ -------
186
+ numpy.ndarray
187
+ Enhanced waveform with the same length as ``x``.
188
+
189
+ References
190
+ ----------
191
+ .. [1] S. F. Boll, "Suppression of acoustic noise in speech using spectral
192
+ subtraction", *IEEE Trans. ASSP*, 27(2):113-120, 1979.
193
+ .. [2] M. Berouti, R. Schwartz and J. Makhoul, "Enhancement of speech
194
+ corrupted by acoustic noise", *ICASSP*, 4:208-211, 1979.
195
+ """
196
+ if sr <= 0:
197
+ raise ValueError("sr must be positive")
198
+ if over_sub < 0.0:
199
+ raise ValueError("over_sub must be non-negative")
200
+ if not 0.0 <= floor < 1.0:
201
+ raise ValueError(f"floor must lie in [0, 1), got {floor}")
202
+ x = np.asarray(x, dtype=np.float64).ravel()
203
+ n_fft, hop = _analysis_sizes(sr)
204
+ if x.size < n_fft:
205
+ _LOGGER.warning("signal shorter than one analysis window; returned unchanged")
206
+ return x.copy()
207
+
208
+ spec = stft(x, n_fft, hop, window="hann", center=True)
209
+ mag = np.abs(spec)
210
+ noise_pow = _noise_power(mag, noise_frames, "spectral_subtraction")
211
+
212
+ obs_pow = mag**2
213
+ clean_pow = np.maximum(obs_pow - over_sub * noise_pow[None, :], floor * obs_pow)
214
+ enhanced = np.sqrt(clean_pow) * np.exp(1j * np.angle(spec))
215
+ y = istft(enhanced, n_fft, hop, window="hann", center=True)
216
+ return _match_length(y, x.size)
217
+
218
+
219
+ def _match_length(y: np.ndarray, n: int) -> np.ndarray:
220
+ """Trim or zero-pad a resynthesised signal to exactly ``n`` samples."""
221
+ if y.size >= n:
222
+ return np.ascontiguousarray(y[:n])
223
+ return np.pad(y, (0, n - y.size))