sonore 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sonore/__init__.py ADDED
@@ -0,0 +1,194 @@
1
+ """sonore: signals and stimuli for auditory research.
2
+
3
+ Typical use in a notebook::
4
+
5
+ import sonore as so
6
+ from sonore import dB
7
+
8
+ x = so.pure_tone(1.0, 44100, 440).ramp(0.01)
9
+ y = x + (so.gaussian_noise(1.0, 44100, rng=0) - 10 * dB) # tone at +10 dB SNR
10
+ y # displays an audio player
11
+ so.overview(y) # waveform, spectrum, spectrogram, modulation spectrum
12
+ """
13
+
14
+ from sonore import texture
15
+ from sonore.analysis.cepstrum import Cepstrum
16
+ from sonore.analysis.envelopes import Envelope, Envelopes
17
+ from sonore.analysis.filterbank import (
18
+ CosineFilterbank,
19
+ ERBFilterbank,
20
+ GammatoneFilterbank,
21
+ MorletFilterbank,
22
+ OctaveFilterbank,
23
+ Subbands,
24
+ noise_vocode,
25
+ subbands,
26
+ )
27
+ from sonore.analysis.frames import Filterbank, Frame, GaborFrame, TVGaborFrame
28
+ from sonore.analysis.modulation import (
29
+ ConstantQModulationFilterbank,
30
+ ModulationFilterbank,
31
+ OctaveModulationFilterbank,
32
+ )
33
+ from sonore.analysis.representations import (
34
+ STFT,
35
+ TVSTFT,
36
+ Mask,
37
+ ModulationSpectrum,
38
+ ReassignedSpectrogram,
39
+ Spectrum,
40
+ TFPower,
41
+ ideal_binary_mask,
42
+ ideal_ratio_mask,
43
+ long_term_spectrum,
44
+ reassigned_spectrogram,
45
+ tandem_power,
46
+ )
47
+ from sonore.core.sound import Sound, load
48
+ from sonore.core.units import Decibels, dB
49
+ from sonore.core.utils import amp_to_db, db_to_amp, erb_to_freq, freq_to_erb, rms
50
+ from sonore.plotting import overview
51
+ from sonore.signals.generators import (
52
+ correlated_noise,
53
+ exponential_chirp,
54
+ gaussian_noise,
55
+ harmonic_complex,
56
+ iterated_ripple_noise,
57
+ linear_chirp,
58
+ pulse_train,
59
+ pure_tone,
60
+ sawtooth_wave,
61
+ schroeder_complex,
62
+ silence,
63
+ square_wave,
64
+ )
65
+ from sonore.signals.processing import (
66
+ amplitude_modulate,
67
+ bandpass,
68
+ butter_filter,
69
+ concat,
70
+ match_channels,
71
+ match_fs,
72
+ mix,
73
+ normalize,
74
+ pad,
75
+ relative_db,
76
+ truncate,
77
+ )
78
+ from sonore.stimuli.binaural import (
79
+ InterauralCues,
80
+ apply_itd_ild,
81
+ interaural_cues,
82
+ oscor,
83
+ phasewarp,
84
+ simple_bir,
85
+ )
86
+ from sonore.stimuli.hrir_data import load_hrirs
87
+ from sonore.stimuli.phasevocoder import PVAnalysis, pitch_shift, pv_analyze, time_stretch
88
+ from sonore.stimuli.reverb import band_rt60s, measure_rt60, synth_ir
89
+ from sonore.stimuli.ripples import DynamicRipple, Ripple, RippleSum, ripple_sound
90
+ from sonore.stimuli.spatialization import (
91
+ HRIRSet,
92
+ circular_trajectory,
93
+ distance_gain_db,
94
+ hcc_to_rect,
95
+ linear_trajectory,
96
+ move_sound,
97
+ rect_to_hcc,
98
+ spatialize,
99
+ )
100
+
101
+ __version__ = "0.3.0"
102
+
103
+ __all__ = [
104
+ "texture",
105
+ "ConstantQModulationFilterbank",
106
+ "ModulationFilterbank",
107
+ "OctaveModulationFilterbank",
108
+ "measure_rt60",
109
+ "band_rt60s",
110
+ "CosineFilterbank",
111
+ "Filterbank",
112
+ "Frame",
113
+ "GaborFrame",
114
+ "GammatoneFilterbank",
115
+ "MorletFilterbank",
116
+ "TVGaborFrame",
117
+ "TVSTFT",
118
+ "Cepstrum",
119
+ "TFPower",
120
+ "tandem_power",
121
+ "ReassignedSpectrogram",
122
+ "reassigned_spectrogram",
123
+ "Envelopes",
124
+ "Envelope",
125
+ "ripple_sound",
126
+ "RippleSum",
127
+ "Ripple",
128
+ "DynamicRipple",
129
+ "OctaveFilterbank",
130
+ "time_stretch",
131
+ "pv_analyze",
132
+ "pitch_shift",
133
+ "PVAnalysis",
134
+ "amp_to_db",
135
+ "amplitude_modulate",
136
+ "apply_itd_ild",
137
+ "bandpass",
138
+ "butter_filter",
139
+ "circular_trajectory",
140
+ "concat",
141
+ "correlated_noise",
142
+ "dB",
143
+ "db_to_amp",
144
+ "Decibels",
145
+ "distance_gain_db",
146
+ "erb_to_freq",
147
+ "ERBFilterbank",
148
+ "exponential_chirp",
149
+ "freq_to_erb",
150
+ "gaussian_noise",
151
+ "harmonic_complex",
152
+ "hcc_to_rect",
153
+ "HRIRSet",
154
+ "ideal_binary_mask",
155
+ "ideal_ratio_mask",
156
+ "interaural_cues",
157
+ "InterauralCues",
158
+ "iterated_ripple_noise",
159
+ "linear_chirp",
160
+ "linear_trajectory",
161
+ "load",
162
+ "load_hrirs",
163
+ "long_term_spectrum",
164
+ "Mask",
165
+ "match_channels",
166
+ "match_fs",
167
+ "mix",
168
+ "ModulationSpectrum",
169
+ "move_sound",
170
+ "noise_vocode",
171
+ "normalize",
172
+ "oscor",
173
+ "overview",
174
+ "pad",
175
+ "phasewarp",
176
+ "pulse_train",
177
+ "pure_tone",
178
+ "rect_to_hcc",
179
+ "relative_db",
180
+ "rms",
181
+ "sawtooth_wave",
182
+ "schroeder_complex",
183
+ "silence",
184
+ "simple_bir",
185
+ "Sound",
186
+ "spatialize",
187
+ "Spectrum",
188
+ "square_wave",
189
+ "STFT",
190
+ "Subbands",
191
+ "subbands",
192
+ "synth_ir",
193
+ "truncate",
194
+ ]
@@ -0,0 +1,5 @@
1
+ """Taking sounds apart: frames and filterbanks, the representations they produce
2
+ (spectra, STFTs, masks, modulation spectra), and envelopes.
3
+
4
+ Imports from :mod:`sonore.core` and :mod:`sonore.signals`.
5
+ """
@@ -0,0 +1,198 @@
1
+ """The real cepstrum of a short-time Fourier transform: liftering,
2
+ resynthesis with the original or minimum phase, and cepstral F0."""
3
+
4
+ from __future__ import annotations
5
+
6
+ from collections.abc import Sequence
7
+
8
+ import numpy as np
9
+
10
+ from sonore.analysis.representations import STFT, TVSTFT
11
+ from sonore.core.sound import Sound
12
+
13
+ __all__ = ["Cepstrum"]
14
+
15
+
16
+ class Cepstrum:
17
+ """The real cepstrum of each frame of an :class:`~sonore.analysis.representations.STFT`
18
+ or :class:`~sonore.analysis.representations.TVSTFT`.
19
+
20
+ For a frame with spectrum ``X[k]`` on ``n_fft`` bins, the cepstrum is
21
+ ``c[n] = IDFT(ln |X[k]|)`` at quefrency ``n / fs`` seconds. It is real and
22
+ even in ``n`` because the sound is real, so only ``n = 0 .. n_fft // 2``
23
+ is stored: ``data`` has shape ``(n_channels, n_fft // 2 + 1, n_frames)``
24
+ on quefrencies :attr:`q` [s] and frame times :attr:`t` [s].
25
+
26
+ The natural log is used, so ``exp`` undoes it exactly: without liftering,
27
+ :meth:`to_sound` gives back the analyzed sound. Scaling the sound changes
28
+ only ``c[0]``. Before the log, magnitudes are floored at ``floor_db``
29
+ below the channel's largest magnitude over all frames, so a frame of
30
+ digital silence has a flat log spectrum rather than ``-inf``; real
31
+ recordings stay far above the default. Only the real cepstrum is
32
+ provided: the complex cepstrum needs phase unwrapping, and the minimum
33
+ phase comes from the real one.
34
+
35
+ Parameters
36
+ ----------
37
+ coefs
38
+ The coefficients to take the cepstrum of. They are kept (as
39
+ :attr:`source`) for their phase, frame and times.
40
+ floor_db
41
+ The floor, in dB below each channel's maximum.
42
+ """
43
+
44
+ def __init__(self, coefs: STFT | TVSTFT, floor_db: float = -200.0):
45
+ if not isinstance(coefs, (STFT, TVSTFT)):
46
+ raise TypeError(f"expected an STFT or TVSTFT, not {type(coefs).__name__}")
47
+ self.source = coefs
48
+ self.fs = coefs.fs
49
+ self.n_fft = _n_fft(coefs)
50
+ mag = np.abs(coefs.data)
51
+ peak = mag.max(axis=(1, 2), keepdims=True)
52
+ floor = np.where(peak > 0, peak * 10 ** (floor_db / 20), np.finfo(float).tiny)
53
+ log_mag = np.log(np.maximum(mag, floor))
54
+ self.data = np.fft.irfft(log_mag, n=self.n_fft, axis=1)[:, : self.n_fft // 2 + 1]
55
+
56
+ @classmethod
57
+ def _from(cls, template: Cepstrum, data: np.ndarray) -> Cepstrum:
58
+ new = cls.__new__(cls)
59
+ new.source, new.fs, new.n_fft, new.data = template.source, template.fs, template.n_fft, data
60
+ return new
61
+
62
+ def __repr__(self) -> str:
63
+ c, q, t = self.data.shape
64
+ return f"Cepstrum({q} quefrencies x {t} frames, {c} ch, up to {self.q[-1] * 1e3:.1f} ms)"
65
+
66
+ @property
67
+ def q(self) -> np.ndarray:
68
+ """Quefrencies [s]."""
69
+ return np.arange(self.data.shape[1]) / self.fs
70
+
71
+ @property
72
+ def t(self) -> np.ndarray:
73
+ """Frame center times [s], those of :attr:`source`."""
74
+ return self.source.t
75
+
76
+ def _full(self) -> np.ndarray:
77
+ """All ``n_fft`` quefrencies, mirrored from the stored half."""
78
+ n_half = self.data.shape[1]
79
+ mirror = self.data[:, 1 : self.n_fft - n_half + 1][:, ::-1]
80
+ return np.concatenate([self.data, mirror], axis=1)
81
+
82
+ def lifter(self, cutoff: float | Sequence[float], keep: str = "low") -> Cepstrum:
83
+ """A rectangular lifter. ``keep="low"`` keeps the quefrencies below
84
+ ``cutoff`` [s], the smooth spectral envelope; ``"high"`` keeps the
85
+ rest, the fine structure such as the harmonics. ``cutoff`` is one
86
+ value or one per frame, so it can follow an F0 track (half a period
87
+ separates the envelope from the harmonics)."""
88
+ cut = np.asarray(cutoff, dtype=float)
89
+ n_frames = self.data.shape[2]
90
+ if cut.ndim > 1 or (cut.ndim == 1 and len(cut) != n_frames):
91
+ raise ValueError(f"cutoff must be a scalar or have one value per frame ({n_frames})")
92
+ below = self.q[:, None] < np.broadcast_to(cut, (n_frames,))[None, :]
93
+ if keep == "low":
94
+ mask = below
95
+ elif keep == "high":
96
+ mask = ~below
97
+ else:
98
+ raise ValueError(f"keep must be 'low' or 'high', not {keep!r}")
99
+ return Cepstrum._from(self, self.data * mask)
100
+
101
+ def envelope(self) -> np.ndarray:
102
+ """``exp(DFT(c))``: the magnitude spectrum this cepstrum stands for,
103
+ shape ``(n_channels, n_freqs, n_frames)`` on the source's frequencies.
104
+ After a low lifter it is the cepstral spectral envelope, which follows
105
+ the shape of the true envelope but sits a few dB below the harmonic
106
+ peaks, because the lifter averages the peaks with the dips between them."""
107
+ return np.exp(np.fft.rfft(self._full(), axis=1).real)
108
+
109
+ def to_stft(self, phase: str = "original") -> STFT | TVSTFT:
110
+ """Coefficients of the source's type and frame with :meth:`envelope`
111
+ as magnitude.
112
+
113
+ ``phase="original"`` takes the phase from :attr:`source`, so an
114
+ unliftered cepstrum returns the source's coefficients. ``"minimum"``
115
+ uses the minimum phase for that magnitude, from the folded cepstrum
116
+ (``c[0]`` kept, ``2 c[n]`` up to ``n_fft / 2``, zero beyond). Each
117
+ frame's response then starts at the frame's phase reference, the
118
+ middle of its window. The fold is exact up to the time aliasing of
119
+ the cepstrum on ``n_fft`` bins, which is negligible once ``n_fft``
120
+ is several times the response's length.
121
+ """
122
+ if phase == "original":
123
+ data = self.envelope() * np.exp(1j * np.angle(self.source.data))
124
+ elif phase == "minimum":
125
+ n_mid = (self.n_fft + 1) // 2
126
+ folded = np.zeros((self.data.shape[0], self.n_fft, self.data.shape[2]))
127
+ folded[:, 0] = self.data[:, 0]
128
+ folded[:, 1:n_mid] = 2 * self.data[:, 1:n_mid]
129
+ if self.n_fft % 2 == 0:
130
+ folded[:, n_mid] = self.data[:, n_mid]
131
+ data = np.exp(np.fft.rfft(folded, axis=1))
132
+ else:
133
+ raise ValueError(f"phase must be 'original' or 'minimum', not {phase!r}")
134
+ return type(self.source)._from(self.source, data)
135
+
136
+ def to_sound(self, phase: str = "original") -> Sound:
137
+ """:meth:`to_stft` synthesized by the source's frame: exact for an
138
+ unliftered cepstrum with the original phase, and otherwise the
139
+ least-squares signal for those coefficients."""
140
+ return self.to_stft(phase).to_sound()
141
+
142
+ def f0(
143
+ self, f_lo: float = 75.0, f_hi: float = 400.0, threshold: float = 0.1
144
+ ) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
145
+ """Classic cepstral F0 (Noll, 1967): the largest cepstral peak between
146
+ quefrencies ``1/f_hi`` and ``1/f_lo``, refined by a parabola through it
147
+ and its neighbors.
148
+
149
+ Returns ``(t, f0, peak)``: frame times [s], and F0 [Hz] and the peak's
150
+ height for each channel and frame, shape ``(n_channels, n_frames)``.
151
+ F0 is 0 where the peak is below ``threshold``, a crude voicing rule.
152
+
153
+ A periodic sound puts ripples in the log spectrum, one per harmonic,
154
+ and they are resolved only if the window holds about three periods: at
155
+ two periods, many frames come out an octave off. So every window must
156
+ be at least ``3 / f_lo`` long, or this raises. On a male spoken
157
+ sentence (CMU ARCTIC ``bdl``, 40 ms Hann frames), the result agrees
158
+ with WORLD's Harvest within 5% on about 79% of the frames Harvest
159
+ calls voiced, and on 95% of those whose peak also exceeds 0.1. It is a
160
+ baseline, not an F0 tracker: each frame is judged alone.
161
+ """
162
+ if not 0 < f_lo < f_hi < self.fs / 2:
163
+ raise ValueError(f"need 0 < f_lo < f_hi < fs/2, got f_lo={f_lo:g}, f_hi={f_hi:g}")
164
+ shortest = _shortest_window(self.source) / self.fs
165
+ if shortest < 3 / f_lo:
166
+ raise ValueError(
167
+ f"the shortest window is {shortest * 1e3:.1f} ms; cepstral F0 down to f_lo={f_lo:g} Hz "
168
+ f"needs windows of at least three periods, {3 / f_lo * 1e3:.1f} ms"
169
+ )
170
+ q_lo, q_hi = max(int(np.floor(self.fs / f_hi)), 1), int(np.ceil(self.fs / f_lo))
171
+ q_hi = min(q_hi, self.data.shape[1] - 2)
172
+ k = q_lo + np.argmax(self.data[:, q_lo : q_hi + 1], axis=1)[:, None, :]
173
+ y0, y1, y2 = (np.take_along_axis(self.data, k + d, axis=1)[:, 0] for d in (-1, 0, 1))
174
+ denom = y0 - 2 * y1 + y2
175
+ with np.errstate(divide="ignore", invalid="ignore"):
176
+ shift = np.where(denom != 0, 0.5 * (y0 - y2) / denom, 0.0)
177
+ f0 = self.fs / (k[:, 0] + shift)
178
+ return self.t, np.where(y1 >= threshold, f0, 0.0), y1
179
+
180
+ def plot(self, ax=None, channel: int = 0, **kwargs):
181
+ """Cepstrum against time and quefrency in ms (see
182
+ :func:`~sonore.plotting.plot_cepstrum`)."""
183
+ from sonore.plotting import plot_cepstrum
184
+
185
+ return plot_cepstrum(self, ax=ax, channel=channel, **kwargs)
186
+
187
+
188
+ def _n_fft(coefs: STFT | TVSTFT) -> int:
189
+ if isinstance(coefs, STFT):
190
+ return int(coefs.sft.mfft)
191
+ return int(coefs.frame.layout(coefs.fs).n_fft)
192
+
193
+
194
+ def _shortest_window(coefs: STFT | TVSTFT) -> int:
195
+ """Shortest window [samples]."""
196
+ if isinstance(coefs, STFT):
197
+ return int(coefs.sft.m_num)
198
+ return int(coefs.frame.layout(coefs.fs).lengths.min())