mea8000-encoder 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mea8000/__init__.py +6 -0
- mea8000/__main__.py +7 -0
- mea8000/analysis.py +326 -0
- mea8000/cli.py +287 -0
- mea8000/codec.py +221 -0
- mea8000/encoder.py +793 -0
- mea8000/fastsim.py +260 -0
- mea8000/filters.py +114 -0
- mea8000/merge.py +129 -0
- mea8000/profile.py +316 -0
- mea8000/profiles/compact.toml +11 -0
- mea8000/profiles/philips.toml +10 -0
- mea8000/profiles/thomson.toml +21 -0
- mea8000/report.py +254 -0
- mea8000/sim.py +427 -0
- mea8000/tables.py +34 -0
- mea8000/tuning.py +96 -0
- mea8000/wav.py +83 -0
- mea8000_encoder-0.1.0.dist-info/METADATA +142 -0
- mea8000_encoder-0.1.0.dist-info/RECORD +25 -0
- mea8000_encoder-0.1.0.dist-info/WHEEL +5 -0
- mea8000_encoder-0.1.0.dist-info/entry_points.txt +2 -0
- mea8000_encoder-0.1.0.dist-info/licenses/LICENSE +21 -0
- mea8000_encoder-0.1.0.dist-info/licenses/THIRD_PARTY.md +41 -0
- mea8000_encoder-0.1.0.dist-info/top_level.txt +1 -0
mea8000/__init__.py
ADDED
mea8000/__main__.py
ADDED
mea8000/analysis.py
ADDED
|
@@ -0,0 +1,326 @@
|
|
|
1
|
+
"""Target analysis for the encoder: pitch, voicing, energy and spectral envelope per 8 ms.
|
|
2
|
+
|
|
3
|
+
Everything works at 8 kHz, the rate of the chip's filter bank: nothing above 4 kHz can be
|
|
4
|
+
represented, so the input is band-limited first.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
from scipy.signal import resample_poly
|
|
13
|
+
|
|
14
|
+
RATE = 8000
|
|
15
|
+
HOP = 64 # 8 ms, the shortest frame of the chip
|
|
16
|
+
ENV_WIN = 256 # 32 ms analysis window for the envelope
|
|
17
|
+
PITCH_WIN = 224 # 28 ms window for the difference function
|
|
18
|
+
MAX_CANDIDATES = 6
|
|
19
|
+
F0_MIN, F0_MAX = 60.0, 510.0
|
|
20
|
+
N_BINS = 129 # envelope grid: 0..4000 Hz in 31.25 Hz steps
|
|
21
|
+
ENV_FREQS = np.linspace(0.0, RATE / 2, N_BINS)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class Analysis:
|
|
26
|
+
rate: int
|
|
27
|
+
hop: int
|
|
28
|
+
f0: np.ndarray # Hz per frame, nan when unvoiced
|
|
29
|
+
voiced: np.ndarray # bool per frame
|
|
30
|
+
periodicity: np.ndarray # 1 - min(cmndf), per frame
|
|
31
|
+
energy_db: np.ndarray # RMS of the frame, dB re full scale
|
|
32
|
+
envelope_db: np.ndarray # (frames, N_BINS) smoothed log magnitude
|
|
33
|
+
lpc: np.ndarray # (frames, order + 1) predictor coefficients
|
|
34
|
+
hnr_db: np.ndarray | None = None # harmonic-to-noise ratio at the tracked pitch
|
|
35
|
+
|
|
36
|
+
@property
|
|
37
|
+
def n_frames(self) -> int:
|
|
38
|
+
return len(self.f0)
|
|
39
|
+
|
|
40
|
+
def times(self) -> np.ndarray:
|
|
41
|
+
return np.arange(self.n_frames) * self.hop / self.rate
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def to_8k(x: np.ndarray, rate: float) -> np.ndarray:
|
|
45
|
+
"""Mono float signal at 8 kHz; `rate` may be fractional (clock compensation)."""
|
|
46
|
+
x = np.asarray(x, dtype=np.float64)
|
|
47
|
+
if x.dtype.kind in "iu" or np.abs(x).max() > 1.5:
|
|
48
|
+
x = x / 32768.0
|
|
49
|
+
rate = int(round(rate))
|
|
50
|
+
if rate == RATE:
|
|
51
|
+
return x
|
|
52
|
+
from math import gcd
|
|
53
|
+
|
|
54
|
+
g = gcd(rate, RATE)
|
|
55
|
+
return resample_poly(x, RATE // g, rate // g)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def n_frames_for(n_samples: int, hop: int = HOP) -> int:
|
|
59
|
+
return max(1, int(np.ceil(n_samples / hop)))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _frame(x: np.ndarray, start: int, win: int) -> np.ndarray:
|
|
63
|
+
"""Window of `win` samples centred on the frame that starts at `start` (zero padded)."""
|
|
64
|
+
c = start + HOP // 2
|
|
65
|
+
a = c - win // 2
|
|
66
|
+
seg = np.zeros(win)
|
|
67
|
+
lo, hi = max(a, 0), min(a + win, len(x))
|
|
68
|
+
if hi > lo:
|
|
69
|
+
seg[lo - a: hi - a] = x[lo:hi]
|
|
70
|
+
return seg
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# ------------------------------------------------------------------ envelope (Burg LPC)
|
|
74
|
+
|
|
75
|
+
def burg(x: np.ndarray, order: int) -> tuple[np.ndarray, float]:
|
|
76
|
+
"""Burg's method. Returns a[0..order] with a[0] = 1 and the prediction error power."""
|
|
77
|
+
x = np.asarray(x, dtype=np.float64)
|
|
78
|
+
n = len(x)
|
|
79
|
+
a = np.zeros(order + 1)
|
|
80
|
+
a[0] = 1.0
|
|
81
|
+
err = float(np.dot(x, x)) / n
|
|
82
|
+
f = x[1:].copy() # forward prediction errors
|
|
83
|
+
b = x[:-1].copy() # backward prediction errors
|
|
84
|
+
for m in range(1, order + 1):
|
|
85
|
+
den = float(np.dot(f, f) + np.dot(b, b))
|
|
86
|
+
k = -2.0 * float(np.dot(f, b)) / den if den > 0 else 0.0
|
|
87
|
+
a_prev = a.copy()
|
|
88
|
+
for i in range(1, m):
|
|
89
|
+
a[i] = a_prev[i] + k * a_prev[m - i]
|
|
90
|
+
a[m] = k
|
|
91
|
+
f, b = f[1:] + k * b[1:], b[:-1] + k * f[:-1]
|
|
92
|
+
err *= (1.0 - k * k)
|
|
93
|
+
return a, err
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def lpc_envelope_db(a: np.ndarray, gain: float, freqs: np.ndarray = ENV_FREQS) -> np.ndarray:
|
|
97
|
+
w = 2 * np.pi * freqs / RATE
|
|
98
|
+
k = np.arange(len(a))
|
|
99
|
+
denom = np.abs(np.exp(-1j * np.outer(w, k)) @ a)
|
|
100
|
+
return 20 * np.log10(gain / np.maximum(denom, 1e-9) + 1e-12)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
# ------------------------------------------------------------------ pitch (YIN)
|
|
104
|
+
|
|
105
|
+
def cmndf(seg: np.ndarray, lag_max: int) -> np.ndarray:
|
|
106
|
+
"""Cumulative mean normalised difference function of YIN for lags 0..lag_max."""
|
|
107
|
+
n = len(seg) - lag_max
|
|
108
|
+
x = seg[:n]
|
|
109
|
+
d = np.empty(lag_max + 1)
|
|
110
|
+
e0 = float(np.dot(x, x))
|
|
111
|
+
# d(tau) = sum (x[j] - x[j+tau])^2 over j < n
|
|
112
|
+
csum = np.concatenate([[0.0], np.cumsum(seg * seg)])
|
|
113
|
+
for tau in range(lag_max + 1):
|
|
114
|
+
y = seg[tau: tau + n]
|
|
115
|
+
d[tau] = e0 + (csum[tau + n] - csum[tau]) - 2.0 * float(np.dot(x, y))
|
|
116
|
+
out = np.ones(lag_max + 1)
|
|
117
|
+
run = np.cumsum(d[1:])
|
|
118
|
+
with np.errstate(divide="ignore", invalid="ignore"):
|
|
119
|
+
out[1:] = np.where(run > 0, d[1:] * np.arange(1, lag_max + 1) / run, 1.0)
|
|
120
|
+
return out
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _parabolic(y: np.ndarray, i: int) -> float:
|
|
124
|
+
if i <= 0 or i >= len(y) - 1:
|
|
125
|
+
return float(i)
|
|
126
|
+
a, b, c = y[i - 1], y[i], y[i + 1]
|
|
127
|
+
den = a - 2 * b + c
|
|
128
|
+
return float(i) if den == 0 else float(i + 0.5 * (a - c) / den)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# the subharmonic threshold and penalty live in tuning.current (subharmonic_threshold, _penalty)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _candidates(c: np.ndarray, lag_min: int) -> tuple[np.ndarray, np.ndarray]:
|
|
135
|
+
"""Local minima of the CMNDF (refined lag, cost), best first, at most MAX_CANDIDATES.
|
|
136
|
+
|
|
137
|
+
YIN's protection against pitch halving: on a periodic signal the CMNDF is as deep at
|
|
138
|
+
two periods as at one, so the shortest lag whose dip is under SUBHARMONIC_THRESHOLD
|
|
139
|
+
is taken as the period and every longer dip pays SUBHARMONIC_PENALTY per octave.
|
|
140
|
+
Without it the tracker halves the pitch of sung voices and of the chip's own output."""
|
|
141
|
+
idx = [i for i in range(max(lag_min, 1), len(c) - 1) if c[i] <= c[i - 1] and c[i] < c[i + 1]]
|
|
142
|
+
if not idx:
|
|
143
|
+
return np.zeros(0), np.zeros(0)
|
|
144
|
+
# a dip counts as "good" under the absolute threshold, or within 0.1 of the deepest
|
|
145
|
+
# one: on a fading note the dip at one period is shallower than 0.15 while the one at
|
|
146
|
+
# two periods is not, and the pitch would still be halved
|
|
147
|
+
from . import tuning
|
|
148
|
+
K = tuning.current
|
|
149
|
+
cmin = min(c[i] for i in idx)
|
|
150
|
+
good = [i for i in idx if c[i] < max(K.subharmonic_threshold, min(0.4, cmin + 0.1))]
|
|
151
|
+
first = min(good) if good else None
|
|
152
|
+
cost = {i: c[i] + (K.subharmonic_penalty * np.log2(i / first) if first is not None and i > 1.5 * first else 0.0)
|
|
153
|
+
for i in idx}
|
|
154
|
+
idx.sort(key=lambda i: cost[i])
|
|
155
|
+
idx = idx[:MAX_CANDIDATES]
|
|
156
|
+
return np.array([_parabolic(c, i) for i in idx]), np.array([cost[i] for i in idx])
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _viterbi(lags: list[np.ndarray], costs: list[np.ndarray], unvoiced_cost: np.ndarray | float,
|
|
160
|
+
octave_weight: float, switch_cost: float) -> np.ndarray:
|
|
161
|
+
"""Pick one candidate (or unvoiced, returned as nan) per frame by dynamic programming.
|
|
162
|
+
`unvoiced_cost` may be one value per frame."""
|
|
163
|
+
n = len(lags)
|
|
164
|
+
uc = np.broadcast_to(np.asarray(unvoiced_cost, dtype=np.float64), (n,))
|
|
165
|
+
best = [None] * n
|
|
166
|
+
back = [None] * n
|
|
167
|
+
prev_score = None
|
|
168
|
+
prev_lags = None
|
|
169
|
+
for k in range(n):
|
|
170
|
+
lk = np.append(lags[k], np.nan) # last state = unvoiced
|
|
171
|
+
local = np.append(costs[k], uc[k])
|
|
172
|
+
if prev_score is None:
|
|
173
|
+
score = local
|
|
174
|
+
back[k] = np.full(len(lk), -1)
|
|
175
|
+
else:
|
|
176
|
+
m = len(prev_lags)
|
|
177
|
+
trans = np.full((m, len(lk)), switch_cost)
|
|
178
|
+
pv = ~np.isnan(prev_lags)
|
|
179
|
+
cv = ~np.isnan(lk)
|
|
180
|
+
both = np.outer(pv, cv)
|
|
181
|
+
ratio = np.abs(np.log2(np.outer(prev_lags, 1.0 / lk)))
|
|
182
|
+
trans[both] = octave_weight * ratio[both]
|
|
183
|
+
trans[np.outer(~pv, ~cv)] = 0.0
|
|
184
|
+
total = prev_score[:, None] + trans
|
|
185
|
+
back[k] = np.argmin(total, axis=0)
|
|
186
|
+
score = total[back[k], np.arange(len(lk))] + local
|
|
187
|
+
best[k] = score
|
|
188
|
+
prev_score, prev_lags = score, lk
|
|
189
|
+
path = np.full(n, np.nan)
|
|
190
|
+
j = int(np.argmin(best[-1]))
|
|
191
|
+
for k in range(n - 1, -1, -1):
|
|
192
|
+
path[k] = lags[k][j] if j < len(lags[k]) else np.nan
|
|
193
|
+
j = int(back[k][j])
|
|
194
|
+
return path
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def analyze(x: np.ndarray, rate: int, order: int = 12, unvoiced_cost: float | None = None,
|
|
198
|
+
switch_cost: float | None = None, min_run: int | None = None, silence_db: float = -60.0,
|
|
199
|
+
hnr_unvoiced: float | None = None, hnr_voiced: float = float("inf"),
|
|
200
|
+
loud_range_db: float = 20.0, loud_unvoiced_bonus: float | None = None) -> Analysis:
|
|
201
|
+
from . import tuning
|
|
202
|
+
K = tuning.current
|
|
203
|
+
unvoiced_cost = K.unvoiced_cost if unvoiced_cost is None else unvoiced_cost
|
|
204
|
+
switch_cost = K.switch_cost if switch_cost is None else switch_cost
|
|
205
|
+
min_run = K.min_run if min_run is None else min_run
|
|
206
|
+
hnr_unvoiced = K.hnr_unvoiced if hnr_unvoiced is None else hnr_unvoiced
|
|
207
|
+
loud_unvoiced_bonus = K.loud_unvoiced_bonus if loud_unvoiced_bonus is None else loud_unvoiced_bonus
|
|
208
|
+
# hnr_voiced (promotion of unvoiced slots) is disabled by default: on the chip's own
|
|
209
|
+
# output, noise through narrow resonators shows a median HNR of 13 dB against 17 dB for
|
|
210
|
+
# voiced slots, so promotion creates far more false voicing than it repairs
|
|
211
|
+
x8 = to_8k(x, rate)
|
|
212
|
+
n = n_frames_for(len(x8))
|
|
213
|
+
lag_min = int(np.floor(RATE / F0_MAX))
|
|
214
|
+
lag_max = int(np.ceil(RATE / F0_MIN))
|
|
215
|
+
|
|
216
|
+
periodicity = np.zeros(n)
|
|
217
|
+
energy = np.full(n, silence_db)
|
|
218
|
+
env = np.zeros((n, N_BINS))
|
|
219
|
+
lpcs = np.zeros((n, order + 1))
|
|
220
|
+
hann_env = np.hanning(ENV_WIN)
|
|
221
|
+
cand_lags: list[np.ndarray] = []
|
|
222
|
+
cand_costs: list[np.ndarray] = []
|
|
223
|
+
|
|
224
|
+
for k in range(n):
|
|
225
|
+
start = k * HOP
|
|
226
|
+
seg = x8[start: start + HOP]
|
|
227
|
+
if len(seg):
|
|
228
|
+
r = float(np.sqrt(np.mean(seg * seg)))
|
|
229
|
+
energy[k] = 20 * np.log10(r) if r > 0 else silence_db
|
|
230
|
+
|
|
231
|
+
# envelope
|
|
232
|
+
w = _frame(x8, start, ENV_WIN) * hann_env
|
|
233
|
+
if np.dot(w, w) > 1e-12:
|
|
234
|
+
a, err = burg(w, order)
|
|
235
|
+
lpcs[k] = a
|
|
236
|
+
env[k] = lpc_envelope_db(a, np.sqrt(max(err, 1e-20)))
|
|
237
|
+
else:
|
|
238
|
+
env[k] = -120.0
|
|
239
|
+
|
|
240
|
+
# pitch candidates
|
|
241
|
+
p = _frame(x8, start, PITCH_WIN + lag_max)
|
|
242
|
+
if np.dot(p, p) < 1e-10 or energy[k] <= silence_db + 10:
|
|
243
|
+
cand_lags.append(np.zeros(0))
|
|
244
|
+
cand_costs.append(np.zeros(0))
|
|
245
|
+
continue
|
|
246
|
+
c = cmndf(p, lag_max)
|
|
247
|
+
lags, costs = _candidates(c, lag_min)
|
|
248
|
+
periodicity[k] = 1.0 - float(costs[0]) if len(costs) else 0.0
|
|
249
|
+
cand_lags.append(lags)
|
|
250
|
+
cand_costs.append(costs)
|
|
251
|
+
|
|
252
|
+
# loud frames are almost always vowels: the unvoiced state costs more near the peak level
|
|
253
|
+
# (reverberation and background lower the measured periodicity of real recordings)
|
|
254
|
+
peak = float(np.max(energy))
|
|
255
|
+
loudness = np.clip((energy - (peak - loud_range_db)) / loud_range_db, 0.0, 1.0)
|
|
256
|
+
uc = unvoiced_cost + loud_unvoiced_bonus * loudness
|
|
257
|
+
lag_path = _viterbi(cand_lags, cand_costs, uc, octave_weight=K.octave_weight, switch_cost=switch_cost)
|
|
258
|
+
f0 = RATE / lag_path
|
|
259
|
+
voiced = ~np.isnan(f0)
|
|
260
|
+
|
|
261
|
+
# harmonic-to-noise ratio at the tracked pitch (or the best candidate) decides the
|
|
262
|
+
# doubtful slots: periodic energy through narrow resonators is not voicing, and weak
|
|
263
|
+
# voiced slots at onsets still show harmonics
|
|
264
|
+
hnr = np.full(n, np.nan)
|
|
265
|
+
for k in range(n):
|
|
266
|
+
if energy[k] <= silence_db + 10:
|
|
267
|
+
continue
|
|
268
|
+
if voiced[k]:
|
|
269
|
+
fk = f0[k]
|
|
270
|
+
elif len(cand_lags[k]):
|
|
271
|
+
fk = RATE / cand_lags[k][0]
|
|
272
|
+
else:
|
|
273
|
+
continue
|
|
274
|
+
hnr[k] = harmonic_to_noise_db(x8, k * HOP, fk)
|
|
275
|
+
with np.errstate(invalid="ignore"):
|
|
276
|
+
voiced = np.where(hnr < hnr_unvoiced, False, voiced)
|
|
277
|
+
promote = (~voiced) & (hnr > hnr_voiced)
|
|
278
|
+
for k in np.where(promote)[0]:
|
|
279
|
+
f0[k] = RATE / cand_lags[k][0]
|
|
280
|
+
voiced = voiced | promote
|
|
281
|
+
voiced = _absorb_short_runs(voiced, min_run)
|
|
282
|
+
# slots voiced by absorption take the pitch of their nearest voiced neighbour
|
|
283
|
+
idx = np.where(~np.isnan(f0))[0]
|
|
284
|
+
if len(idx):
|
|
285
|
+
for k in np.where(voiced & np.isnan(f0))[0]:
|
|
286
|
+
f0[k] = f0[idx[np.argmin(np.abs(idx - k))]]
|
|
287
|
+
f0[~voiced] = np.nan
|
|
288
|
+
return Analysis(RATE, HOP, f0, voiced, periodicity, energy, env, lpcs, hnr)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def harmonic_to_noise_db(x8: np.ndarray, start: int, f0_hz: float, fmax: float = 2500.0) -> float:
|
|
292
|
+
"""Mean level difference (dB) between the harmonic peaks of f0 and the valleys between
|
|
293
|
+
them, over the harmonics below `fmax`, on a window of three periods."""
|
|
294
|
+
if not np.isfinite(f0_hz) or f0_hz <= 0:
|
|
295
|
+
return float("nan")
|
|
296
|
+
win = max(256, int(round(3 * RATE / f0_hz)))
|
|
297
|
+
win += win % 2
|
|
298
|
+
seg = _frame(x8, start, win) * np.hanning(win)
|
|
299
|
+
nfft = 4096
|
|
300
|
+
spec = 20 * np.log10(np.abs(np.fft.rfft(seg, nfft)) + 1e-9)
|
|
301
|
+
grid = np.fft.rfftfreq(nfft, 1.0 / RATE)
|
|
302
|
+
diffs = []
|
|
303
|
+
k = 1
|
|
304
|
+
while (k + 0.5) * f0_hz < fmax:
|
|
305
|
+
fc = k * f0_hz
|
|
306
|
+
lo, hi = np.searchsorted(grid, fc - 0.25 * f0_hz), np.searchsorted(grid, fc + 0.25 * f0_hz)
|
|
307
|
+
vlo, vhi = np.searchsorted(grid, fc + 0.3 * f0_hz), np.searchsorted(grid, fc + 0.7 * f0_hz)
|
|
308
|
+
if hi > lo and vhi > vlo:
|
|
309
|
+
diffs.append(spec[lo:hi].max() - spec[vlo:vhi].min())
|
|
310
|
+
k += 1
|
|
311
|
+
return float(np.mean(diffs)) if diffs else float("nan")
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _absorb_short_runs(flags: np.ndarray, min_run: int) -> np.ndarray:
|
|
315
|
+
"""Flip runs shorter than `min_run` that sit between two runs of the other value."""
|
|
316
|
+
out = flags.copy()
|
|
317
|
+
n = len(out)
|
|
318
|
+
k = 0
|
|
319
|
+
while k < n:
|
|
320
|
+
j = k
|
|
321
|
+
while j < n and out[j] == out[k]:
|
|
322
|
+
j += 1
|
|
323
|
+
if 0 < k and j < n and j - k < min_run:
|
|
324
|
+
out[k:j] = not out[k]
|
|
325
|
+
k = j
|
|
326
|
+
return out
|
mea8000/cli.py
ADDED
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
"""Command line.
|
|
2
|
+
|
|
3
|
+
mea8000 encode voice.wav voice.mea # profile thomson
|
|
4
|
+
mea8000 encode voice.wav voice.mea --profile compact
|
|
5
|
+
mea8000 encode voice.wav voice.mea --profile voice.toml --report
|
|
6
|
+
mea8000 render voice.mea voice.wav # what the chip says
|
|
7
|
+
mea8000 inspect voice.mea # what the file holds
|
|
8
|
+
mea8000 profiles # the shipped profiles
|
|
9
|
+
mea8000 profiles thomson > mine.toml # one of them, to start from
|
|
10
|
+
|
|
11
|
+
A profile (`--profile`) says how to convert: thomson (default), philips, compact, or a
|
|
12
|
+
TOML file of your own (docs/profile.md). Options override its values. `python -m mea8000`
|
|
13
|
+
is the same command.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import argparse
|
|
19
|
+
import sys
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
import numpy as np
|
|
23
|
+
|
|
24
|
+
from . import profile as profiles
|
|
25
|
+
from . import sim
|
|
26
|
+
from .codec import build_stream, describe_stream, detect_layout, parse_stream
|
|
27
|
+
from .filters import convert
|
|
28
|
+
from .wav import AudioError, describe, read_wav, write_wav
|
|
29
|
+
|
|
30
|
+
MODELS = {"default": sim.DEFAULT, "mame-int": sim.MAME_INT, "mame-float": sim.MAME_FLOAT, "java": sim.LEGACY_JAVA}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Progress:
|
|
34
|
+
"""A percentage on one line of stderr, only when stderr is a terminal."""
|
|
35
|
+
|
|
36
|
+
def __init__(self, label: str) -> None:
|
|
37
|
+
self.label = label
|
|
38
|
+
self.on = sys.stderr.isatty()
|
|
39
|
+
self.last = -1
|
|
40
|
+
|
|
41
|
+
def __call__(self, fraction: float) -> None:
|
|
42
|
+
pct = int(100 * min(1.0, max(0.0, fraction)))
|
|
43
|
+
if self.on and pct != self.last:
|
|
44
|
+
self.last = pct
|
|
45
|
+
sys.stderr.write(f"\r{self.label} {pct:3d}%")
|
|
46
|
+
sys.stderr.flush()
|
|
47
|
+
|
|
48
|
+
def done(self) -> None:
|
|
49
|
+
if self.on and self.last >= 0:
|
|
50
|
+
sys.stderr.write("\r" + " " * (len(self.label) + 6) + "\r")
|
|
51
|
+
sys.stderr.flush()
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def fail(message: str) -> int:
|
|
55
|
+
print(message, file=sys.stderr)
|
|
56
|
+
return 2
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _profile(args) -> profiles.Profile:
|
|
60
|
+
prof = profiles.load(args.profile)
|
|
61
|
+
return prof.with_overrides(format=getattr(args, "format", None), pitch=getattr(args, "pitch", None),
|
|
62
|
+
frame_ms=getattr(args, "frame", None), quality=getattr(args, "quality", None),
|
|
63
|
+
highpass_hz=getattr(args, "highpass", None), channel=getattr(args, "channel", None),
|
|
64
|
+
normalize=getattr(args, "normalize", None), trim=getattr(args, "trim", None))
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _load_stream(path: str, fmt: str | None):
|
|
68
|
+
p = Path(path)
|
|
69
|
+
if not p.is_file():
|
|
70
|
+
raise FileNotFoundError(f"{path}: no such file")
|
|
71
|
+
data = p.read_bytes()
|
|
72
|
+
layout = fmt or detect_layout(data)
|
|
73
|
+
try:
|
|
74
|
+
return parse_stream(data, layout), layout
|
|
75
|
+
except ValueError as e:
|
|
76
|
+
raise ValueError(f"{path}: not a {'vocabulary image' if layout == 'vocabulary' else 'speech file'} ({e})") from e
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _check_output(output: Path, source: Path, wrong_suffix: str, wrong_what: str) -> str | None:
|
|
80
|
+
"""Why the output cannot be written there, or None: never over the input, never under
|
|
81
|
+
the other kind's extension, only in a directory that exists."""
|
|
82
|
+
if output.resolve() == source.resolve():
|
|
83
|
+
return f"{output}: the output is the input"
|
|
84
|
+
if output.suffix.lower() == wrong_suffix:
|
|
85
|
+
return f"{output}: {wrong_what} cannot be written under a {wrong_suffix} name"
|
|
86
|
+
if not output.parent.is_dir():
|
|
87
|
+
return f"{output.parent}: no such directory"
|
|
88
|
+
return None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# ----------------------------------------------------------------- commands
|
|
92
|
+
|
|
93
|
+
def cmd_encode(args) -> int:
|
|
94
|
+
try:
|
|
95
|
+
prof = _profile(args)
|
|
96
|
+
except (FileNotFoundError, ValueError) as e:
|
|
97
|
+
return fail(str(e))
|
|
98
|
+
output = Path(args.output)
|
|
99
|
+
if not output.suffix:
|
|
100
|
+
output = output.with_name(output.name + prof.extension)
|
|
101
|
+
name = output.name.lower()
|
|
102
|
+
if name.endswith(".voc.mea") and prof.format == "speech":
|
|
103
|
+
return fail(f"{output}: the name says vocabulary image, the format is speech (pass --format vocabulary, or name it .mea)")
|
|
104
|
+
if name.endswith(".mea") and not name.endswith(".voc.mea") and prof.format == "vocabulary":
|
|
105
|
+
return fail(f"{output}: the name says speech files, the format is vocabulary (pass --format speech, or name it .voc.mea)")
|
|
106
|
+
why = _check_output(output, Path(args.source), ".wav", "speech data")
|
|
107
|
+
if why:
|
|
108
|
+
return fail(why)
|
|
109
|
+
try:
|
|
110
|
+
x, rate = read_wav(args.source)
|
|
111
|
+
progress = Progress("encoding")
|
|
112
|
+
prepared, utts, result = convert(prof, x, rate, progress=progress)
|
|
113
|
+
progress.done()
|
|
114
|
+
except (AudioError, ValueError) as e:
|
|
115
|
+
return fail(str(e))
|
|
116
|
+
print(f"{describe(x, rate, args.source)} -> {prepared.describe()}")
|
|
117
|
+
data = build_stream(utts, prof.format)
|
|
118
|
+
output.write_bytes(data)
|
|
119
|
+
what = "vocabulary image" if prof.format == "vocabulary" else "speech file" + ("s" if len(utts) != 1 else "")
|
|
120
|
+
frames = sum(len(u.frames) for u in utts)
|
|
121
|
+
print(f"{output}: {what}, {len(utts)} word group{'s' if len(utts) != 1 else ''}, {frames} frames, {len(data)} bytes "
|
|
122
|
+
f"(profile {prof.name}, {prof.clock_hz / 1e6:.2f} MHz, pitch {prof.pitch}, frame {prof.frame_ms} ms, {prof.quality})")
|
|
123
|
+
if args.report:
|
|
124
|
+
from .report import write_report
|
|
125
|
+
|
|
126
|
+
page = output.with_suffix("").with_suffix(".html") if output.name.endswith(".voc.mea") else output.with_suffix(".html")
|
|
127
|
+
source_wav = page.with_name(page.stem + "-source.wav")
|
|
128
|
+
chip_wav = page.with_name(page.stem + "-chip.wav")
|
|
129
|
+
write_report(page, source_wav, chip_wav, Path(args.source).name, prepared, utts, result, prof, len(data))
|
|
130
|
+
print(f"{page}: report ({source_wav.name}, the recording after the filters, and {chip_wav.name} beside it)")
|
|
131
|
+
return 0
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def cmd_render(args) -> int:
|
|
135
|
+
from .fastsim import HAVE_NUMBA, render_fast
|
|
136
|
+
|
|
137
|
+
try:
|
|
138
|
+
prof = profiles.load(args.profile)
|
|
139
|
+
utts, layout = _load_stream(args.file, args.format)
|
|
140
|
+
except (FileNotFoundError, ValueError) as e:
|
|
141
|
+
return fail(str(e))
|
|
142
|
+
if args.rate is not None and args.rate < 1:
|
|
143
|
+
return fail(f"--rate {args.rate}: a rate is a positive number of Hz")
|
|
144
|
+
why = _check_output(Path(args.output), Path(args.file), ".mea", "a rendering")
|
|
145
|
+
if why:
|
|
146
|
+
return fail(why)
|
|
147
|
+
model = sim.with_model(MODELS[args.model], clock_hz=prof.clock_hz)
|
|
148
|
+
if args.pitch_scale:
|
|
149
|
+
model = sim.with_model(model, pitch_scale=args.pitch_scale)
|
|
150
|
+
if args.truncate_bits:
|
|
151
|
+
model = sim.with_model(model, truncate_bits=args.truncate_bits)
|
|
152
|
+
noise = np.fromfile(args.noise_table, dtype="<i4").astype(np.int64) if args.noise_table else None
|
|
153
|
+
seconds = sum(u.duration_ms() for u in utts) / 1000
|
|
154
|
+
if args.policy == "java-exact":
|
|
155
|
+
samples = sim.render_java_like(utts, noise)
|
|
156
|
+
elif HAVE_NUMBA:
|
|
157
|
+
if sys.stderr.isatty():
|
|
158
|
+
sys.stderr.write("rendering ...\r")
|
|
159
|
+
samples = render_fast(utts, model, args.policy, noise)
|
|
160
|
+
else:
|
|
161
|
+
progress = Progress("rendering")
|
|
162
|
+
samples = sim.render(utts, model, args.policy, noise, progress=progress)
|
|
163
|
+
progress.done()
|
|
164
|
+
rate = model.sample_rate
|
|
165
|
+
if args.rate and args.rate != rate:
|
|
166
|
+
from math import gcd
|
|
167
|
+
from scipy.signal import resample_poly
|
|
168
|
+
|
|
169
|
+
g = gcd(int(args.rate), rate)
|
|
170
|
+
samples = np.clip(resample_poly(samples.astype(np.float64), int(args.rate) // g, rate // g), -32767, 32767)
|
|
171
|
+
rate = int(args.rate)
|
|
172
|
+
write_wav(args.output, samples, rate)
|
|
173
|
+
print(f"{args.output}: {len(samples) / rate:.2f} s at {rate} Hz (chip clock {prof.clock_hz / 1e6:.2f} MHz, "
|
|
174
|
+
f"profile {prof.name}, {layout} of {len(utts)} word group{'s' if len(utts) != 1 else ''})")
|
|
175
|
+
return 0
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def cmd_inspect(args) -> int:
|
|
179
|
+
try:
|
|
180
|
+
utts, layout = _load_stream(args.file, args.format)
|
|
181
|
+
except (FileNotFoundError, ValueError) as e:
|
|
182
|
+
return fail(str(e))
|
|
183
|
+
what = "vocabulary image" if layout == "vocabulary" else "speech file" + ("s" if len(utts) != 1 else "")
|
|
184
|
+
print(f"{args.file}: {what}, {len(utts)} word group{'s' if len(utts) != 1 else ''}, "
|
|
185
|
+
f"{sum(len(u.frames) for u in utts)} frames, {sum(u.duration_ms() for u in utts) / 1000:.2f} s")
|
|
186
|
+
if not args.summary:
|
|
187
|
+
print(describe_stream(utts))
|
|
188
|
+
return 0
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def cmd_profiles(args) -> int:
|
|
192
|
+
if args.name:
|
|
193
|
+
try:
|
|
194
|
+
print(profiles.path_of(args.name).read_text(), end="")
|
|
195
|
+
except FileNotFoundError as e:
|
|
196
|
+
return fail(str(e))
|
|
197
|
+
return 0
|
|
198
|
+
for name in profiles.names():
|
|
199
|
+
p = profiles.load(name)
|
|
200
|
+
print(f"{name:10s} clock {p.clock_hz / 1e6:.2f} MHz format {p.format:10s} pitch {p.pitch:6s} "
|
|
201
|
+
f"frame {p.frame_ms:2d} ms quality {p.quality:8s} channel {p.channel:5s} highpass {p.highpass_hz:g} Hz "
|
|
202
|
+
f"normalize {'on' if p.normalize else 'off':3s} trim {'on' if p.trim else 'off'}")
|
|
203
|
+
return 0
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
# ----------------------------------------------------------------- parser
|
|
207
|
+
|
|
208
|
+
class Formatter(argparse.RawDescriptionHelpFormatter):
|
|
209
|
+
pass
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def main(argv=None) -> int:
|
|
213
|
+
ap = argparse.ArgumentParser(prog="mea8000", description="A recording in, MEA8000 speech data out.",
|
|
214
|
+
formatter_class=Formatter, epilog=__doc__.split("\n", 2)[2])
|
|
215
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
216
|
+
|
|
217
|
+
e = sub.add_parser("encode", help="encode a recording into MEA8000 speech data",
|
|
218
|
+
usage="mea8000 encode SOURCE.wav OUTPUT.mea [--profile NAME|FILE] [options]",
|
|
219
|
+
description="Encode a recording (WAV, any rate, any bit depth, mono or stereo).",
|
|
220
|
+
formatter_class=Formatter)
|
|
221
|
+
e.add_argument("source", metavar="SOURCE.wav", help="the recording")
|
|
222
|
+
e.add_argument("output", metavar="OUTPUT.mea", help="the output (.mea for speech files, .voc.mea for a vocabulary image;"
|
|
223
|
+
" added when the name has no extension)")
|
|
224
|
+
e.add_argument("--profile", default="thomson", metavar="NAME|FILE",
|
|
225
|
+
help="how to convert: thomson (default), philips, compact, or a TOML file of your own")
|
|
226
|
+
e.add_argument("--format", choices=profiles.FORMATS,
|
|
227
|
+
help="speech: speech files one after the other, one per word group (default); "
|
|
228
|
+
"vocabulary: an image with an offset table, each file reachable by its number")
|
|
229
|
+
e.add_argument("--pitch", choices=profiles.PITCHES,
|
|
230
|
+
help="local: a starting pitch per word group, each group its own speech file (default); "
|
|
231
|
+
"global: one starting pitch for the whole recording, gliding through the pauses, in a single file")
|
|
232
|
+
e.add_argument("--frame", type=int, choices=profiles.FRAMES_MS, metavar="{8,16,32,64}",
|
|
233
|
+
help="the shortest frame allowed, in ms (default 8, the chip's finest grain; 64 makes "
|
|
234
|
+
"every frame 64 ms whatever the quality; only where noise meets voice inside a frame, "
|
|
235
|
+
"or at the end of a file, can a frame be shorter)")
|
|
236
|
+
e.add_argument("--quality", choices=tuple(profiles.QUALITY_PRICE_DB),
|
|
237
|
+
help="how much degradation is accepted to lengthen frames beyond --frame: "
|
|
238
|
+
"best: none (default); balanced: a little, about 40 %% fewer bytes at --frame; "
|
|
239
|
+
"compact: what stays intelligible, about 60 %% fewer bytes at --frame")
|
|
240
|
+
e.add_argument("--highpass", type=float, metavar="HZ",
|
|
241
|
+
help="a high-pass on the recording before analysis (80-150 Hz removes rumble and kick drums)")
|
|
242
|
+
e.add_argument("--channel", choices=profiles.CHANNELS,
|
|
243
|
+
help="which channel of a stereo recording to encode (default left; a mono file has only one)")
|
|
244
|
+
e.add_argument("--normalize", action=argparse.BooleanOptionalAction, default=None,
|
|
245
|
+
help="scale the recording so that its loudest sample reaches full scale (default on; the chip's "
|
|
246
|
+
"amplitude codes are absolute, a quiet recording would come out as silence)")
|
|
247
|
+
e.add_argument("--trim", action=argparse.BooleanOptionalAction, default=None,
|
|
248
|
+
help="drop the silence before the first sound and after the last (default on)")
|
|
249
|
+
e.add_argument("--report", action="store_true",
|
|
250
|
+
help="also write, next to the output, the recording after the filters (-source.wav), the chip's "
|
|
251
|
+
"rendering (-chip.wav) and an HTML page (.html): spectrograms of both, pitch, voicing and "
|
|
252
|
+
"level lanes, the two sounds to play")
|
|
253
|
+
e.set_defaults(func=cmd_encode)
|
|
254
|
+
|
|
255
|
+
r = sub.add_parser("render", help="render speech data to a WAV with the chip simulator",
|
|
256
|
+
description="Hear what the chip says: a WAV at the chip's own rate.", formatter_class=Formatter)
|
|
257
|
+
r.add_argument("file", metavar="INPUT.mea")
|
|
258
|
+
r.add_argument("output", metavar="OUTPUT.wav")
|
|
259
|
+
r.add_argument("--profile", default="thomson", metavar="NAME|FILE",
|
|
260
|
+
help="the clock of the target machine, taken from the profile: thomson 4 MHz (default), philips 3.84 MHz")
|
|
261
|
+
r.add_argument("--rate", type=int, metavar="HZ",
|
|
262
|
+
help="resample the WAV to this rate (default: the chip's own, 64000 Hz at 3.84 MHz, 66664 Hz at 4 MHz)")
|
|
263
|
+
r.add_argument("--format", choices=profiles.FORMATS, help="speech or vocabulary; detected from the content when omitted")
|
|
264
|
+
r.add_argument("--model", choices=list(MODELS), default="default", help=argparse.SUPPRESS)
|
|
265
|
+
r.add_argument("--policy", choices=["chunk", "philips", "java-exact"], default="chunk", help=argparse.SUPPRESS)
|
|
266
|
+
r.add_argument("--pitch-scale", type=float, help=argparse.SUPPRESS)
|
|
267
|
+
r.add_argument("--truncate-bits", type=int, help=argparse.SUPPRESS)
|
|
268
|
+
r.add_argument("--noise-table", help=argparse.SUPPRESS)
|
|
269
|
+
r.set_defaults(func=cmd_render)
|
|
270
|
+
|
|
271
|
+
i = sub.add_parser("inspect", help="print what a .mea file holds", formatter_class=Formatter)
|
|
272
|
+
i.add_argument("file", metavar="INPUT.mea")
|
|
273
|
+
i.add_argument("--format", choices=profiles.FORMATS, help="speech or vocabulary; detected from the content when omitted")
|
|
274
|
+
i.add_argument("--summary", action="store_true", help="the first line only")
|
|
275
|
+
i.set_defaults(func=cmd_inspect)
|
|
276
|
+
|
|
277
|
+
pr = sub.add_parser("profiles", help="list the shipped profiles, or print one",
|
|
278
|
+
description="Without a name, one line per shipped profile; with one, its file, to start yours from.")
|
|
279
|
+
pr.add_argument("name", nargs="?", metavar="NAME", help="thomson, philips or compact: print that profile's file")
|
|
280
|
+
pr.set_defaults(func=cmd_profiles)
|
|
281
|
+
|
|
282
|
+
args = ap.parse_args(argv)
|
|
283
|
+
return args.func(args)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
if __name__ == "__main__":
|
|
287
|
+
sys.exit(main())
|