mea8000-encoder 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mea8000/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ """MEA8000 speech synthesizer tools."""
2
+
3
+ from .codec import Frame, Utterance, parse_stream, build_stream
4
+ from .sim import Chip, Model, render
5
+
6
+ __all__ = ["Frame", "Utterance", "parse_stream", "build_stream", "Chip", "Model", "render"]
mea8000/__main__.py ADDED
@@ -0,0 +1,7 @@
1
+ """`python -m mea8000` is the `mea8000` command."""
2
+
3
+ import sys
4
+
5
+ from .cli import main
6
+
7
+ sys.exit(main())
mea8000/analysis.py ADDED
@@ -0,0 +1,326 @@
1
+ """Target analysis for the encoder: pitch, voicing, energy and spectral envelope per 8 ms.
2
+
3
+ Everything works at 8 kHz, the rate of the chip's filter bank: nothing above 4 kHz can be
4
+ represented, so the input is band-limited first.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass
10
+
11
+ import numpy as np
12
+ from scipy.signal import resample_poly
13
+
14
+ RATE = 8000
15
+ HOP = 64 # 8 ms, the shortest frame of the chip
16
+ ENV_WIN = 256 # 32 ms analysis window for the envelope
17
+ PITCH_WIN = 224 # 28 ms window for the difference function
18
+ MAX_CANDIDATES = 6
19
+ F0_MIN, F0_MAX = 60.0, 510.0
20
+ N_BINS = 129 # envelope grid: 0..4000 Hz in 31.25 Hz steps
21
+ ENV_FREQS = np.linspace(0.0, RATE / 2, N_BINS)
22
+
23
+
24
+ @dataclass
25
+ class Analysis:
26
+ rate: int
27
+ hop: int
28
+ f0: np.ndarray # Hz per frame, nan when unvoiced
29
+ voiced: np.ndarray # bool per frame
30
+ periodicity: np.ndarray # 1 - min(cmndf), per frame
31
+ energy_db: np.ndarray # RMS of the frame, dB re full scale
32
+ envelope_db: np.ndarray # (frames, N_BINS) smoothed log magnitude
33
+ lpc: np.ndarray # (frames, order + 1) predictor coefficients
34
+ hnr_db: np.ndarray | None = None # harmonic-to-noise ratio at the tracked pitch
35
+
36
+ @property
37
+ def n_frames(self) -> int:
38
+ return len(self.f0)
39
+
40
+ def times(self) -> np.ndarray:
41
+ return np.arange(self.n_frames) * self.hop / self.rate
42
+
43
+
44
+ def to_8k(x: np.ndarray, rate: float) -> np.ndarray:
45
+ """Mono float signal at 8 kHz; `rate` may be fractional (clock compensation)."""
46
+ x = np.asarray(x, dtype=np.float64)
47
+ if x.dtype.kind in "iu" or np.abs(x).max() > 1.5:
48
+ x = x / 32768.0
49
+ rate = int(round(rate))
50
+ if rate == RATE:
51
+ return x
52
+ from math import gcd
53
+
54
+ g = gcd(rate, RATE)
55
+ return resample_poly(x, RATE // g, rate // g)
56
+
57
+
58
+ def n_frames_for(n_samples: int, hop: int = HOP) -> int:
59
+ return max(1, int(np.ceil(n_samples / hop)))
60
+
61
+
62
+ def _frame(x: np.ndarray, start: int, win: int) -> np.ndarray:
63
+ """Window of `win` samples centred on the frame that starts at `start` (zero padded)."""
64
+ c = start + HOP // 2
65
+ a = c - win // 2
66
+ seg = np.zeros(win)
67
+ lo, hi = max(a, 0), min(a + win, len(x))
68
+ if hi > lo:
69
+ seg[lo - a: hi - a] = x[lo:hi]
70
+ return seg
71
+
72
+
73
+ # ------------------------------------------------------------------ envelope (Burg LPC)
74
+
75
+ def burg(x: np.ndarray, order: int) -> tuple[np.ndarray, float]:
76
+ """Burg's method. Returns a[0..order] with a[0] = 1 and the prediction error power."""
77
+ x = np.asarray(x, dtype=np.float64)
78
+ n = len(x)
79
+ a = np.zeros(order + 1)
80
+ a[0] = 1.0
81
+ err = float(np.dot(x, x)) / n
82
+ f = x[1:].copy() # forward prediction errors
83
+ b = x[:-1].copy() # backward prediction errors
84
+ for m in range(1, order + 1):
85
+ den = float(np.dot(f, f) + np.dot(b, b))
86
+ k = -2.0 * float(np.dot(f, b)) / den if den > 0 else 0.0
87
+ a_prev = a.copy()
88
+ for i in range(1, m):
89
+ a[i] = a_prev[i] + k * a_prev[m - i]
90
+ a[m] = k
91
+ f, b = f[1:] + k * b[1:], b[:-1] + k * f[:-1]
92
+ err *= (1.0 - k * k)
93
+ return a, err
94
+
95
+
96
+ def lpc_envelope_db(a: np.ndarray, gain: float, freqs: np.ndarray = ENV_FREQS) -> np.ndarray:
97
+ w = 2 * np.pi * freqs / RATE
98
+ k = np.arange(len(a))
99
+ denom = np.abs(np.exp(-1j * np.outer(w, k)) @ a)
100
+ return 20 * np.log10(gain / np.maximum(denom, 1e-9) + 1e-12)
101
+
102
+
103
+ # ------------------------------------------------------------------ pitch (YIN)
104
+
105
+ def cmndf(seg: np.ndarray, lag_max: int) -> np.ndarray:
106
+ """Cumulative mean normalised difference function of YIN for lags 0..lag_max."""
107
+ n = len(seg) - lag_max
108
+ x = seg[:n]
109
+ d = np.empty(lag_max + 1)
110
+ e0 = float(np.dot(x, x))
111
+ # d(tau) = sum (x[j] - x[j+tau])^2 over j < n
112
+ csum = np.concatenate([[0.0], np.cumsum(seg * seg)])
113
+ for tau in range(lag_max + 1):
114
+ y = seg[tau: tau + n]
115
+ d[tau] = e0 + (csum[tau + n] - csum[tau]) - 2.0 * float(np.dot(x, y))
116
+ out = np.ones(lag_max + 1)
117
+ run = np.cumsum(d[1:])
118
+ with np.errstate(divide="ignore", invalid="ignore"):
119
+ out[1:] = np.where(run > 0, d[1:] * np.arange(1, lag_max + 1) / run, 1.0)
120
+ return out
121
+
122
+
123
+ def _parabolic(y: np.ndarray, i: int) -> float:
124
+ if i <= 0 or i >= len(y) - 1:
125
+ return float(i)
126
+ a, b, c = y[i - 1], y[i], y[i + 1]
127
+ den = a - 2 * b + c
128
+ return float(i) if den == 0 else float(i + 0.5 * (a - c) / den)
129
+
130
+
131
+ # the subharmonic threshold and penalty live in tuning.current (subharmonic_threshold, _penalty)
132
+
133
+
134
+ def _candidates(c: np.ndarray, lag_min: int) -> tuple[np.ndarray, np.ndarray]:
135
+ """Local minima of the CMNDF (refined lag, cost), best first, at most MAX_CANDIDATES.
136
+
137
+ YIN's protection against pitch halving: on a periodic signal the CMNDF is as deep at
138
+ two periods as at one, so the shortest lag whose dip is under SUBHARMONIC_THRESHOLD
139
+ is taken as the period and every longer dip pays SUBHARMONIC_PENALTY per octave.
140
+ Without it the tracker halves the pitch of sung voices and of the chip's own output."""
141
+ idx = [i for i in range(max(lag_min, 1), len(c) - 1) if c[i] <= c[i - 1] and c[i] < c[i + 1]]
142
+ if not idx:
143
+ return np.zeros(0), np.zeros(0)
144
+ # a dip counts as "good" under the absolute threshold, or within 0.1 of the deepest
145
+ # one: on a fading note the dip at one period is shallower than 0.15 while the one at
146
+ # two periods is not, and the pitch would still be halved
147
+ from . import tuning
148
+ K = tuning.current
149
+ cmin = min(c[i] for i in idx)
150
+ good = [i for i in idx if c[i] < max(K.subharmonic_threshold, min(0.4, cmin + 0.1))]
151
+ first = min(good) if good else None
152
+ cost = {i: c[i] + (K.subharmonic_penalty * np.log2(i / first) if first is not None and i > 1.5 * first else 0.0)
153
+ for i in idx}
154
+ idx.sort(key=lambda i: cost[i])
155
+ idx = idx[:MAX_CANDIDATES]
156
+ return np.array([_parabolic(c, i) for i in idx]), np.array([cost[i] for i in idx])
157
+
158
+
159
+ def _viterbi(lags: list[np.ndarray], costs: list[np.ndarray], unvoiced_cost: np.ndarray | float,
160
+ octave_weight: float, switch_cost: float) -> np.ndarray:
161
+ """Pick one candidate (or unvoiced, returned as nan) per frame by dynamic programming.
162
+ `unvoiced_cost` may be one value per frame."""
163
+ n = len(lags)
164
+ uc = np.broadcast_to(np.asarray(unvoiced_cost, dtype=np.float64), (n,))
165
+ best = [None] * n
166
+ back = [None] * n
167
+ prev_score = None
168
+ prev_lags = None
169
+ for k in range(n):
170
+ lk = np.append(lags[k], np.nan) # last state = unvoiced
171
+ local = np.append(costs[k], uc[k])
172
+ if prev_score is None:
173
+ score = local
174
+ back[k] = np.full(len(lk), -1)
175
+ else:
176
+ m = len(prev_lags)
177
+ trans = np.full((m, len(lk)), switch_cost)
178
+ pv = ~np.isnan(prev_lags)
179
+ cv = ~np.isnan(lk)
180
+ both = np.outer(pv, cv)
181
+ ratio = np.abs(np.log2(np.outer(prev_lags, 1.0 / lk)))
182
+ trans[both] = octave_weight * ratio[both]
183
+ trans[np.outer(~pv, ~cv)] = 0.0
184
+ total = prev_score[:, None] + trans
185
+ back[k] = np.argmin(total, axis=0)
186
+ score = total[back[k], np.arange(len(lk))] + local
187
+ best[k] = score
188
+ prev_score, prev_lags = score, lk
189
+ path = np.full(n, np.nan)
190
+ j = int(np.argmin(best[-1]))
191
+ for k in range(n - 1, -1, -1):
192
+ path[k] = lags[k][j] if j < len(lags[k]) else np.nan
193
+ j = int(back[k][j])
194
+ return path
195
+
196
+
197
+ def analyze(x: np.ndarray, rate: int, order: int = 12, unvoiced_cost: float | None = None,
198
+ switch_cost: float | None = None, min_run: int | None = None, silence_db: float = -60.0,
199
+ hnr_unvoiced: float | None = None, hnr_voiced: float = float("inf"),
200
+ loud_range_db: float = 20.0, loud_unvoiced_bonus: float | None = None) -> Analysis:
201
+ from . import tuning
202
+ K = tuning.current
203
+ unvoiced_cost = K.unvoiced_cost if unvoiced_cost is None else unvoiced_cost
204
+ switch_cost = K.switch_cost if switch_cost is None else switch_cost
205
+ min_run = K.min_run if min_run is None else min_run
206
+ hnr_unvoiced = K.hnr_unvoiced if hnr_unvoiced is None else hnr_unvoiced
207
+ loud_unvoiced_bonus = K.loud_unvoiced_bonus if loud_unvoiced_bonus is None else loud_unvoiced_bonus
208
+ # hnr_voiced (promotion of unvoiced slots) is disabled by default: on the chip's own
209
+ # output, noise through narrow resonators shows a median HNR of 13 dB against 17 dB for
210
+ # voiced slots, so promotion creates far more false voicing than it repairs
211
+ x8 = to_8k(x, rate)
212
+ n = n_frames_for(len(x8))
213
+ lag_min = int(np.floor(RATE / F0_MAX))
214
+ lag_max = int(np.ceil(RATE / F0_MIN))
215
+
216
+ periodicity = np.zeros(n)
217
+ energy = np.full(n, silence_db)
218
+ env = np.zeros((n, N_BINS))
219
+ lpcs = np.zeros((n, order + 1))
220
+ hann_env = np.hanning(ENV_WIN)
221
+ cand_lags: list[np.ndarray] = []
222
+ cand_costs: list[np.ndarray] = []
223
+
224
+ for k in range(n):
225
+ start = k * HOP
226
+ seg = x8[start: start + HOP]
227
+ if len(seg):
228
+ r = float(np.sqrt(np.mean(seg * seg)))
229
+ energy[k] = 20 * np.log10(r) if r > 0 else silence_db
230
+
231
+ # envelope
232
+ w = _frame(x8, start, ENV_WIN) * hann_env
233
+ if np.dot(w, w) > 1e-12:
234
+ a, err = burg(w, order)
235
+ lpcs[k] = a
236
+ env[k] = lpc_envelope_db(a, np.sqrt(max(err, 1e-20)))
237
+ else:
238
+ env[k] = -120.0
239
+
240
+ # pitch candidates
241
+ p = _frame(x8, start, PITCH_WIN + lag_max)
242
+ if np.dot(p, p) < 1e-10 or energy[k] <= silence_db + 10:
243
+ cand_lags.append(np.zeros(0))
244
+ cand_costs.append(np.zeros(0))
245
+ continue
246
+ c = cmndf(p, lag_max)
247
+ lags, costs = _candidates(c, lag_min)
248
+ periodicity[k] = 1.0 - float(costs[0]) if len(costs) else 0.0
249
+ cand_lags.append(lags)
250
+ cand_costs.append(costs)
251
+
252
+ # loud frames are almost always vowels: the unvoiced state costs more near the peak level
253
+ # (reverberation and background lower the measured periodicity of real recordings)
254
+ peak = float(np.max(energy))
255
+ loudness = np.clip((energy - (peak - loud_range_db)) / loud_range_db, 0.0, 1.0)
256
+ uc = unvoiced_cost + loud_unvoiced_bonus * loudness
257
+ lag_path = _viterbi(cand_lags, cand_costs, uc, octave_weight=K.octave_weight, switch_cost=switch_cost)
258
+ f0 = RATE / lag_path
259
+ voiced = ~np.isnan(f0)
260
+
261
+ # harmonic-to-noise ratio at the tracked pitch (or the best candidate) decides the
262
+ # doubtful slots: periodic energy through narrow resonators is not voicing, and weak
263
+ # voiced slots at onsets still show harmonics
264
+ hnr = np.full(n, np.nan)
265
+ for k in range(n):
266
+ if energy[k] <= silence_db + 10:
267
+ continue
268
+ if voiced[k]:
269
+ fk = f0[k]
270
+ elif len(cand_lags[k]):
271
+ fk = RATE / cand_lags[k][0]
272
+ else:
273
+ continue
274
+ hnr[k] = harmonic_to_noise_db(x8, k * HOP, fk)
275
+ with np.errstate(invalid="ignore"):
276
+ voiced = np.where(hnr < hnr_unvoiced, False, voiced)
277
+ promote = (~voiced) & (hnr > hnr_voiced)
278
+ for k in np.where(promote)[0]:
279
+ f0[k] = RATE / cand_lags[k][0]
280
+ voiced = voiced | promote
281
+ voiced = _absorb_short_runs(voiced, min_run)
282
+ # slots voiced by absorption take the pitch of their nearest voiced neighbour
283
+ idx = np.where(~np.isnan(f0))[0]
284
+ if len(idx):
285
+ for k in np.where(voiced & np.isnan(f0))[0]:
286
+ f0[k] = f0[idx[np.argmin(np.abs(idx - k))]]
287
+ f0[~voiced] = np.nan
288
+ return Analysis(RATE, HOP, f0, voiced, periodicity, energy, env, lpcs, hnr)
289
+
290
+
291
+ def harmonic_to_noise_db(x8: np.ndarray, start: int, f0_hz: float, fmax: float = 2500.0) -> float:
292
+ """Mean level difference (dB) between the harmonic peaks of f0 and the valleys between
293
+ them, over the harmonics below `fmax`, on a window of three periods."""
294
+ if not np.isfinite(f0_hz) or f0_hz <= 0:
295
+ return float("nan")
296
+ win = max(256, int(round(3 * RATE / f0_hz)))
297
+ win += win % 2
298
+ seg = _frame(x8, start, win) * np.hanning(win)
299
+ nfft = 4096
300
+ spec = 20 * np.log10(np.abs(np.fft.rfft(seg, nfft)) + 1e-9)
301
+ grid = np.fft.rfftfreq(nfft, 1.0 / RATE)
302
+ diffs = []
303
+ k = 1
304
+ while (k + 0.5) * f0_hz < fmax:
305
+ fc = k * f0_hz
306
+ lo, hi = np.searchsorted(grid, fc - 0.25 * f0_hz), np.searchsorted(grid, fc + 0.25 * f0_hz)
307
+ vlo, vhi = np.searchsorted(grid, fc + 0.3 * f0_hz), np.searchsorted(grid, fc + 0.7 * f0_hz)
308
+ if hi > lo and vhi > vlo:
309
+ diffs.append(spec[lo:hi].max() - spec[vlo:vhi].min())
310
+ k += 1
311
+ return float(np.mean(diffs)) if diffs else float("nan")
312
+
313
+
314
+ def _absorb_short_runs(flags: np.ndarray, min_run: int) -> np.ndarray:
315
+ """Flip runs shorter than `min_run` that sit between two runs of the other value."""
316
+ out = flags.copy()
317
+ n = len(out)
318
+ k = 0
319
+ while k < n:
320
+ j = k
321
+ while j < n and out[j] == out[k]:
322
+ j += 1
323
+ if 0 < k and j < n and j - k < min_run:
324
+ out[k:j] = not out[k]
325
+ k = j
326
+ return out
mea8000/cli.py ADDED
@@ -0,0 +1,287 @@
1
+ """Command line.
2
+
3
+ mea8000 encode voice.wav voice.mea # profile thomson
4
+ mea8000 encode voice.wav voice.mea --profile compact
5
+ mea8000 encode voice.wav voice.mea --profile voice.toml --report
6
+ mea8000 render voice.mea voice.wav # what the chip says
7
+ mea8000 inspect voice.mea # what the file holds
8
+ mea8000 profiles # the shipped profiles
9
+ mea8000 profiles thomson > mine.toml # one of them, to start from
10
+
11
+ A profile (`--profile`) says how to convert: thomson (default), philips, compact, or a
12
+ TOML file of your own (docs/profile.md). Options override its values. `python -m mea8000`
13
+ is the same command.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import argparse
19
+ import sys
20
+ from pathlib import Path
21
+
22
+ import numpy as np
23
+
24
+ from . import profile as profiles
25
+ from . import sim
26
+ from .codec import build_stream, describe_stream, detect_layout, parse_stream
27
+ from .filters import convert
28
+ from .wav import AudioError, describe, read_wav, write_wav
29
+
30
+ MODELS = {"default": sim.DEFAULT, "mame-int": sim.MAME_INT, "mame-float": sim.MAME_FLOAT, "java": sim.LEGACY_JAVA}
31
+
32
+
33
+ class Progress:
34
+ """A percentage on one line of stderr, only when stderr is a terminal."""
35
+
36
+ def __init__(self, label: str) -> None:
37
+ self.label = label
38
+ self.on = sys.stderr.isatty()
39
+ self.last = -1
40
+
41
+ def __call__(self, fraction: float) -> None:
42
+ pct = int(100 * min(1.0, max(0.0, fraction)))
43
+ if self.on and pct != self.last:
44
+ self.last = pct
45
+ sys.stderr.write(f"\r{self.label} {pct:3d}%")
46
+ sys.stderr.flush()
47
+
48
+ def done(self) -> None:
49
+ if self.on and self.last >= 0:
50
+ sys.stderr.write("\r" + " " * (len(self.label) + 6) + "\r")
51
+ sys.stderr.flush()
52
+
53
+
54
+ def fail(message: str) -> int:
55
+ print(message, file=sys.stderr)
56
+ return 2
57
+
58
+
59
+ def _profile(args) -> profiles.Profile:
60
+ prof = profiles.load(args.profile)
61
+ return prof.with_overrides(format=getattr(args, "format", None), pitch=getattr(args, "pitch", None),
62
+ frame_ms=getattr(args, "frame", None), quality=getattr(args, "quality", None),
63
+ highpass_hz=getattr(args, "highpass", None), channel=getattr(args, "channel", None),
64
+ normalize=getattr(args, "normalize", None), trim=getattr(args, "trim", None))
65
+
66
+
67
+ def _load_stream(path: str, fmt: str | None):
68
+ p = Path(path)
69
+ if not p.is_file():
70
+ raise FileNotFoundError(f"{path}: no such file")
71
+ data = p.read_bytes()
72
+ layout = fmt or detect_layout(data)
73
+ try:
74
+ return parse_stream(data, layout), layout
75
+ except ValueError as e:
76
+ raise ValueError(f"{path}: not a {'vocabulary image' if layout == 'vocabulary' else 'speech file'} ({e})") from e
77
+
78
+
79
+ def _check_output(output: Path, source: Path, wrong_suffix: str, wrong_what: str) -> str | None:
80
+ """Why the output cannot be written there, or None: never over the input, never under
81
+ the other kind's extension, only in a directory that exists."""
82
+ if output.resolve() == source.resolve():
83
+ return f"{output}: the output is the input"
84
+ if output.suffix.lower() == wrong_suffix:
85
+ return f"{output}: {wrong_what} cannot be written under a {wrong_suffix} name"
86
+ if not output.parent.is_dir():
87
+ return f"{output.parent}: no such directory"
88
+ return None
89
+
90
+
91
+ # ----------------------------------------------------------------- commands
92
+
93
+ def cmd_encode(args) -> int:
94
+ try:
95
+ prof = _profile(args)
96
+ except (FileNotFoundError, ValueError) as e:
97
+ return fail(str(e))
98
+ output = Path(args.output)
99
+ if not output.suffix:
100
+ output = output.with_name(output.name + prof.extension)
101
+ name = output.name.lower()
102
+ if name.endswith(".voc.mea") and prof.format == "speech":
103
+ return fail(f"{output}: the name says vocabulary image, the format is speech (pass --format vocabulary, or name it .mea)")
104
+ if name.endswith(".mea") and not name.endswith(".voc.mea") and prof.format == "vocabulary":
105
+ return fail(f"{output}: the name says speech files, the format is vocabulary (pass --format speech, or name it .voc.mea)")
106
+ why = _check_output(output, Path(args.source), ".wav", "speech data")
107
+ if why:
108
+ return fail(why)
109
+ try:
110
+ x, rate = read_wav(args.source)
111
+ progress = Progress("encoding")
112
+ prepared, utts, result = convert(prof, x, rate, progress=progress)
113
+ progress.done()
114
+ except (AudioError, ValueError) as e:
115
+ return fail(str(e))
116
+ print(f"{describe(x, rate, args.source)} -> {prepared.describe()}")
117
+ data = build_stream(utts, prof.format)
118
+ output.write_bytes(data)
119
+ what = "vocabulary image" if prof.format == "vocabulary" else "speech file" + ("s" if len(utts) != 1 else "")
120
+ frames = sum(len(u.frames) for u in utts)
121
+ print(f"{output}: {what}, {len(utts)} word group{'s' if len(utts) != 1 else ''}, {frames} frames, {len(data)} bytes "
122
+ f"(profile {prof.name}, {prof.clock_hz / 1e6:.2f} MHz, pitch {prof.pitch}, frame {prof.frame_ms} ms, {prof.quality})")
123
+ if args.report:
124
+ from .report import write_report
125
+
126
+ page = output.with_suffix("").with_suffix(".html") if output.name.endswith(".voc.mea") else output.with_suffix(".html")
127
+ source_wav = page.with_name(page.stem + "-source.wav")
128
+ chip_wav = page.with_name(page.stem + "-chip.wav")
129
+ write_report(page, source_wav, chip_wav, Path(args.source).name, prepared, utts, result, prof, len(data))
130
+ print(f"{page}: report ({source_wav.name}, the recording after the filters, and {chip_wav.name} beside it)")
131
+ return 0
132
+
133
+
134
+ def cmd_render(args) -> int:
135
+ from .fastsim import HAVE_NUMBA, render_fast
136
+
137
+ try:
138
+ prof = profiles.load(args.profile)
139
+ utts, layout = _load_stream(args.file, args.format)
140
+ except (FileNotFoundError, ValueError) as e:
141
+ return fail(str(e))
142
+ if args.rate is not None and args.rate < 1:
143
+ return fail(f"--rate {args.rate}: a rate is a positive number of Hz")
144
+ why = _check_output(Path(args.output), Path(args.file), ".mea", "a rendering")
145
+ if why:
146
+ return fail(why)
147
+ model = sim.with_model(MODELS[args.model], clock_hz=prof.clock_hz)
148
+ if args.pitch_scale:
149
+ model = sim.with_model(model, pitch_scale=args.pitch_scale)
150
+ if args.truncate_bits:
151
+ model = sim.with_model(model, truncate_bits=args.truncate_bits)
152
+ noise = np.fromfile(args.noise_table, dtype="<i4").astype(np.int64) if args.noise_table else None
153
+ seconds = sum(u.duration_ms() for u in utts) / 1000
154
+ if args.policy == "java-exact":
155
+ samples = sim.render_java_like(utts, noise)
156
+ elif HAVE_NUMBA:
157
+ if sys.stderr.isatty():
158
+ sys.stderr.write("rendering ...\r")
159
+ samples = render_fast(utts, model, args.policy, noise)
160
+ else:
161
+ progress = Progress("rendering")
162
+ samples = sim.render(utts, model, args.policy, noise, progress=progress)
163
+ progress.done()
164
+ rate = model.sample_rate
165
+ if args.rate and args.rate != rate:
166
+ from math import gcd
167
+ from scipy.signal import resample_poly
168
+
169
+ g = gcd(int(args.rate), rate)
170
+ samples = np.clip(resample_poly(samples.astype(np.float64), int(args.rate) // g, rate // g), -32767, 32767)
171
+ rate = int(args.rate)
172
+ write_wav(args.output, samples, rate)
173
+ print(f"{args.output}: {len(samples) / rate:.2f} s at {rate} Hz (chip clock {prof.clock_hz / 1e6:.2f} MHz, "
174
+ f"profile {prof.name}, {layout} of {len(utts)} word group{'s' if len(utts) != 1 else ''})")
175
+ return 0
176
+
177
+
178
+ def cmd_inspect(args) -> int:
179
+ try:
180
+ utts, layout = _load_stream(args.file, args.format)
181
+ except (FileNotFoundError, ValueError) as e:
182
+ return fail(str(e))
183
+ what = "vocabulary image" if layout == "vocabulary" else "speech file" + ("s" if len(utts) != 1 else "")
184
+ print(f"{args.file}: {what}, {len(utts)} word group{'s' if len(utts) != 1 else ''}, "
185
+ f"{sum(len(u.frames) for u in utts)} frames, {sum(u.duration_ms() for u in utts) / 1000:.2f} s")
186
+ if not args.summary:
187
+ print(describe_stream(utts))
188
+ return 0
189
+
190
+
191
+ def cmd_profiles(args) -> int:
192
+ if args.name:
193
+ try:
194
+ print(profiles.path_of(args.name).read_text(), end="")
195
+ except FileNotFoundError as e:
196
+ return fail(str(e))
197
+ return 0
198
+ for name in profiles.names():
199
+ p = profiles.load(name)
200
+ print(f"{name:10s} clock {p.clock_hz / 1e6:.2f} MHz format {p.format:10s} pitch {p.pitch:6s} "
201
+ f"frame {p.frame_ms:2d} ms quality {p.quality:8s} channel {p.channel:5s} highpass {p.highpass_hz:g} Hz "
202
+ f"normalize {'on' if p.normalize else 'off':3s} trim {'on' if p.trim else 'off'}")
203
+ return 0
204
+
205
+
206
+ # ----------------------------------------------------------------- parser
207
+
208
+ class Formatter(argparse.RawDescriptionHelpFormatter):
209
+ pass
210
+
211
+
212
+ def main(argv=None) -> int:
213
+ ap = argparse.ArgumentParser(prog="mea8000", description="A recording in, MEA8000 speech data out.",
214
+ formatter_class=Formatter, epilog=__doc__.split("\n", 2)[2])
215
+ sub = ap.add_subparsers(dest="cmd", required=True)
216
+
217
+ e = sub.add_parser("encode", help="encode a recording into MEA8000 speech data",
218
+ usage="mea8000 encode SOURCE.wav OUTPUT.mea [--profile NAME|FILE] [options]",
219
+ description="Encode a recording (WAV, any rate, any bit depth, mono or stereo).",
220
+ formatter_class=Formatter)
221
+ e.add_argument("source", metavar="SOURCE.wav", help="the recording")
222
+ e.add_argument("output", metavar="OUTPUT.mea", help="the output (.mea for speech files, .voc.mea for a vocabulary image;"
223
+ " added when the name has no extension)")
224
+ e.add_argument("--profile", default="thomson", metavar="NAME|FILE",
225
+ help="how to convert: thomson (default), philips, compact, or a TOML file of your own")
226
+ e.add_argument("--format", choices=profiles.FORMATS,
227
+ help="speech: speech files one after the other, one per word group (default); "
228
+ "vocabulary: an image with an offset table, each file reachable by its number")
229
+ e.add_argument("--pitch", choices=profiles.PITCHES,
230
+ help="local: a starting pitch per word group, each group its own speech file (default); "
231
+ "global: one starting pitch for the whole recording, gliding through the pauses, in a single file")
232
+ e.add_argument("--frame", type=int, choices=profiles.FRAMES_MS, metavar="{8,16,32,64}",
233
+ help="the shortest frame allowed, in ms (default 8, the chip's finest grain; 64 makes "
234
+ "every frame 64 ms whatever the quality; only where noise meets voice inside a frame, "
235
+ "or at the end of a file, can a frame be shorter)")
236
+ e.add_argument("--quality", choices=tuple(profiles.QUALITY_PRICE_DB),
237
+ help="how much degradation is accepted to lengthen frames beyond --frame: "
238
+ "best: none (default); balanced: a little, about 40 %% fewer bytes at --frame; "
239
+ "compact: what stays intelligible, about 60 %% fewer bytes at --frame")
240
+ e.add_argument("--highpass", type=float, metavar="HZ",
241
+ help="a high-pass on the recording before analysis (80-150 Hz removes rumble and kick drums)")
242
+ e.add_argument("--channel", choices=profiles.CHANNELS,
243
+ help="which channel of a stereo recording to encode (default left; a mono file has only one)")
244
+ e.add_argument("--normalize", action=argparse.BooleanOptionalAction, default=None,
245
+ help="scale the recording so that its loudest sample reaches full scale (default on; the chip's "
246
+ "amplitude codes are absolute, a quiet recording would come out as silence)")
247
+ e.add_argument("--trim", action=argparse.BooleanOptionalAction, default=None,
248
+ help="drop the silence before the first sound and after the last (default on)")
249
+ e.add_argument("--report", action="store_true",
250
+ help="also write, next to the output, the recording after the filters (-source.wav), the chip's "
251
+ "rendering (-chip.wav) and an HTML page (.html): spectrograms of both, pitch, voicing and "
252
+ "level lanes, the two sounds to play")
253
+ e.set_defaults(func=cmd_encode)
254
+
255
+ r = sub.add_parser("render", help="render speech data to a WAV with the chip simulator",
256
+ description="Hear what the chip says: a WAV at the chip's own rate.", formatter_class=Formatter)
257
+ r.add_argument("file", metavar="INPUT.mea")
258
+ r.add_argument("output", metavar="OUTPUT.wav")
259
+ r.add_argument("--profile", default="thomson", metavar="NAME|FILE",
260
+ help="the clock of the target machine, taken from the profile: thomson 4 MHz (default), philips 3.84 MHz")
261
+ r.add_argument("--rate", type=int, metavar="HZ",
262
+ help="resample the WAV to this rate (default: the chip's own, 64000 Hz at 3.84 MHz, 66664 Hz at 4 MHz)")
263
+ r.add_argument("--format", choices=profiles.FORMATS, help="speech or vocabulary; detected from the content when omitted")
264
+ r.add_argument("--model", choices=list(MODELS), default="default", help=argparse.SUPPRESS)
265
+ r.add_argument("--policy", choices=["chunk", "philips", "java-exact"], default="chunk", help=argparse.SUPPRESS)
266
+ r.add_argument("--pitch-scale", type=float, help=argparse.SUPPRESS)
267
+ r.add_argument("--truncate-bits", type=int, help=argparse.SUPPRESS)
268
+ r.add_argument("--noise-table", help=argparse.SUPPRESS)
269
+ r.set_defaults(func=cmd_render)
270
+
271
+ i = sub.add_parser("inspect", help="print what a .mea file holds", formatter_class=Formatter)
272
+ i.add_argument("file", metavar="INPUT.mea")
273
+ i.add_argument("--format", choices=profiles.FORMATS, help="speech or vocabulary; detected from the content when omitted")
274
+ i.add_argument("--summary", action="store_true", help="the first line only")
275
+ i.set_defaults(func=cmd_inspect)
276
+
277
+ pr = sub.add_parser("profiles", help="list the shipped profiles, or print one",
278
+ description="Without a name, one line per shipped profile; with one, its file, to start yours from.")
279
+ pr.add_argument("name", nargs="?", metavar="NAME", help="thomson, philips or compact: print that profile's file")
280
+ pr.set_defaults(func=cmd_profiles)
281
+
282
+ args = ap.parse_args(argv)
283
+ return args.func(args)
284
+
285
+
286
+ if __name__ == "__main__":
287
+ sys.exit(main())