reelkit-cli 0.10.6 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,85 @@
1
+ #!/usr/bin/env python3
2
+ """What a recording actually says, as phonemes, and whether it says what was meant.
3
+
4
+ phonemes.py take1.wav take2.wav [--want "m e a f j e n i m"] [--from 1.2 --to 2.3]
5
+
6
+ Speech-to-text writes the word it expects, so a mispronounced word still comes back spelled correctly. A phoneme recogniser (wav2vec2 trained on
7
+ espeak phonemes, any language) writes what was said. With --want, each file passes when the wanted phonemes appear in order (spaces and length marks
8
+ are ignored); the nearest stretch is shown when they do not. Also printed: the length of the stretch, so an over-processed take (stretched, pitched)
9
+ shows up as too long for its phonemes.
10
+ Check the clean voice, not the mix: music degrades the result. Needs `torch`, `transformers` and `huggingface_hub`; the model (about 1.2 GB) is
11
+ downloaded once. If they are missing, the script says so and exits with code 2 rather than failing a build.
12
+ """
13
+ import argparse, json, subprocess, sys
14
+
15
+ np = None # numpy, imported with the other optional packages so that a missing one is a sentence, not a traceback
16
+
17
+ MODEL = "facebook/wav2vec2-xlsr-53-espeak-cv-ft"
18
+
19
+
20
+ def load16(path, t0=None, t1=None):
21
+ cut = (["-ss", str(t0)] if t0 else []) + (["-to", str(t1)] if t1 else [])
22
+ raw = subprocess.check_output(["ffmpeg", "-nostdin", "-v", "error", *cut, "-i", path, "-ac", "1", "-ar", "16000", "-f", "f32le", "-"])
23
+ a = np.frombuffer(raw, np.float32)
24
+ return (a - a.mean()) / (a.std() + 1e-7)
25
+
26
+
27
+ def squash(s):
28
+ return "".join(c for c in s if c not in " ːˈˌ")
29
+
30
+
31
+ def nearest(got, want):
32
+ """The stretch of `got` closest to `want` by edit distance, and that distance."""
33
+ g, w = squash(got), squash(want)
34
+ best = (len(w), "")
35
+ for i in range(len(g)):
36
+ for n in (len(w) - 1, len(w), len(w) + 1):
37
+ s = g[i: i + n]
38
+ if not s:
39
+ continue
40
+ prev = list(range(len(w) + 1))
41
+ for a_ in s:
42
+ cur = [prev[0] + 1]
43
+ for j, b_ in enumerate(w, 1):
44
+ cur.append(min(prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + (a_ != b_)))
45
+ prev = cur
46
+ if prev[-1] < best[0]:
47
+ best = (prev[-1], s)
48
+ return best
49
+
50
+
51
+ if __name__ == "__main__":
52
+ p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
53
+ p.add_argument("files", nargs="+"); p.add_argument("--want"); p.add_argument("--from", dest="t0", type=float); p.add_argument("--to", dest="t1", type=float)
54
+ a = p.parse_args()
55
+ try:
56
+ import numpy as np
57
+ import torch
58
+ from transformers import Wav2Vec2ForCTC
59
+ from huggingface_hub import hf_hub_download
60
+ except ImportError as e:
61
+ print(f"phoneme check skipped: {e.name} is not installed. In a virtual environment: pip install torch transformers huggingface_hub", file=sys.stderr)
62
+ sys.exit(2)
63
+ model = Wav2Vec2ForCTC.from_pretrained(MODEL).eval()
64
+ vocab = json.load(open(hf_hub_download(MODEL, "vocab.json")))
65
+ inv = {v: k for k, v in vocab.items()}
66
+ ok = True
67
+ for f in a.files:
68
+ x = load16(f, a.t0, a.t1)
69
+ with torch.no_grad():
70
+ ids = model(torch.tensor(x)[None]).logits[0].argmax(-1).tolist()
71
+ out, prev = [], None
72
+ for i in ids:
73
+ if i != prev and inv[i] not in ("<pad>", "<s>", "</s>", "<unk>", "|"):
74
+ out.append(inv[i])
75
+ prev = i
76
+ said = " ".join(out)
77
+ row = {"file": f, "seconds": round(len(x) / 16000, 2), "phonemes": said}
78
+ if a.want:
79
+ row["has_wanted"] = squash(a.want) in squash(said)
80
+ if not row["has_wanted"]:
81
+ dist, near = nearest(said, a.want)
82
+ row["nearest"] = near; row["distance"] = dist
83
+ ok = False
84
+ print(json.dumps(row, ensure_ascii=False))
85
+ sys.exit(0 if ok else 1)
@@ -0,0 +1,51 @@
1
+ #!/usr/bin/env python3
2
+ """Does a take jump an octave, or end like a question?
3
+
4
+ pitch_check.py take1.wav take2.wav [--floor 70 --ceil 400]
5
+
6
+ For each file: the pitch range, the largest step between neighbouring 10 ms frames (a ratio near 2 is an octave jump, the commonest fault of a
7
+ synthetic voice), and the pitch over the last voiced 0.6 s in four quarters. A statement falls; a line whose last quarter is clearly above its
8
+ first reads as a question. The verdicts (octave_jump, rises_at_end) are given only when Praat is installed (`praat-parselmouth`); the built-in
9
+ fallback tracker makes octave errors of its own, so with it only the numbers are printed and the exit code is 0.
10
+ """
11
+ import argparse, json, sys
12
+ import numpy as np
13
+ from _audio import SR, f0_track, load
14
+
15
+
16
+ def has_praat():
17
+ try:
18
+ import parselmouth # noqa: F401
19
+ return True
20
+ except ImportError:
21
+ return False
22
+
23
+
24
+ def check(path, lo, hi):
25
+ t, f = f0_track(load(path), SR, lo, hi)
26
+ if len(f) < 8:
27
+ return {"file": path, "voiced": False}
28
+ steps = [max(f[i + 1] / f[i], f[i] / f[i + 1]) for i in range(len(f) - 1) if t[i + 1] - t[i] < 0.025]
29
+ last = f[t > t[-1] - 0.6]
30
+ q = [float(np.median(p)) for p in np.array_split(last, 4) if len(p)]
31
+ worst = max(steps) if steps else 1.0
32
+ out = {"file": path, "voiced": True, "range_hz": [round(float(f.min())), round(float(f.max()))], "max_step_ratio": round(float(worst), 2), "last_quarters_hz": [round(v) for v in q]}
33
+ if has_praat():
34
+ out["octave_jump"] = bool(1.8 < worst < 2.25)
35
+ out["rises_at_end"] = bool(len(q) == 4 and q[-1] > q[0] * 1.12)
36
+ else:
37
+ # The plain tracker makes octave errors of its own, so its steps prove nothing either way.
38
+ out["tracker"] = "fallback autocorrelation: coarse. Install praat-parselmouth (pip, in a virtual environment) for a verdict on octave jumps and rising endings."
39
+ return out
40
+
41
+
42
+ if __name__ == "__main__":
43
+ p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
44
+ p.add_argument("files", nargs="+"); p.add_argument("--floor", type=float, default=70.0); p.add_argument("--ceil", type=float, default=400.0)
45
+ a = p.parse_args()
46
+ bad = False
47
+ for f in a.files:
48
+ r = check(f, a.floor, a.ceil)
49
+ bad = bad or r.get("octave_jump") or r.get("rises_at_end")
50
+ print(json.dumps(r, ensure_ascii=False))
51
+ sys.exit(1 if bad else 0)
@@ -0,0 +1,67 @@
1
+ #!/usr/bin/env python3
2
+ """Find where a line sits inside an assembled voice track, and replace part of it without touching any other sample.
3
+
4
+ place_line.py locate track.wav line.mp3 [--near 5.9]
5
+ place_line.py replace track.wav new.wav --at 8.485 --until 10.45 --from 2.705 [--match-old] --out track_new.wav
6
+
7
+ locate: slides 0.4 s windows of the line over the track and prints, for each, the offset where it fits best and how well. A line is often in the track
8
+ in several pieces with different offsets (a pause was shortened when the track was built), so one offset for the whole line is not enough.
9
+ replace: writes new[--from:] into the track from --at, silence up to --until, and leaves every sample outside --at..--until exactly as it was.
10
+ With --match-old the new audio is scaled to the RMS level of what it replaces. The report says where the new audio ends and how long the silence after it is.
11
+ """
12
+ import argparse, json, sys
13
+ import numpy as np
14
+ from _audio import SR, load, rms
15
+
16
+
17
+ def locate(track, line, near=None, sr=SR, win=0.4, step=0.6, reach=1.0):
18
+ out = []
19
+ t = 0.3
20
+ while t + win < len(line) / sr:
21
+ w = line[int(t * sr): int((t + win) * sr)]
22
+ if rms(w) > 0.005:
23
+ centre = (near if near is not None else 0.0) + t
24
+ lo = max(0, int((centre - reach) * sr)) if near is not None else 0
25
+ hi = min(len(track), int((centre + reach + win) * sr)) if near is not None else len(track)
26
+ seg = track[lo:hi]
27
+ c = np.correlate(seg, w, "valid")
28
+ energy = np.sqrt(np.convolve(seg ** 2, np.ones(len(w)), "valid") * (w ** 2).sum()) + 1e-12
29
+ k = int(np.argmax(np.abs(c) / energy))
30
+ out.append({"line_s": round(t, 2), "track_s": round((lo + k) / sr, 4), "line_start_in_track_s": round((lo + k) / sr - t, 4), "fit": round(float(abs(c[k]) / energy[k]), 3)})
31
+ t += step
32
+ return out
33
+
34
+
35
+ if __name__ == "__main__":
36
+ p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
37
+ sub = p.add_subparsers(dest="cmd", required=True)
38
+ l = sub.add_parser("locate"); l.add_argument("track"); l.add_argument("line"); l.add_argument("--near", type=float)
39
+ r = sub.add_parser("replace"); r.add_argument("track"); r.add_argument("new"); r.add_argument("--at", type=float, required=True); r.add_argument("--until", type=float, required=True)
40
+ r.add_argument("--from", dest="src_from", type=float, default=0.0); r.add_argument("--to", dest="src_to", type=float); r.add_argument("--match-old", action="store_true"); r.add_argument("--gain", type=float); r.add_argument("--out", required=True)
41
+ a = p.parse_args()
42
+ if a.cmd == "locate":
43
+ rows = locate(load(a.track), load(a.line), a.near)
44
+ offsets = sorted({round(x["line_start_in_track_s"], 2) for x in rows if x["fit"] > 0.8})
45
+ json.dump({"windows": rows, "offsets_found": offsets, "note": "more than one offset means the line is in the track in pieces" if len(offsets) > 1 else "one offset"}, sys.stdout, indent=1); print()
46
+ else:
47
+ import soundfile as sf
48
+ x, sr = sf.read(a.track, dtype="int16", always_2d=True)
49
+ new = load(a.new, sr)[int(a.src_from * sr): int(a.src_to * sr) if a.src_to else None]
50
+ s0, s1 = int(round(a.at * sr)), int(round(a.until * sr))
51
+ if len(new) > s1 - s0:
52
+ sys.exit(f"the new audio is {len(new) / sr:.3f} s but the slot is {(s1 - s0) / sr:.3f} s: trim a pause inside the take, do not move later lines")
53
+ old = x[s0:s1, 0].astype(np.float64) / 32768
54
+ active = lambda v: v[np.abs(v) > 0.003]
55
+ gain = a.gain if a.gain else (rms(active(old)) / max(rms(active(new)), 1e-9) if a.match_old else 1.0)
56
+ seg = np.zeros(s1 - s0); seg[: len(new)] = new * gain
57
+ n = int(0.004 * sr); seg[:n] *= np.linspace(0, 1, n)
58
+ q = np.clip(np.round(seg * 32767), -32768, 32767).astype("int16")
59
+ out = x.copy()
60
+ for ch in range(out.shape[1]):
61
+ out[s0:s1, ch] = q
62
+ assert np.array_equal(out[:s0], x[:s0]) and np.array_equal(out[s1:], x[s1:])
63
+ sf.write(a.out, out, sr, subtype="PCM_16")
64
+ after = np.nonzero(np.abs(x[s1:, 0]) > 30)[0]
65
+ end = (s0 + len(new)) / sr
66
+ json.dump({"written_s": [a.at, a.until], "new_audio_ends_s": round(end, 3), "next_sound_at_s": round((s1 + after[0]) / sr, 3) if len(after) else None,
67
+ "margin_s": round((s1 + after[0]) / sr - end, 3) if len(after) else None, "gain": round(gain, 3), "peak": round(float(np.abs(seg).max()), 3), "samples_outside_unchanged": True}, sys.stdout, indent=1); print()
@@ -0,0 +1,65 @@
1
+ #!/usr/bin/env python3
2
+ """Join two recordings at a quiet point, at a zero crossing, with a short crossfade, and report the levels and the pitch either side.
3
+
4
+ splice.py head.wav tail.wav --head-end 2.68 --tail-start 0.05 --out joined.wav [--gap 0.04] [--fade-ms 10] [--match-rms]
5
+
6
+ --head-end / --tail-start are where you want to cut, in seconds. Each is moved to the quietest point within --search seconds (a pause, the closure of a
7
+ stop consonant, or just before a fricative starts), and then to the nearest zero crossing, so the join has no click.
8
+ The crossfade is head[-n:] * fade-out + tail[:n] * fade-in over the same n samples, so the joined length is exactly head + gap + tail - n.
9
+ --match-rms scales the tail so that its level over --rms-window seconds next to the join equals the head's (a ratio of RMS levels over the spoken parts, pauses left out: a projection gain
10
+ reads low when the two takes differ in phase).
11
+ """
12
+ import argparse, json, sys
13
+ import numpy as np
14
+ from _audio import SR, db, f0_median, load, nearest_zero_crossing, quietest, rms, save, speech_rms
15
+
16
+
17
+ def splice(head, tail, head_end, tail_start, sr=SR, gap=0.0, fade_ms=10.0, search=0.03, match=False, rms_window=0.5):
18
+ a = quietest(head, max(0, int((head_end - search) * sr)), min(len(head), int((head_end + search) * sr)), sr)
19
+ a = nearest_zero_crossing(head, a, sr)
20
+ b = quietest(tail, max(0, int((tail_start - search) * sr)), min(len(tail), int((tail_start + search) * sr)), sr)
21
+ b = nearest_zero_crossing(tail, b, sr)
22
+ h, t = head[:a].copy(), tail[b:].copy()
23
+ w = int(rms_window * sr)
24
+ level_h, level_t = speech_rms(h[-w:], sr), speech_rms(t[:w], sr)
25
+ gain = level_h / level_t if match and level_t > 0 else 1.0
26
+ t *= gain
27
+ n = max(1, int(sr * fade_ms / 1000))
28
+ n = min(n, len(h), len(t))
29
+ ramp = np.linspace(0.0, 1.0, n)
30
+ if gap > 0:
31
+ h[-n:] *= ramp[::-1]
32
+ t[:n] *= ramp
33
+ out = np.concatenate([h, np.zeros(int(gap * sr)), t])
34
+ else:
35
+ mix = h[-n:] * ramp[::-1] + t[:n] * ramp
36
+ out = np.concatenate([h[:-n], mix, t[n:]])
37
+ report = {
38
+ "head_cut_s": round(a / sr, 4), "tail_cut_s": round(b / sr, 4), "join_at_s": round((len(h) - (0 if gap > 0 else n)) / sr, 4),
39
+ "length_s": round(len(out) / sr, 4), "expected_length_s": round((len(h) + int(gap * sr) + len(t) - (0 if gap > 0 else n)) / sr, 4),
40
+ "fade_ms": round(1000 * n / sr, 1), "gap_s": gap,
41
+ "level_at_cut_db": {"head": round(db(rms(head[max(0, a - 240): a + 240])), 1), "tail": round(db(rms(tail[max(0, b - 240): b + 240])), 1)},
42
+ "rms_db": {"head_side": round(db(level_h), 1), "tail_side_before_gain": round(db(level_t), 1), "gain_applied": round(gain, 3)},
43
+ "f0_hz": {"head_side": f0_median(h[-w:], sr), "tail_side": f0_median(t[:w], sr)},
44
+ "peak": round(float(np.abs(out).max()), 3),
45
+ }
46
+ f = report["f0_hz"]
47
+ if f["head_side"] and f["tail_side"]:
48
+ ratio = max(f["head_side"], f["tail_side"]) / min(f["head_side"], f["tail_side"])
49
+ report["f0_ratio"] = round(ratio, 2)
50
+ report["warnings"] = (["pitch differs by more than 25% across the join: listen for a jump"] if ratio > 1.25 else []) + (["pitch nearly doubles across the join: an octave jump"] if 1.8 < ratio < 2.2 else [])
51
+ if report["level_at_cut_db"]["head"] > -35 or report["level_at_cut_db"]["tail"] > -35:
52
+ report.setdefault("warnings", []).append("a cut point is not in a quiet place (above -35 dB): move it into a pause or a stop closure")
53
+ return out, report
54
+
55
+
56
+ if __name__ == "__main__":
57
+ p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
58
+ p.add_argument("head"); p.add_argument("tail")
59
+ p.add_argument("--head-end", type=float, required=True); p.add_argument("--tail-start", type=float, default=0.0)
60
+ p.add_argument("--out", required=True); p.add_argument("--gap", type=float, default=0.0); p.add_argument("--fade-ms", type=float, default=10.0)
61
+ p.add_argument("--search", type=float, default=0.03); p.add_argument("--match-rms", action="store_true"); p.add_argument("--rms-window", type=float, default=0.5)
62
+ a = p.parse_args()
63
+ out, report = splice(load(a.head), load(a.tail), a.head_end, a.tail_start, gap=a.gap, fade_ms=a.fade_ms, search=a.search, match=a.match_rms, rms_window=a.rms_window)
64
+ save(a.out, out)
65
+ json.dump(report, sys.stdout, indent=1, ensure_ascii=False); print()