reelkit-cli 0.10.5 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -1
- package/README.md +12 -2
- package/package.json +2 -1
- package/skill/SKILL.md +40 -5
- package/skill/reference/delivery.md +39 -0
- package/skill/reference/hebrew-rtl.md +10 -1
- package/skill/reference/reference-recreation.md +33 -0
- package/skill/reference/revisions.md +36 -0
- package/skill/reference/rights.md +27 -0
- package/skill/reference/voice-fixes.md +33 -0
- package/skill/reference/voice-sync.md +4 -0
- package/src/cli.ts +31 -2
- package/src/clock/clock.ts +76 -0
- package/src/commands/build.ts +82 -7
- package/src/commands/clock.ts +37 -0
- package/src/commands/diff.ts +81 -0
- package/src/commands/export.ts +51 -0
- package/src/commands/lint.ts +94 -0
- package/src/diff/framediff.ts +121 -0
- package/src/export/presets.ts +147 -0
- package/src/lint/pixel-rules.ts +149 -0
- package/src/lint/source-rules.ts +186 -0
- package/src/media/ffmpeg.ts +84 -0
- package/src/render/chunked.ts +139 -0
- package/src/render/render.ts +19 -2
- package/src/render/worker.ts +98 -54
- package/tools/voice/README.md +26 -0
- package/tools/voice/_audio.py +118 -0
- package/tools/voice/ab_video.py +93 -0
- package/tools/voice/credits.py +35 -0
- package/tools/voice/onsets.py +70 -0
- package/tools/voice/phonemes.py +85 -0
- package/tools/voice/pitch_check.py +51 -0
- package/tools/voice/place_line.py +67 -0
- package/tools/voice/splice.py +65 -0
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Where a word really starts in a recording, to place its caption within a frame.
|
|
3
|
+
|
|
4
|
+
onsets.py voice.wav --from 8.4 --to 9.3 [--fps 30] [--hf 5500]
|
|
5
|
+
|
|
6
|
+
Prints the level of each 10 ms window between --from and --to in two bands (everything, and above --hf Hz), then the moments that look like onsets:
|
|
7
|
+
a rise out of a pause, the start of a fricative (s, sh, ts, f: high-band energy with little low-band energy), and the release after a stop closure
|
|
8
|
+
(a dip of 30 ms or more). Speech-to-text word times are good to about 60 ms and are thrown off by pauses; these are good to about 10 ms.
|
|
9
|
+
Place a caption one or two frames before the onset. Each time is given in seconds and as a frame at --fps.
|
|
10
|
+
"""
|
|
11
|
+
import argparse, json, sys
|
|
12
|
+
import numpy as np
|
|
13
|
+
from _audio import SR, envelope_db, load
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def bands(x, sr, hf):
|
|
17
|
+
spec = np.fft.rfft(x)
|
|
18
|
+
freqs = np.fft.rfftfreq(len(x), 1 / sr)
|
|
19
|
+
hi = np.fft.irfft(np.where(freqs >= hf, spec, 0), len(x))
|
|
20
|
+
lo = np.fft.irfft(np.where(freqs <= 1500, spec, 0), len(x))
|
|
21
|
+
return hi, lo
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def onsets(x, sr=SR, t0=0.0, hf=5500.0, win=0.01):
|
|
25
|
+
hi, lo = bands(x, sr, hf)
|
|
26
|
+
full, e_hi, e_lo = envelope_db(x, sr, win), envelope_db(hi, sr, win), envelope_db(lo, sr, win)
|
|
27
|
+
peak = full.max() if len(full) else 0
|
|
28
|
+
found = []
|
|
29
|
+
for i in range(1, len(full)):
|
|
30
|
+
t = t0 + i * win
|
|
31
|
+
# out of a pause: at least 30 ms more than 30 dB under the peak, then a rise
|
|
32
|
+
if i >= 3 and (full[i - 3:i] < peak - 30).all() and full[i] > peak - 24:
|
|
33
|
+
found.append({"t": round(t, 3), "kind": "rise out of a pause"})
|
|
34
|
+
# a fricative: the high band comes up while the low band stays down
|
|
35
|
+
if e_hi[i] > e_hi.max() - 12 and e_hi[i - 1] < e_hi.max() - 18 and e_lo[i] < e_lo.max() - 12:
|
|
36
|
+
found.append({"t": round(t, 3), "kind": "fricative starts"})
|
|
37
|
+
# release after a stop closure: a dip of 30 ms or more inside speech
|
|
38
|
+
i = 1
|
|
39
|
+
while i < len(full) - 1:
|
|
40
|
+
if full[i] < peak - 26 and full[i - 1] >= peak - 26:
|
|
41
|
+
j = i
|
|
42
|
+
while j < len(full) and full[j] < peak - 26:
|
|
43
|
+
j += 1
|
|
44
|
+
if 3 <= j - i <= 12 and j < len(full):
|
|
45
|
+
found.append({"t": round(t0 + j * win, 3), "kind": f"release after a {int((j - i) * win * 1000)} ms closure"})
|
|
46
|
+
i = j
|
|
47
|
+
else:
|
|
48
|
+
i += 1
|
|
49
|
+
return full, e_hi, e_lo, sorted(found, key=lambda f: f["t"])
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
if __name__ == "__main__":
|
|
53
|
+
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
54
|
+
p.add_argument("file"); p.add_argument("--from", dest="t0", type=float, default=0.0); p.add_argument("--to", dest="t1", type=float)
|
|
55
|
+
p.add_argument("--fps", type=float, default=30.0); p.add_argument("--hf", type=float, default=5500.0); p.add_argument("--json", action="store_true")
|
|
56
|
+
a = p.parse_args()
|
|
57
|
+
x = load(a.file)
|
|
58
|
+
x = x[int(a.t0 * SR): int(a.t1 * SR) if a.t1 else None]
|
|
59
|
+
full, hi, lo, found = onsets(x, SR, a.t0, a.hf)
|
|
60
|
+
for f in found:
|
|
61
|
+
f["frame"] = round(f["t"] * a.fps, 1)
|
|
62
|
+
if a.json:
|
|
63
|
+
json.dump({"onsets": found}, sys.stdout, indent=1); print()
|
|
64
|
+
else:
|
|
65
|
+
print("time all high low (dB per 10 ms)")
|
|
66
|
+
for i in range(len(full)):
|
|
67
|
+
print(f"{a.t0 + i * 0.01:6.2f} {full[i]:5.0f} {hi[i]:5.0f} {lo[i]:5.0f}")
|
|
68
|
+
print("\nonsets:")
|
|
69
|
+
for f in found:
|
|
70
|
+
print(f" {f['t']:.3f} s frame {f['frame']:.1f} {f['kind']}")
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""What a recording actually says, as phonemes, and whether it says what was meant.
|
|
3
|
+
|
|
4
|
+
phonemes.py take1.wav take2.wav [--want "m e a f j e n i m"] [--from 1.2 --to 2.3]
|
|
5
|
+
|
|
6
|
+
Speech-to-text writes the word it expects, so a mispronounced word still comes back spelled correctly. A phoneme recogniser (wav2vec2 trained on
|
|
7
|
+
espeak phonemes, any language) writes what was said. With --want, each file passes when the wanted phonemes appear in order (spaces and length marks
|
|
8
|
+
are ignored); the nearest stretch is shown when they do not. Also printed: the length of the stretch, so an over-processed take (stretched, pitched)
|
|
9
|
+
shows up as too long for its phonemes.
|
|
10
|
+
Check the clean voice, not the mix: music degrades the result. Needs `torch`, `transformers` and `huggingface_hub`; the model (about 1.2 GB) is
|
|
11
|
+
downloaded once. If they are missing, the script says so and exits with code 2 rather than failing a build.
|
|
12
|
+
"""
|
|
13
|
+
import argparse, json, subprocess, sys
|
|
14
|
+
|
|
15
|
+
np = None # numpy, imported with the other optional packages so that a missing one is a sentence, not a traceback
|
|
16
|
+
|
|
17
|
+
MODEL = "facebook/wav2vec2-xlsr-53-espeak-cv-ft"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def load16(path, t0=None, t1=None):
|
|
21
|
+
cut = (["-ss", str(t0)] if t0 else []) + (["-to", str(t1)] if t1 else [])
|
|
22
|
+
raw = subprocess.check_output(["ffmpeg", "-nostdin", "-v", "error", *cut, "-i", path, "-ac", "1", "-ar", "16000", "-f", "f32le", "-"])
|
|
23
|
+
a = np.frombuffer(raw, np.float32)
|
|
24
|
+
return (a - a.mean()) / (a.std() + 1e-7)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def squash(s):
|
|
28
|
+
return "".join(c for c in s if c not in " ːˈˌ")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def nearest(got, want):
|
|
32
|
+
"""The stretch of `got` closest to `want` by edit distance, and that distance."""
|
|
33
|
+
g, w = squash(got), squash(want)
|
|
34
|
+
best = (len(w), "")
|
|
35
|
+
for i in range(len(g)):
|
|
36
|
+
for n in (len(w) - 1, len(w), len(w) + 1):
|
|
37
|
+
s = g[i: i + n]
|
|
38
|
+
if not s:
|
|
39
|
+
continue
|
|
40
|
+
prev = list(range(len(w) + 1))
|
|
41
|
+
for a_ in s:
|
|
42
|
+
cur = [prev[0] + 1]
|
|
43
|
+
for j, b_ in enumerate(w, 1):
|
|
44
|
+
cur.append(min(prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + (a_ != b_)))
|
|
45
|
+
prev = cur
|
|
46
|
+
if prev[-1] < best[0]:
|
|
47
|
+
best = (prev[-1], s)
|
|
48
|
+
return best
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
if __name__ == "__main__":
|
|
52
|
+
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
53
|
+
p.add_argument("files", nargs="+"); p.add_argument("--want"); p.add_argument("--from", dest="t0", type=float); p.add_argument("--to", dest="t1", type=float)
|
|
54
|
+
a = p.parse_args()
|
|
55
|
+
try:
|
|
56
|
+
import numpy as np
|
|
57
|
+
import torch
|
|
58
|
+
from transformers import Wav2Vec2ForCTC
|
|
59
|
+
from huggingface_hub import hf_hub_download
|
|
60
|
+
except ImportError as e:
|
|
61
|
+
print(f"phoneme check skipped: {e.name} is not installed. In a virtual environment: pip install torch transformers huggingface_hub", file=sys.stderr)
|
|
62
|
+
sys.exit(2)
|
|
63
|
+
model = Wav2Vec2ForCTC.from_pretrained(MODEL).eval()
|
|
64
|
+
vocab = json.load(open(hf_hub_download(MODEL, "vocab.json")))
|
|
65
|
+
inv = {v: k for k, v in vocab.items()}
|
|
66
|
+
ok = True
|
|
67
|
+
for f in a.files:
|
|
68
|
+
x = load16(f, a.t0, a.t1)
|
|
69
|
+
with torch.no_grad():
|
|
70
|
+
ids = model(torch.tensor(x)[None]).logits[0].argmax(-1).tolist()
|
|
71
|
+
out, prev = [], None
|
|
72
|
+
for i in ids:
|
|
73
|
+
if i != prev and inv[i] not in ("<pad>", "<s>", "</s>", "<unk>", "|"):
|
|
74
|
+
out.append(inv[i])
|
|
75
|
+
prev = i
|
|
76
|
+
said = " ".join(out)
|
|
77
|
+
row = {"file": f, "seconds": round(len(x) / 16000, 2), "phonemes": said}
|
|
78
|
+
if a.want:
|
|
79
|
+
row["has_wanted"] = squash(a.want) in squash(said)
|
|
80
|
+
if not row["has_wanted"]:
|
|
81
|
+
dist, near = nearest(said, a.want)
|
|
82
|
+
row["nearest"] = near; row["distance"] = dist
|
|
83
|
+
ok = False
|
|
84
|
+
print(json.dumps(row, ensure_ascii=False))
|
|
85
|
+
sys.exit(0 if ok else 1)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Does a take jump an octave, or end like a question?
|
|
3
|
+
|
|
4
|
+
pitch_check.py take1.wav take2.wav [--floor 70 --ceil 400]
|
|
5
|
+
|
|
6
|
+
For each file: the pitch range, the largest step between neighbouring 10 ms frames (a ratio near 2 is an octave jump, the commonest fault of a
|
|
7
|
+
synthetic voice), and the pitch over the last voiced 0.6 s in four quarters. A statement falls; a line whose last quarter is clearly above its
|
|
8
|
+
first reads as a question. The verdicts (octave_jump, rises_at_end) are given only when Praat is installed (`praat-parselmouth`); the built-in
|
|
9
|
+
fallback tracker makes octave errors of its own, so with it only the numbers are printed and the exit code is 0.
|
|
10
|
+
"""
|
|
11
|
+
import argparse, json, sys
|
|
12
|
+
import numpy as np
|
|
13
|
+
from _audio import SR, f0_track, load
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def has_praat():
|
|
17
|
+
try:
|
|
18
|
+
import parselmouth # noqa: F401
|
|
19
|
+
return True
|
|
20
|
+
except ImportError:
|
|
21
|
+
return False
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def check(path, lo, hi):
|
|
25
|
+
t, f = f0_track(load(path), SR, lo, hi)
|
|
26
|
+
if len(f) < 8:
|
|
27
|
+
return {"file": path, "voiced": False}
|
|
28
|
+
steps = [max(f[i + 1] / f[i], f[i] / f[i + 1]) for i in range(len(f) - 1) if t[i + 1] - t[i] < 0.025]
|
|
29
|
+
last = f[t > t[-1] - 0.6]
|
|
30
|
+
q = [float(np.median(p)) for p in np.array_split(last, 4) if len(p)]
|
|
31
|
+
worst = max(steps) if steps else 1.0
|
|
32
|
+
out = {"file": path, "voiced": True, "range_hz": [round(float(f.min())), round(float(f.max()))], "max_step_ratio": round(float(worst), 2), "last_quarters_hz": [round(v) for v in q]}
|
|
33
|
+
if has_praat():
|
|
34
|
+
out["octave_jump"] = bool(1.8 < worst < 2.25)
|
|
35
|
+
out["rises_at_end"] = bool(len(q) == 4 and q[-1] > q[0] * 1.12)
|
|
36
|
+
else:
|
|
37
|
+
# The plain tracker makes octave errors of its own, so its steps prove nothing either way.
|
|
38
|
+
out["tracker"] = "fallback autocorrelation: coarse. Install praat-parselmouth (pip, in a virtual environment) for a verdict on octave jumps and rising endings."
|
|
39
|
+
return out
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
if __name__ == "__main__":
|
|
43
|
+
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
44
|
+
p.add_argument("files", nargs="+"); p.add_argument("--floor", type=float, default=70.0); p.add_argument("--ceil", type=float, default=400.0)
|
|
45
|
+
a = p.parse_args()
|
|
46
|
+
bad = False
|
|
47
|
+
for f in a.files:
|
|
48
|
+
r = check(f, a.floor, a.ceil)
|
|
49
|
+
bad = bad or r.get("octave_jump") or r.get("rises_at_end")
|
|
50
|
+
print(json.dumps(r, ensure_ascii=False))
|
|
51
|
+
sys.exit(1 if bad else 0)
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Find where a line sits inside an assembled voice track, and replace part of it without touching any other sample.
|
|
3
|
+
|
|
4
|
+
place_line.py locate track.wav line.mp3 [--near 5.9]
|
|
5
|
+
place_line.py replace track.wav new.wav --at 8.485 --until 10.45 --from 2.705 [--match-old] --out track_new.wav
|
|
6
|
+
|
|
7
|
+
locate: slides 0.4 s windows of the line over the track and prints, for each, the offset where it fits best and how well. A line is often in the track
|
|
8
|
+
in several pieces with different offsets (a pause was shortened when the track was built), so one offset for the whole line is not enough.
|
|
9
|
+
replace: writes new[--from:] into the track from --at, silence up to --until, and leaves every sample outside --at..--until exactly as it was.
|
|
10
|
+
With --match-old the new audio is scaled to the RMS level of what it replaces. The report says where the new audio ends and how long the silence after it is.
|
|
11
|
+
"""
|
|
12
|
+
import argparse, json, sys
|
|
13
|
+
import numpy as np
|
|
14
|
+
from _audio import SR, load, rms
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def locate(track, line, near=None, sr=SR, win=0.4, step=0.6, reach=1.0):
|
|
18
|
+
out = []
|
|
19
|
+
t = 0.3
|
|
20
|
+
while t + win < len(line) / sr:
|
|
21
|
+
w = line[int(t * sr): int((t + win) * sr)]
|
|
22
|
+
if rms(w) > 0.005:
|
|
23
|
+
centre = (near if near is not None else 0.0) + t
|
|
24
|
+
lo = max(0, int((centre - reach) * sr)) if near is not None else 0
|
|
25
|
+
hi = min(len(track), int((centre + reach + win) * sr)) if near is not None else len(track)
|
|
26
|
+
seg = track[lo:hi]
|
|
27
|
+
c = np.correlate(seg, w, "valid")
|
|
28
|
+
energy = np.sqrt(np.convolve(seg ** 2, np.ones(len(w)), "valid") * (w ** 2).sum()) + 1e-12
|
|
29
|
+
k = int(np.argmax(np.abs(c) / energy))
|
|
30
|
+
out.append({"line_s": round(t, 2), "track_s": round((lo + k) / sr, 4), "line_start_in_track_s": round((lo + k) / sr - t, 4), "fit": round(float(abs(c[k]) / energy[k]), 3)})
|
|
31
|
+
t += step
|
|
32
|
+
return out
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
if __name__ == "__main__":
|
|
36
|
+
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
37
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
38
|
+
l = sub.add_parser("locate"); l.add_argument("track"); l.add_argument("line"); l.add_argument("--near", type=float)
|
|
39
|
+
r = sub.add_parser("replace"); r.add_argument("track"); r.add_argument("new"); r.add_argument("--at", type=float, required=True); r.add_argument("--until", type=float, required=True)
|
|
40
|
+
r.add_argument("--from", dest="src_from", type=float, default=0.0); r.add_argument("--to", dest="src_to", type=float); r.add_argument("--match-old", action="store_true"); r.add_argument("--gain", type=float); r.add_argument("--out", required=True)
|
|
41
|
+
a = p.parse_args()
|
|
42
|
+
if a.cmd == "locate":
|
|
43
|
+
rows = locate(load(a.track), load(a.line), a.near)
|
|
44
|
+
offsets = sorted({round(x["line_start_in_track_s"], 2) for x in rows if x["fit"] > 0.8})
|
|
45
|
+
json.dump({"windows": rows, "offsets_found": offsets, "note": "more than one offset means the line is in the track in pieces" if len(offsets) > 1 else "one offset"}, sys.stdout, indent=1); print()
|
|
46
|
+
else:
|
|
47
|
+
import soundfile as sf
|
|
48
|
+
x, sr = sf.read(a.track, dtype="int16", always_2d=True)
|
|
49
|
+
new = load(a.new, sr)[int(a.src_from * sr): int(a.src_to * sr) if a.src_to else None]
|
|
50
|
+
s0, s1 = int(round(a.at * sr)), int(round(a.until * sr))
|
|
51
|
+
if len(new) > s1 - s0:
|
|
52
|
+
sys.exit(f"the new audio is {len(new) / sr:.3f} s but the slot is {(s1 - s0) / sr:.3f} s: trim a pause inside the take, do not move later lines")
|
|
53
|
+
old = x[s0:s1, 0].astype(np.float64) / 32768
|
|
54
|
+
active = lambda v: v[np.abs(v) > 0.003]
|
|
55
|
+
gain = a.gain if a.gain else (rms(active(old)) / max(rms(active(new)), 1e-9) if a.match_old else 1.0)
|
|
56
|
+
seg = np.zeros(s1 - s0); seg[: len(new)] = new * gain
|
|
57
|
+
n = int(0.004 * sr); seg[:n] *= np.linspace(0, 1, n)
|
|
58
|
+
q = np.clip(np.round(seg * 32767), -32768, 32767).astype("int16")
|
|
59
|
+
out = x.copy()
|
|
60
|
+
for ch in range(out.shape[1]):
|
|
61
|
+
out[s0:s1, ch] = q
|
|
62
|
+
assert np.array_equal(out[:s0], x[:s0]) and np.array_equal(out[s1:], x[s1:])
|
|
63
|
+
sf.write(a.out, out, sr, subtype="PCM_16")
|
|
64
|
+
after = np.nonzero(np.abs(x[s1:, 0]) > 30)[0]
|
|
65
|
+
end = (s0 + len(new)) / sr
|
|
66
|
+
json.dump({"written_s": [a.at, a.until], "new_audio_ends_s": round(end, 3), "next_sound_at_s": round((s1 + after[0]) / sr, 3) if len(after) else None,
|
|
67
|
+
"margin_s": round((s1 + after[0]) / sr - end, 3) if len(after) else None, "gain": round(gain, 3), "peak": round(float(np.abs(seg).max()), 3), "samples_outside_unchanged": True}, sys.stdout, indent=1); print()
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Join two recordings at a quiet point, at a zero crossing, with a short crossfade, and report the levels and the pitch either side.
|
|
3
|
+
|
|
4
|
+
splice.py head.wav tail.wav --head-end 2.68 --tail-start 0.05 --out joined.wav [--gap 0.04] [--fade-ms 10] [--match-rms]
|
|
5
|
+
|
|
6
|
+
--head-end / --tail-start are where you want to cut, in seconds. Each is moved to the quietest point within --search seconds (a pause, the closure of a
|
|
7
|
+
stop consonant, or just before a fricative starts), and then to the nearest zero crossing, so the join has no click.
|
|
8
|
+
The crossfade is head[-n:] * fade-out + tail[:n] * fade-in over the same n samples, so the joined length is exactly head + gap + tail - n.
|
|
9
|
+
--match-rms scales the tail so that its level over --rms-window seconds next to the join equals the head's (a ratio of RMS levels over the spoken parts, pauses left out: a projection gain
|
|
10
|
+
reads low when the two takes differ in phase).
|
|
11
|
+
"""
|
|
12
|
+
import argparse, json, sys
|
|
13
|
+
import numpy as np
|
|
14
|
+
from _audio import SR, db, f0_median, load, nearest_zero_crossing, quietest, rms, save, speech_rms
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def splice(head, tail, head_end, tail_start, sr=SR, gap=0.0, fade_ms=10.0, search=0.03, match=False, rms_window=0.5):
|
|
18
|
+
a = quietest(head, max(0, int((head_end - search) * sr)), min(len(head), int((head_end + search) * sr)), sr)
|
|
19
|
+
a = nearest_zero_crossing(head, a, sr)
|
|
20
|
+
b = quietest(tail, max(0, int((tail_start - search) * sr)), min(len(tail), int((tail_start + search) * sr)), sr)
|
|
21
|
+
b = nearest_zero_crossing(tail, b, sr)
|
|
22
|
+
h, t = head[:a].copy(), tail[b:].copy()
|
|
23
|
+
w = int(rms_window * sr)
|
|
24
|
+
level_h, level_t = speech_rms(h[-w:], sr), speech_rms(t[:w], sr)
|
|
25
|
+
gain = level_h / level_t if match and level_t > 0 else 1.0
|
|
26
|
+
t *= gain
|
|
27
|
+
n = max(1, int(sr * fade_ms / 1000))
|
|
28
|
+
n = min(n, len(h), len(t))
|
|
29
|
+
ramp = np.linspace(0.0, 1.0, n)
|
|
30
|
+
if gap > 0:
|
|
31
|
+
h[-n:] *= ramp[::-1]
|
|
32
|
+
t[:n] *= ramp
|
|
33
|
+
out = np.concatenate([h, np.zeros(int(gap * sr)), t])
|
|
34
|
+
else:
|
|
35
|
+
mix = h[-n:] * ramp[::-1] + t[:n] * ramp
|
|
36
|
+
out = np.concatenate([h[:-n], mix, t[n:]])
|
|
37
|
+
report = {
|
|
38
|
+
"head_cut_s": round(a / sr, 4), "tail_cut_s": round(b / sr, 4), "join_at_s": round((len(h) - (0 if gap > 0 else n)) / sr, 4),
|
|
39
|
+
"length_s": round(len(out) / sr, 4), "expected_length_s": round((len(h) + int(gap * sr) + len(t) - (0 if gap > 0 else n)) / sr, 4),
|
|
40
|
+
"fade_ms": round(1000 * n / sr, 1), "gap_s": gap,
|
|
41
|
+
"level_at_cut_db": {"head": round(db(rms(head[max(0, a - 240): a + 240])), 1), "tail": round(db(rms(tail[max(0, b - 240): b + 240])), 1)},
|
|
42
|
+
"rms_db": {"head_side": round(db(level_h), 1), "tail_side_before_gain": round(db(level_t), 1), "gain_applied": round(gain, 3)},
|
|
43
|
+
"f0_hz": {"head_side": f0_median(h[-w:], sr), "tail_side": f0_median(t[:w], sr)},
|
|
44
|
+
"peak": round(float(np.abs(out).max()), 3),
|
|
45
|
+
}
|
|
46
|
+
f = report["f0_hz"]
|
|
47
|
+
if f["head_side"] and f["tail_side"]:
|
|
48
|
+
ratio = max(f["head_side"], f["tail_side"]) / min(f["head_side"], f["tail_side"])
|
|
49
|
+
report["f0_ratio"] = round(ratio, 2)
|
|
50
|
+
report["warnings"] = (["pitch differs by more than 25% across the join: listen for a jump"] if ratio > 1.25 else []) + (["pitch nearly doubles across the join: an octave jump"] if 1.8 < ratio < 2.2 else [])
|
|
51
|
+
if report["level_at_cut_db"]["head"] > -35 or report["level_at_cut_db"]["tail"] > -35:
|
|
52
|
+
report.setdefault("warnings", []).append("a cut point is not in a quiet place (above -35 dB): move it into a pause or a stop closure")
|
|
53
|
+
return out, report
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
if __name__ == "__main__":
|
|
57
|
+
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
58
|
+
p.add_argument("head"); p.add_argument("tail")
|
|
59
|
+
p.add_argument("--head-end", type=float, required=True); p.add_argument("--tail-start", type=float, default=0.0)
|
|
60
|
+
p.add_argument("--out", required=True); p.add_argument("--gap", type=float, default=0.0); p.add_argument("--fade-ms", type=float, default=10.0)
|
|
61
|
+
p.add_argument("--search", type=float, default=0.03); p.add_argument("--match-rms", action="store_true"); p.add_argument("--rms-window", type=float, default=0.5)
|
|
62
|
+
a = p.parse_args()
|
|
63
|
+
out, report = splice(load(a.head), load(a.tail), a.head_end, a.tail_start, gap=a.gap, fade_ms=a.fade_ms, search=a.search, match=a.match_rms, rms_window=a.rms_window)
|
|
64
|
+
save(a.out, out)
|
|
65
|
+
json.dump(report, sys.stdout, indent=1, ensure_ascii=False); print()
|