recito 0.7.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- recito/__init__.py +5 -0
- recito/_version.py +24 -0
- recito/assemble.py +608 -0
- recito/chunk.py +119 -0
- recito/clean.py +2497 -0
- recito/config.py +282 -0
- recito/cover.py +71 -0
- recito/engines/__init__.py +31 -0
- recito/engines/base.py +245 -0
- recito/engines/qwen3.py +194 -0
- recito/engines/tone.py +50 -0
- recito/engines/vllm_omni.py +435 -0
- recito/extract.py +71 -0
- recito/extract_epub.py +799 -0
- recito/extract_pdf.py +1989 -0
- recito/main.py +661 -0
- recito/models.py +463 -0
- recito/synthesize.py +3710 -0
- recito/textnorm/__init__.py +243 -0
- recito/textnorm/backends/__init__.py +1 -0
- recito/textnorm/backends/basic.py +50 -0
- recito/textnorm/backends/own_en.py +1130 -0
- recito/textnorm/data/__init__.py +63 -0
- recito/textnorm/data/en/acronyms.tsv +39 -0
- recito/textnorm/data/en/common.tsv +286607 -0
- recito/textnorm/data/en/currency.tsv +6 -0
- recito/textnorm/data/en/glue.tsv +11 -0
- recito/textnorm/data/en/hotkeys.tsv +35 -0
- recito/textnorm/data/en/months.tsv +18 -0
- recito/textnorm/data/en/seconds_context.tsv +11 -0
- recito/textnorm/data/en/symbols.tsv +11 -0
- recito/textnorm/data/en/units.tsv +40 -0
- recito/textnorm/data/en/whitelist.tsv +58 -0
- recito/textnorm/data/en/year_context.tsv +23 -0
- recito/textnorm/overlay.py +116 -0
- recito/textnorm/prepare.py +879 -0
- recito/textnorm/profile.py +795 -0
- recito/textnorm/rules_en.py +46 -0
- recito/textnorm/segment.py +265 -0
- recito/textnorm/urls.py +366 -0
- recito/textnorm/verbalize.py +89 -0
- recito-0.7.2.dist-info/METADATA +123 -0
- recito-0.7.2.dist-info/RECORD +47 -0
- recito-0.7.2.dist-info/WHEEL +5 -0
- recito-0.7.2.dist-info/entry_points.txt +2 -0
- recito-0.7.2.dist-info/licenses/LICENSE.txt +661 -0
- recito-0.7.2.dist-info/top_level.txt +1 -0
recito/__init__.py
ADDED
recito/_version.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# file generated by vcs-versioning
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"__version__",
|
|
7
|
+
"__version_tuple__",
|
|
8
|
+
"version",
|
|
9
|
+
"version_tuple",
|
|
10
|
+
"__commit_id__",
|
|
11
|
+
"commit_id",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
version: str
|
|
15
|
+
__version__: str
|
|
16
|
+
__version_tuple__: tuple[int | str, ...]
|
|
17
|
+
version_tuple: tuple[int | str, ...]
|
|
18
|
+
commit_id: str | None
|
|
19
|
+
__commit_id__: str | None
|
|
20
|
+
|
|
21
|
+
__version__ = version = '0.7.2'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 7, 2)
|
|
23
|
+
|
|
24
|
+
__commit_id__ = commit_id = None
|
recito/assemble.py
ADDED
|
@@ -0,0 +1,608 @@
|
|
|
1
|
+
"""Stage 5 — ASSEMBLE: cached chunk wavs -> final audiobook file.
|
|
2
|
+
|
|
3
|
+
Per chunk: leading/trailing silence is trimmed (energy threshold relative
|
|
4
|
+
to peak, tts-audiobook-tool pattern, MIT) so chunk joins aren't choppy.
|
|
5
|
+
Per chapter: chunks and boundary silences are concatenated in memory and
|
|
6
|
+
written as FLAC intermediates, one per chapter, built in parallel.
|
|
7
|
+
|
|
8
|
+
Every ffmpeg audio codec/filter used here is single-threaded, so wall
|
|
9
|
+
time is governed by the longest serial pass over the audio. All passes
|
|
10
|
+
are therefore kept chapter-sized and run in parallel: each chapter gets
|
|
11
|
+
a two-pass EBU R128 loudnorm (I=-14 / TP=-3 / LRA=7, linear mode
|
|
12
|
+
honoring the true-peak ceiling) measured and applied per chapter, then
|
|
13
|
+
is encoded to the target codec. Chapter-level normalization is audibly
|
|
14
|
+
identical to book-wide for TTS narration (chapter loudness varies by
|
|
15
|
+
~1 LU) while keeping chapter-to-chapter levels consistent. The encoded
|
|
16
|
+
chapter segments are concatenated with stream copy (no re-encode) and
|
|
17
|
+
muxed with ffmetadata chapter marks; AAC encoder priming at each join
|
|
18
|
+
(~50 ms) lands inside the boundary silence.
|
|
19
|
+
|
|
20
|
+
:class:`ChapterPipeline` drives the per-chapter work; it can run after
|
|
21
|
+
synthesis (:func:`assemble`) or overlapped with it, firing each
|
|
22
|
+
chapter's task as soon as that chapter's audio is final in the chunk
|
|
23
|
+
cache (the output is identical either way).
|
|
24
|
+
|
|
25
|
+
Metadata is written as classic iTunes-style ilst atoms
|
|
26
|
+
(``mdta``-style ``use_metadata_tags`` is deliberately avoided: most
|
|
27
|
+
players — Cozy, VLC, mutagen-based tools — cannot read it), and a cover
|
|
28
|
+
image is embedded as an attached_pic stream (covr/APIC/PICTURE). mp3 and
|
|
29
|
+
flac outputs are also supported (no chapter marks — container
|
|
30
|
+
limitation).
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
import json
|
|
36
|
+
import os
|
|
37
|
+
import re
|
|
38
|
+
import shutil
|
|
39
|
+
import subprocess
|
|
40
|
+
from concurrent.futures import Future, ThreadPoolExecutor
|
|
41
|
+
from dataclasses import dataclass
|
|
42
|
+
from pathlib import Path
|
|
43
|
+
from typing import Self
|
|
44
|
+
|
|
45
|
+
import numpy as np
|
|
46
|
+
import soundfile as sf
|
|
47
|
+
|
|
48
|
+
from recito.config import Pauses
|
|
49
|
+
from recito.models import BookMeta, Boundary, Chunk, Manifest
|
|
50
|
+
|
|
51
|
+
#: Silence-trim threshold relative to each chunk's peak RMS (dB).
|
|
52
|
+
_TRIM_THRESHOLD_DB = -42.0
|
|
53
|
+
#: Audio retained before/after the detected speech (seconds).
|
|
54
|
+
_TRIM_PAD_S = 0.04
|
|
55
|
+
|
|
56
|
+
_LOUDNORM = "I=-14:TP=-3:LRA=7"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class AssemblyError(Exception):
|
|
60
|
+
"""Missing audio, ffmpeg failure, or unsupported combination."""
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class AssemblyResult:
|
|
65
|
+
output: Path
|
|
66
|
+
duration_s: float
|
|
67
|
+
chapters: list[tuple[str, float]] # (title, duration seconds)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# ---------------------------------------------------------------------------
|
|
71
|
+
# ffmpeg/ffprobe helpers
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def require_ffmpeg() -> None:
|
|
76
|
+
for tool in ("ffmpeg", "ffprobe"):
|
|
77
|
+
if shutil.which(tool) is None:
|
|
78
|
+
raise AssemblyError(
|
|
79
|
+
f"{tool} not found on PATH; install ffmpeg (see README)"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _run(cmd: list[str], what: str) -> subprocess.CompletedProcess:
|
|
84
|
+
# stdin must not be the tty: ffmpeg's interactive keyboard mode would
|
|
85
|
+
# snapshot/restore termios per process, and with concurrent chapter
|
|
86
|
+
# tasks (or a child orphaned by an abort) the last restore can be a
|
|
87
|
+
# stale no-echo snapshot — leaving the terminal echo off after the run.
|
|
88
|
+
proc = subprocess.run(cmd, capture_output=True, text=True, stdin=subprocess.DEVNULL)
|
|
89
|
+
if proc.returncode != 0:
|
|
90
|
+
tail = "\n".join(proc.stderr.strip().splitlines()[-12:])
|
|
91
|
+
raise AssemblyError(f"{what} failed (exit {proc.returncode}):\n{tail}")
|
|
92
|
+
return proc
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _probe_duration(path: Path) -> float:
|
|
96
|
+
proc = _run(
|
|
97
|
+
[
|
|
98
|
+
"ffprobe",
|
|
99
|
+
"-v",
|
|
100
|
+
"error",
|
|
101
|
+
"-show_entries",
|
|
102
|
+
"format=duration",
|
|
103
|
+
"-of",
|
|
104
|
+
"default=noprint_wrappers=1:nokey=1",
|
|
105
|
+
str(path),
|
|
106
|
+
],
|
|
107
|
+
f"ffprobe {path.name}",
|
|
108
|
+
)
|
|
109
|
+
return float(proc.stdout.strip())
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _concat_demuxer(
|
|
113
|
+
inputs: list[Path], output: Path, extra: list[str], what: str
|
|
114
|
+
) -> None:
|
|
115
|
+
# Absolute paths: the demuxer resolves relative entries against the
|
|
116
|
+
# list file's directory, not the process cwd.
|
|
117
|
+
list_file = output.with_suffix(".concat.txt").absolute()
|
|
118
|
+
# ffmpeg concat-demuxer quoting: wrap in single quotes, escape inner ones.
|
|
119
|
+
body = "\n".join(
|
|
120
|
+
f"file '{str(p.absolute()).replace(chr(39), chr(39) + chr(92) + chr(39) + chr(39))}'"
|
|
121
|
+
for p in inputs
|
|
122
|
+
)
|
|
123
|
+
list_file.write_text(body + "\n")
|
|
124
|
+
try:
|
|
125
|
+
_run(
|
|
126
|
+
[
|
|
127
|
+
"ffmpeg",
|
|
128
|
+
"-hide_banner",
|
|
129
|
+
"-nostats",
|
|
130
|
+
"-y",
|
|
131
|
+
"-f",
|
|
132
|
+
"concat",
|
|
133
|
+
"-safe",
|
|
134
|
+
"0",
|
|
135
|
+
"-i",
|
|
136
|
+
str(list_file),
|
|
137
|
+
*extra,
|
|
138
|
+
str(output.absolute()),
|
|
139
|
+
],
|
|
140
|
+
what,
|
|
141
|
+
)
|
|
142
|
+
finally:
|
|
143
|
+
list_file.unlink(missing_ok=True)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# ---------------------------------------------------------------------------
|
|
147
|
+
# Silence trimming
|
|
148
|
+
# ---------------------------------------------------------------------------
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def trim_silence(wav: np.ndarray, sample_rate: int) -> np.ndarray:
|
|
152
|
+
"""Trim leading/trailing silence using peak-relative frame RMS."""
|
|
153
|
+
if wav.ndim != 1 or len(wav) == 0:
|
|
154
|
+
return wav
|
|
155
|
+
frame = max(1, int(0.025 * sample_rate))
|
|
156
|
+
n_frames = len(wav) // frame
|
|
157
|
+
if n_frames < 3:
|
|
158
|
+
return wav
|
|
159
|
+
frames = wav[: n_frames * frame].reshape(n_frames, frame)
|
|
160
|
+
rms = np.sqrt(np.mean(frames * frames, axis=1))
|
|
161
|
+
peak = float(rms.max())
|
|
162
|
+
if peak <= 0:
|
|
163
|
+
return wav
|
|
164
|
+
threshold = peak * (10.0 ** (_TRIM_THRESHOLD_DB / 20.0))
|
|
165
|
+
active = np.flatnonzero(rms >= threshold)
|
|
166
|
+
if len(active) == 0:
|
|
167
|
+
return wav
|
|
168
|
+
pad = int(_TRIM_PAD_S * sample_rate)
|
|
169
|
+
start = max(0, active[0] * frame - pad)
|
|
170
|
+
end = min(len(wav), (active[-1] + 1) * frame + pad)
|
|
171
|
+
return wav[start:end]
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# ---------------------------------------------------------------------------
|
|
175
|
+
# ffmetadata
|
|
176
|
+
# ---------------------------------------------------------------------------
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _escape_meta(value: str) -> str:
|
|
180
|
+
return re.sub(r"([=;#\\\n])", lambda m: "\\" + m.group(1), value)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _ffmetadata(meta: BookMeta, chapters: list[tuple[str, float]]) -> str:
|
|
184
|
+
out = [";FFMETADATA1"]
|
|
185
|
+
# iTunes/ilst-style keys; ffmpeg maps them to the classic atoms
|
|
186
|
+
# (composer = author, artist = narrator: Cozy's author/reader fields).
|
|
187
|
+
tags = {
|
|
188
|
+
"title": meta.title,
|
|
189
|
+
"album": meta.title, # book-grouping key in players (e.g. Cozy)
|
|
190
|
+
"artist": meta.narrator,
|
|
191
|
+
"composer": meta.author,
|
|
192
|
+
"album_artist": meta.author,
|
|
193
|
+
"date": meta.date,
|
|
194
|
+
"genre": meta.genre,
|
|
195
|
+
"description": meta.description,
|
|
196
|
+
}
|
|
197
|
+
for key, value in tags.items():
|
|
198
|
+
value = value.strip()
|
|
199
|
+
if value:
|
|
200
|
+
out.append(f"{key}={_escape_meta(value)}")
|
|
201
|
+
start_ms = 0
|
|
202
|
+
for chapter_title, duration in chapters:
|
|
203
|
+
end_ms = start_ms + round(duration * 1000)
|
|
204
|
+
out.append("[CHAPTER]")
|
|
205
|
+
out.append("TIMEBASE=1/1000")
|
|
206
|
+
out.append(f"START={start_ms}")
|
|
207
|
+
out.append(f"END={end_ms}")
|
|
208
|
+
out.append(f"title={_escape_meta(chapter_title)}")
|
|
209
|
+
start_ms = end_ms
|
|
210
|
+
return "\n".join(out) + "\n"
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
# ---------------------------------------------------------------------------
|
|
214
|
+
# Loudnorm (two-pass, per chapter)
|
|
215
|
+
# ---------------------------------------------------------------------------
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _loudnorm_measure(source: Path) -> dict[str, str]:
|
|
219
|
+
proc = _run(
|
|
220
|
+
[
|
|
221
|
+
"ffmpeg",
|
|
222
|
+
"-hide_banner",
|
|
223
|
+
"-nostats",
|
|
224
|
+
"-i",
|
|
225
|
+
str(source),
|
|
226
|
+
"-af",
|
|
227
|
+
f"loudnorm={_LOUDNORM}:print_format=json",
|
|
228
|
+
"-f",
|
|
229
|
+
"null",
|
|
230
|
+
"-",
|
|
231
|
+
],
|
|
232
|
+
"loudnorm measurement pass",
|
|
233
|
+
)
|
|
234
|
+
match = re.search(r"\{[^{}]*\"input_i\"[^{}]*\}", proc.stderr, re.DOTALL)
|
|
235
|
+
if not match:
|
|
236
|
+
raise AssemblyError("could not parse loudnorm measurement output")
|
|
237
|
+
measured = json.loads(match.group(0))
|
|
238
|
+
return {
|
|
239
|
+
"measured_I": measured["input_i"],
|
|
240
|
+
"measured_TP": measured["input_tp"],
|
|
241
|
+
"measured_LRA": measured["input_lra"],
|
|
242
|
+
"measured_thresh": measured["input_thresh"],
|
|
243
|
+
"offset": measured["target_offset"],
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _loudnorm_applied(measured: dict[str, str]) -> str:
|
|
248
|
+
return (
|
|
249
|
+
f"loudnorm={_LOUDNORM}:"
|
|
250
|
+
+ ":".join(f"{k}={v}" for k, v in measured.items())
|
|
251
|
+
+ ":linear=true:print_format=summary"
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
# ---------------------------------------------------------------------------
|
|
256
|
+
# Per-chapter pipeline stages (run in parallel across chapters)
|
|
257
|
+
# ---------------------------------------------------------------------------
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _build_chapter(
|
|
261
|
+
chapter_index: int,
|
|
262
|
+
entries: list[tuple[str, str]], # (chunk_id, boundary_after value)
|
|
263
|
+
cache_dir: Path,
|
|
264
|
+
chapter_dir: Path,
|
|
265
|
+
pause_seconds: dict[str, float], # boundary value -> seconds
|
|
266
|
+
target_rate: int,
|
|
267
|
+
) -> tuple[int, Path]:
|
|
268
|
+
"""Trim one chapter's chunks, interleave boundary silences, write FLAC."""
|
|
269
|
+
parts: list[np.ndarray] = []
|
|
270
|
+
for chunk_id, boundary in entries:
|
|
271
|
+
wav, rate = sf.read(
|
|
272
|
+
cache_dir / f"{chunk_id}.wav", dtype="float32", always_2d=False
|
|
273
|
+
)
|
|
274
|
+
wav = np.asarray(wav, dtype=np.float32).reshape(-1)
|
|
275
|
+
wav = trim_silence(wav, rate)
|
|
276
|
+
if rate != target_rate:
|
|
277
|
+
raw = chapter_dir / f"{chunk_id}.raw.wav"
|
|
278
|
+
resampled = chapter_dir / f"{chunk_id}.rs.wav"
|
|
279
|
+
try:
|
|
280
|
+
sf.write(raw, wav, rate, format="WAV", subtype="FLOAT")
|
|
281
|
+
_run(
|
|
282
|
+
[
|
|
283
|
+
"ffmpeg",
|
|
284
|
+
"-hide_banner",
|
|
285
|
+
"-nostats",
|
|
286
|
+
"-y",
|
|
287
|
+
"-i",
|
|
288
|
+
str(raw),
|
|
289
|
+
"-af",
|
|
290
|
+
f"aresample={target_rate}",
|
|
291
|
+
"-ar",
|
|
292
|
+
str(target_rate),
|
|
293
|
+
str(resampled),
|
|
294
|
+
],
|
|
295
|
+
f"resample chunk {chunk_id}",
|
|
296
|
+
)
|
|
297
|
+
wav = sf.read(resampled, dtype="float32", always_2d=False)[0]
|
|
298
|
+
wav = np.asarray(wav, dtype=np.float32).reshape(-1)
|
|
299
|
+
finally:
|
|
300
|
+
raw.unlink(missing_ok=True)
|
|
301
|
+
resampled.unlink(missing_ok=True)
|
|
302
|
+
parts.append(wav)
|
|
303
|
+
silence_s = pause_seconds.get(boundary, 0.0)
|
|
304
|
+
if silence_s > 0:
|
|
305
|
+
parts.append(np.zeros(int(silence_s * target_rate), dtype=np.float32))
|
|
306
|
+
audio = np.concatenate(parts) if parts else np.zeros(target_rate, dtype=np.float32)
|
|
307
|
+
chapter_path = chapter_dir / f"chapter_{chapter_index:03d}.flac"
|
|
308
|
+
# 24-bit FLAC: lossless, matching the previous float32 -> s32 pipeline.
|
|
309
|
+
sf.write(chapter_path, audio, target_rate, format="FLAC", subtype="PCM_24")
|
|
310
|
+
return chapter_index, chapter_path
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _measure_encode_chapter(
|
|
314
|
+
chapter_index: int,
|
|
315
|
+
chapter_path: Path,
|
|
316
|
+
segment_path: Path,
|
|
317
|
+
codec: list[str],
|
|
318
|
+
) -> tuple[int, Path]:
|
|
319
|
+
"""Two-pass loudnorm measured and applied per chapter, then encode."""
|
|
320
|
+
measured = _loudnorm_measure(chapter_path)
|
|
321
|
+
_run(
|
|
322
|
+
[
|
|
323
|
+
"ffmpeg",
|
|
324
|
+
"-hide_banner",
|
|
325
|
+
"-nostats",
|
|
326
|
+
"-y",
|
|
327
|
+
"-i",
|
|
328
|
+
str(chapter_path.absolute()),
|
|
329
|
+
"-af",
|
|
330
|
+
_loudnorm_applied(measured),
|
|
331
|
+
*codec,
|
|
332
|
+
str(segment_path.absolute()),
|
|
333
|
+
],
|
|
334
|
+
f"encode chapter {chapter_index + 1}",
|
|
335
|
+
)
|
|
336
|
+
return chapter_index, segment_path
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _mux_extra(
|
|
340
|
+
meta_file: Path, cover: Path | None, fmt: str, rate: int | None = None
|
|
341
|
+
) -> list[str]:
|
|
342
|
+
"""Final-mux args placed after the concat input.
|
|
343
|
+
|
|
344
|
+
Input 0 is the concatenated encoded audio (stream-copied, no
|
|
345
|
+
re-encode); the optional cover image is input 1 (mapped as an
|
|
346
|
+
attached_pic video stream -> covr/APIC/PICTURE); the ffmetadata file
|
|
347
|
+
is the last input, so ``-map_metadata`` must track the input count.
|
|
348
|
+
``rate`` pins the flac re-encode's sample rate (book-modal).
|
|
349
|
+
"""
|
|
350
|
+
args: list[str] = []
|
|
351
|
+
if cover is not None:
|
|
352
|
+
args += ["-i", str(cover.absolute())]
|
|
353
|
+
args += ["-f", "ffmetadata", "-i", str(meta_file.absolute())]
|
|
354
|
+
meta_index = 2 if cover is not None else 1
|
|
355
|
+
args += ["-map", "0:a"]
|
|
356
|
+
if cover is not None:
|
|
357
|
+
cover_codec = (
|
|
358
|
+
"copy" if cover.suffix.lower() in (".png", ".jpg", ".jpeg") else "mjpeg"
|
|
359
|
+
)
|
|
360
|
+
args += ["-map", "1:v", "-c:v", cover_codec, "-disposition:v:0", "attached_pic"]
|
|
361
|
+
args += ["-map_metadata", str(meta_index)]
|
|
362
|
+
if fmt == "flac":
|
|
363
|
+
# Stream copy into a raw FLAC container writes a STREAMINFO header
|
|
364
|
+
# the demuxer trusts (total samples = first segment only), making
|
|
365
|
+
# everything past the first segment unreachable. Re-encode instead:
|
|
366
|
+
# lossless -> lossless is bit-exact, and FLAC encode is fast.
|
|
367
|
+
args += ["-c:a", "flac", "-compression_level", "5"]
|
|
368
|
+
if rate is not None:
|
|
369
|
+
args += ["-ar", str(rate)]
|
|
370
|
+
else:
|
|
371
|
+
args += ["-c:a", "copy"]
|
|
372
|
+
if fmt == "m4b":
|
|
373
|
+
args += ["-movflags", "+faststart"]
|
|
374
|
+
return args
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
#: (codec args, segment suffix) per output format; same encoders/bitrates
|
|
378
|
+
#: as the previous single-pass pipeline — only the segmentation changed.
|
|
379
|
+
def _codec_args(fmt: str, target_rate: int) -> tuple[list[str], str]:
|
|
380
|
+
if fmt == "m4b":
|
|
381
|
+
return ["-c:a", "aac", "-b:a", "128k", "-ar", "44100"], ".m4a"
|
|
382
|
+
if fmt == "mp3":
|
|
383
|
+
return ["-c:a", "libmp3lame", "-b:a", "192k", "-ar", "44100"], ".mp3"
|
|
384
|
+
if fmt == "flac":
|
|
385
|
+
return [
|
|
386
|
+
"-c:a",
|
|
387
|
+
"flac",
|
|
388
|
+
"-compression_level",
|
|
389
|
+
"5",
|
|
390
|
+
"-ar",
|
|
391
|
+
str(target_rate),
|
|
392
|
+
], ".flac"
|
|
393
|
+
raise AssemblyError(f"unsupported output format {fmt!r} (use .m4b, .mp3 or .flac)")
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _worker_count(n_tasks: int) -> int:
|
|
397
|
+
return max(1, min(n_tasks, os.cpu_count() or 4))
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
# ---------------------------------------------------------------------------
|
|
401
|
+
# Overlappable per-chapter pipeline
|
|
402
|
+
# ---------------------------------------------------------------------------
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
class ChapterPipeline:
|
|
406
|
+
"""Per-chapter assembly that can overlap synthesis.
|
|
407
|
+
|
|
408
|
+
``chunk_done`` is the ``on_chunk_complete`` hook of
|
|
409
|
+
:func:`recito.synthesize.synthesize_all`: as each chunk's audio
|
|
410
|
+
becomes final in the cache, the chapter's pending set shrinks, and
|
|
411
|
+
once a chapter's unique chunk ids are all cached, its build ->
|
|
412
|
+
loudnorm -> encode task is submitted to a thread pool — chapters
|
|
413
|
+
assemble while later chapters are still being synthesized.
|
|
414
|
+
``finalize`` waits for the stragglers and muxes the final file.
|
|
415
|
+
|
|
416
|
+
The output is identical to non-overlapped assembly: the gate is
|
|
417
|
+
cache completeness (writes are atomic), chapter intermediates are
|
|
418
|
+
rebuilt unconditionally on every run, and each chapter's output
|
|
419
|
+
depends only on the cache contents, the manifest, and the pauses.
|
|
420
|
+
|
|
421
|
+
Contract: ``chunk_done`` must be called serially, from one thread
|
|
422
|
+
(true for every ``synthesize_all`` completion path). Chapter task
|
|
423
|
+
failures surface at ``finalize``, not mid-synthesis: they are
|
|
424
|
+
deterministic and re-fail on resume, and letting synthesis finish
|
|
425
|
+
leaves a warm cache for that resume.
|
|
426
|
+
"""
|
|
427
|
+
|
|
428
|
+
def __init__(
|
|
429
|
+
self,
|
|
430
|
+
manifest: Manifest,
|
|
431
|
+
cache_dir: Path,
|
|
432
|
+
workdir: Path,
|
|
433
|
+
pauses: Pauses,
|
|
434
|
+
fmt: str,
|
|
435
|
+
) -> None:
|
|
436
|
+
_, self._seg_ext = _codec_args(fmt, 44100) # validates fmt early
|
|
437
|
+
self._fmt = fmt
|
|
438
|
+
self._manifest = manifest
|
|
439
|
+
self._cache_dir = cache_dir
|
|
440
|
+
self._workdir = workdir
|
|
441
|
+
self._chapter_dir = workdir / "chapters"
|
|
442
|
+
self._segment_dir = workdir / "segments"
|
|
443
|
+
self._chapter_dir.mkdir(parents=True, exist_ok=True)
|
|
444
|
+
self._segment_dir.mkdir(parents=True, exist_ok=True)
|
|
445
|
+
# Every Boundary value maps to a Pauses field of the same name, so
|
|
446
|
+
# a new boundary is wired automatically; a Boundary value without
|
|
447
|
+
# a matching Pauses field fails loudly here instead of silently
|
|
448
|
+
# assembling 0.0 s pauses.
|
|
449
|
+
self._pause_seconds = {b.value: getattr(pauses, b.value) for b in Boundary}
|
|
450
|
+
self._entries: dict[int, list[tuple[str, str]]] = {}
|
|
451
|
+
for chunk in manifest.chunks:
|
|
452
|
+
self._entries.setdefault(chunk.chapter_index, []).append(
|
|
453
|
+
(chunk.id, chunk.boundary_after.value)
|
|
454
|
+
)
|
|
455
|
+
# Duplicate chunk ids (identical text) share one cache entry and
|
|
456
|
+
# may span chapters; synthesis queues only the FIRST occurrence
|
|
457
|
+
# of an uncached id, so its completion callback carries just one
|
|
458
|
+
# chapter_index. Completion is therefore tracked per id: one
|
|
459
|
+
# callback clears the id in EVERY chapter that contains it.
|
|
460
|
+
self._pending: dict[int, set[str]] = {
|
|
461
|
+
ci: {cid for cid, _ in entries} for ci, entries in self._entries.items()
|
|
462
|
+
}
|
|
463
|
+
self._id_chapters: dict[str, set[int]] = {}
|
|
464
|
+
for ci, ids in self._pending.items():
|
|
465
|
+
for cid in ids:
|
|
466
|
+
self._id_chapters.setdefault(cid, set()).add(ci)
|
|
467
|
+
workers = _worker_count(len(self._entries))
|
|
468
|
+
self._pool = ThreadPoolExecutor(max_workers=workers)
|
|
469
|
+
self._futures: dict[int, Future] = {}
|
|
470
|
+
|
|
471
|
+
def __enter__(self) -> Self:
|
|
472
|
+
return self
|
|
473
|
+
|
|
474
|
+
def __exit__(self, exc_type: object, *_: object) -> None:
|
|
475
|
+
# On abort, cancel not-yet-started tasks but always wait for
|
|
476
|
+
# running ones: their ffmpeg/soundfile writes into the workdir
|
|
477
|
+
# must quiesce before the caller's workdir lock
|
|
478
|
+
# (main._locked_workdir) is released, or a promptly re-run
|
|
479
|
+
# `speak` could interleave writes with the orphaned tasks.
|
|
480
|
+
# Running tasks are chapter-sized ffmpeg passes (seconds each),
|
|
481
|
+
# so the abort delay stays bounded; chapter intermediates are
|
|
482
|
+
# rebuilt unconditionally on the resumed run.
|
|
483
|
+
self._pool.shutdown(wait=True, cancel_futures=exc_type is not None)
|
|
484
|
+
|
|
485
|
+
def chunk_done(self, chunk: Chunk) -> None:
|
|
486
|
+
"""Record one cached chunk; fire each chapter whose ids are all in."""
|
|
487
|
+
for ci in self._id_chapters.get(chunk.id, ()):
|
|
488
|
+
pending = self._pending[ci]
|
|
489
|
+
if chunk.id not in pending:
|
|
490
|
+
continue # already cleared by an earlier occurrence
|
|
491
|
+
pending.discard(chunk.id)
|
|
492
|
+
if not pending:
|
|
493
|
+
self._futures[ci] = self._pool.submit(
|
|
494
|
+
self._assemble_chapter, ci, self._entries[ci]
|
|
495
|
+
)
|
|
496
|
+
|
|
497
|
+
def _assemble_chapter(
|
|
498
|
+
self, chapter_index: int, entries: list[tuple[str, str]]
|
|
499
|
+
) -> int:
|
|
500
|
+
"""Build, loudnorm and encode one chapter. Returns its sample rate."""
|
|
501
|
+
try:
|
|
502
|
+
# Chapter-modal rate from headers (chunks are all cached by
|
|
503
|
+
# construction). Equals the book-modal rate whenever chapters
|
|
504
|
+
# are internally uniform — the only case seen in practice.
|
|
505
|
+
rates: dict[int, int] = {}
|
|
506
|
+
for chunk_id, _ in entries:
|
|
507
|
+
rate = sf.info(self._cache_dir / f"{chunk_id}.wav").samplerate
|
|
508
|
+
rates[rate] = rates.get(rate, 0) + 1
|
|
509
|
+
rate = max(rates, key=rates.get)
|
|
510
|
+
codec, _ = _codec_args(self._fmt, rate)
|
|
511
|
+
_, chapter_path = _build_chapter(
|
|
512
|
+
chapter_index,
|
|
513
|
+
entries,
|
|
514
|
+
self._cache_dir,
|
|
515
|
+
self._chapter_dir,
|
|
516
|
+
self._pause_seconds,
|
|
517
|
+
rate,
|
|
518
|
+
)
|
|
519
|
+
_measure_encode_chapter(
|
|
520
|
+
chapter_index,
|
|
521
|
+
chapter_path,
|
|
522
|
+
self._segment_dir / f"chapter_{chapter_index:03d}{self._seg_ext}",
|
|
523
|
+
codec,
|
|
524
|
+
)
|
|
525
|
+
return rate
|
|
526
|
+
except Exception as exc:
|
|
527
|
+
raise AssemblyError(
|
|
528
|
+
f"chapter {chapter_index + 1} assembly failed: {exc}"
|
|
529
|
+
) from exc
|
|
530
|
+
|
|
531
|
+
def finalize(self, output: Path, meta: BookMeta) -> AssemblyResult:
|
|
532
|
+
"""Wait for all chapter tasks, then mux the final audiobook."""
|
|
533
|
+
unfinished = sorted(set(self._pending) - set(self._futures))
|
|
534
|
+
if unfinished:
|
|
535
|
+
raise AssemblyError(
|
|
536
|
+
f"{len(unfinished)} chapter(s) lack completed audio; "
|
|
537
|
+
"run the synthesis stage first (re-run `speak` to resume)"
|
|
538
|
+
)
|
|
539
|
+
chapter_indexes = sorted(self._futures)
|
|
540
|
+
# In chapter order: the first task failure propagates.
|
|
541
|
+
chapter_rates = {ci: self._futures[ci].result() for ci in chapter_indexes}
|
|
542
|
+
segment_paths = [
|
|
543
|
+
self._segment_dir / f"chapter_{i:03d}{self._seg_ext}"
|
|
544
|
+
for i in chapter_indexes
|
|
545
|
+
]
|
|
546
|
+
|
|
547
|
+
# Chapter marks follow the encoded segments' real durations (which
|
|
548
|
+
# include per-segment encoder priming, so marks stay on the joins).
|
|
549
|
+
with ThreadPoolExecutor(
|
|
550
|
+
max_workers=_worker_count(len(chapter_indexes))
|
|
551
|
+
) as pool:
|
|
552
|
+
segment_durations = list(pool.map(_probe_duration, segment_paths))
|
|
553
|
+
chapters_meta = list(
|
|
554
|
+
zip(
|
|
555
|
+
(self._manifest.chapters[i] for i in chapter_indexes), segment_durations
|
|
556
|
+
)
|
|
557
|
+
)
|
|
558
|
+
|
|
559
|
+
meta_file = self._workdir / "ffmetadata.txt"
|
|
560
|
+
meta_file.write_text(_ffmetadata(meta, chapters_meta))
|
|
561
|
+
|
|
562
|
+
# Book-modal rate for the flac mux (chunk-weighted, matching the
|
|
563
|
+
# historical whole-manifest modal rate). Computed now — on an
|
|
564
|
+
# overlapped run the cache was incomplete at __init__ time.
|
|
565
|
+
tally: dict[int, int] = {}
|
|
566
|
+
for ci, rate in chapter_rates.items():
|
|
567
|
+
tally[rate] = tally.get(rate, 0) + len(self._entries[ci])
|
|
568
|
+
mux_rate = max(tally, key=tally.get)
|
|
569
|
+
|
|
570
|
+
# Final mux: stream copy, metadata + chapter marks + cover, faststart.
|
|
571
|
+
_concat_demuxer(
|
|
572
|
+
segment_paths,
|
|
573
|
+
output,
|
|
574
|
+
_mux_extra(meta_file, meta.cover, self._fmt, mux_rate),
|
|
575
|
+
f"mux {output.name}",
|
|
576
|
+
)
|
|
577
|
+
duration = _probe_duration(output)
|
|
578
|
+
return AssemblyResult(
|
|
579
|
+
output=output, duration_s=duration, chapters=chapters_meta
|
|
580
|
+
)
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
# ---------------------------------------------------------------------------
|
|
584
|
+
# Entry point
|
|
585
|
+
# ---------------------------------------------------------------------------
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
def assemble(
|
|
589
|
+
manifest: Manifest,
|
|
590
|
+
cache_dir: Path,
|
|
591
|
+
output: Path,
|
|
592
|
+
workdir: Path,
|
|
593
|
+
pauses: Pauses,
|
|
594
|
+
meta: BookMeta,
|
|
595
|
+
) -> AssemblyResult:
|
|
596
|
+
"""Assemble the final audiobook from cached chunk audio."""
|
|
597
|
+
require_ffmpeg()
|
|
598
|
+
missing = [c for c in manifest.chunks if not (cache_dir / f"{c.id}.wav").exists()]
|
|
599
|
+
if missing:
|
|
600
|
+
raise AssemblyError(
|
|
601
|
+
f"{len(missing)} of {len(manifest.chunks)} chunks lack audio; "
|
|
602
|
+
"run the synthesis stage first (re-run `speak` to resume)"
|
|
603
|
+
)
|
|
604
|
+
fmt = output.suffix.lstrip(".").lower()
|
|
605
|
+
with ChapterPipeline(manifest, cache_dir, workdir, pauses, fmt) as pipe:
|
|
606
|
+
for chunk in manifest.chunks:
|
|
607
|
+
pipe.chunk_done(chunk)
|
|
608
|
+
return pipe.finalize(output, meta)
|