recito 0.7.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. recito/__init__.py +5 -0
  2. recito/_version.py +24 -0
  3. recito/assemble.py +608 -0
  4. recito/chunk.py +119 -0
  5. recito/clean.py +2497 -0
  6. recito/config.py +282 -0
  7. recito/cover.py +71 -0
  8. recito/engines/__init__.py +31 -0
  9. recito/engines/base.py +245 -0
  10. recito/engines/qwen3.py +194 -0
  11. recito/engines/tone.py +50 -0
  12. recito/engines/vllm_omni.py +435 -0
  13. recito/extract.py +71 -0
  14. recito/extract_epub.py +799 -0
  15. recito/extract_pdf.py +1989 -0
  16. recito/main.py +661 -0
  17. recito/models.py +463 -0
  18. recito/synthesize.py +3710 -0
  19. recito/textnorm/__init__.py +243 -0
  20. recito/textnorm/backends/__init__.py +1 -0
  21. recito/textnorm/backends/basic.py +50 -0
  22. recito/textnorm/backends/own_en.py +1130 -0
  23. recito/textnorm/data/__init__.py +63 -0
  24. recito/textnorm/data/en/acronyms.tsv +39 -0
  25. recito/textnorm/data/en/common.tsv +286607 -0
  26. recito/textnorm/data/en/currency.tsv +6 -0
  27. recito/textnorm/data/en/glue.tsv +11 -0
  28. recito/textnorm/data/en/hotkeys.tsv +35 -0
  29. recito/textnorm/data/en/months.tsv +18 -0
  30. recito/textnorm/data/en/seconds_context.tsv +11 -0
  31. recito/textnorm/data/en/symbols.tsv +11 -0
  32. recito/textnorm/data/en/units.tsv +40 -0
  33. recito/textnorm/data/en/whitelist.tsv +58 -0
  34. recito/textnorm/data/en/year_context.tsv +23 -0
  35. recito/textnorm/overlay.py +116 -0
  36. recito/textnorm/prepare.py +879 -0
  37. recito/textnorm/profile.py +795 -0
  38. recito/textnorm/rules_en.py +46 -0
  39. recito/textnorm/segment.py +265 -0
  40. recito/textnorm/urls.py +366 -0
  41. recito/textnorm/verbalize.py +89 -0
  42. recito-0.7.2.dist-info/METADATA +123 -0
  43. recito-0.7.2.dist-info/RECORD +47 -0
  44. recito-0.7.2.dist-info/WHEEL +5 -0
  45. recito-0.7.2.dist-info/entry_points.txt +2 -0
  46. recito-0.7.2.dist-info/licenses/LICENSE.txt +661 -0
  47. recito-0.7.2.dist-info/top_level.txt +1 -0
recito/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ # __init__.py
2
+
3
+ from recito._version import __version__, __version_tuple__
4
+
5
+ __all__ = ["__version__", "__version_tuple__"]
recito/_version.py ADDED
@@ -0,0 +1,24 @@
1
+ # file generated by vcs-versioning
2
+ # don't change, don't track in version control
3
+ from __future__ import annotations
4
+
5
+ __all__ = [
6
+ "__version__",
7
+ "__version_tuple__",
8
+ "version",
9
+ "version_tuple",
10
+ "__commit_id__",
11
+ "commit_id",
12
+ ]
13
+
14
+ version: str
15
+ __version__: str
16
+ __version_tuple__: tuple[int | str, ...]
17
+ version_tuple: tuple[int | str, ...]
18
+ commit_id: str | None
19
+ __commit_id__: str | None
20
+
21
+ __version__ = version = '0.7.2'
22
+ __version_tuple__ = version_tuple = (0, 7, 2)
23
+
24
+ __commit_id__ = commit_id = None
recito/assemble.py ADDED
@@ -0,0 +1,608 @@
1
+ """Stage 5 — ASSEMBLE: cached chunk wavs -> final audiobook file.
2
+
3
+ Per chunk: leading/trailing silence is trimmed (energy threshold relative
4
+ to peak, tts-audiobook-tool pattern, MIT) so chunk joins aren't choppy.
5
+ Per chapter: chunks and boundary silences are concatenated in memory and
6
+ written as FLAC intermediates, one per chapter, built in parallel.
7
+
8
+ Every ffmpeg audio codec/filter used here is single-threaded, so wall
9
+ time is governed by the longest serial pass over the audio. All passes
10
+ are therefore kept chapter-sized and run in parallel: each chapter gets
11
+ a two-pass EBU R128 loudnorm (I=-14 / TP=-3 / LRA=7, linear mode
12
+ honoring the true-peak ceiling) measured and applied per chapter, then
13
+ is encoded to the target codec. Chapter-level normalization is audibly
14
+ identical to book-wide for TTS narration (chapter loudness varies by
15
+ ~1 LU) while keeping chapter-to-chapter levels consistent. The encoded
16
+ chapter segments are concatenated with stream copy (no re-encode) and
17
+ muxed with ffmetadata chapter marks; AAC encoder priming at each join
18
+ (~50 ms) lands inside the boundary silence.
19
+
20
+ :class:`ChapterPipeline` drives the per-chapter work; it can run after
21
+ synthesis (:func:`assemble`) or overlapped with it, firing each
22
+ chapter's task as soon as that chapter's audio is final in the chunk
23
+ cache (the output is identical either way).
24
+
25
+ Metadata is written as classic iTunes-style ilst atoms
26
+ (``mdta``-style ``use_metadata_tags`` is deliberately avoided: most
27
+ players — Cozy, VLC, mutagen-based tools — cannot read it), and a cover
28
+ image is embedded as an attached_pic stream (covr/APIC/PICTURE). mp3 and
29
+ flac outputs are also supported (no chapter marks — container
30
+ limitation).
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ import json
36
+ import os
37
+ import re
38
+ import shutil
39
+ import subprocess
40
+ from concurrent.futures import Future, ThreadPoolExecutor
41
+ from dataclasses import dataclass
42
+ from pathlib import Path
43
+ from typing import Self
44
+
45
+ import numpy as np
46
+ import soundfile as sf
47
+
48
+ from recito.config import Pauses
49
+ from recito.models import BookMeta, Boundary, Chunk, Manifest
50
+
51
+ #: Silence-trim threshold relative to each chunk's peak RMS (dB).
52
+ _TRIM_THRESHOLD_DB = -42.0
53
+ #: Audio retained before/after the detected speech (seconds).
54
+ _TRIM_PAD_S = 0.04
55
+
56
+ _LOUDNORM = "I=-14:TP=-3:LRA=7"
57
+
58
+
59
+ class AssemblyError(Exception):
60
+ """Missing audio, ffmpeg failure, or unsupported combination."""
61
+
62
+
63
+ @dataclass
64
+ class AssemblyResult:
65
+ output: Path
66
+ duration_s: float
67
+ chapters: list[tuple[str, float]] # (title, duration seconds)
68
+
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # ffmpeg/ffprobe helpers
72
+ # ---------------------------------------------------------------------------
73
+
74
+
75
+ def require_ffmpeg() -> None:
76
+ for tool in ("ffmpeg", "ffprobe"):
77
+ if shutil.which(tool) is None:
78
+ raise AssemblyError(
79
+ f"{tool} not found on PATH; install ffmpeg (see README)"
80
+ )
81
+
82
+
83
+ def _run(cmd: list[str], what: str) -> subprocess.CompletedProcess:
84
+ # stdin must not be the tty: ffmpeg's interactive keyboard mode would
85
+ # snapshot/restore termios per process, and with concurrent chapter
86
+ # tasks (or a child orphaned by an abort) the last restore can be a
87
+ # stale no-echo snapshot — leaving the terminal echo off after the run.
88
+ proc = subprocess.run(cmd, capture_output=True, text=True, stdin=subprocess.DEVNULL)
89
+ if proc.returncode != 0:
90
+ tail = "\n".join(proc.stderr.strip().splitlines()[-12:])
91
+ raise AssemblyError(f"{what} failed (exit {proc.returncode}):\n{tail}")
92
+ return proc
93
+
94
+
95
+ def _probe_duration(path: Path) -> float:
96
+ proc = _run(
97
+ [
98
+ "ffprobe",
99
+ "-v",
100
+ "error",
101
+ "-show_entries",
102
+ "format=duration",
103
+ "-of",
104
+ "default=noprint_wrappers=1:nokey=1",
105
+ str(path),
106
+ ],
107
+ f"ffprobe {path.name}",
108
+ )
109
+ return float(proc.stdout.strip())
110
+
111
+
112
+ def _concat_demuxer(
113
+ inputs: list[Path], output: Path, extra: list[str], what: str
114
+ ) -> None:
115
+ # Absolute paths: the demuxer resolves relative entries against the
116
+ # list file's directory, not the process cwd.
117
+ list_file = output.with_suffix(".concat.txt").absolute()
118
+ # ffmpeg concat-demuxer quoting: wrap in single quotes, escape inner ones.
119
+ body = "\n".join(
120
+ f"file '{str(p.absolute()).replace(chr(39), chr(39) + chr(92) + chr(39) + chr(39))}'"
121
+ for p in inputs
122
+ )
123
+ list_file.write_text(body + "\n")
124
+ try:
125
+ _run(
126
+ [
127
+ "ffmpeg",
128
+ "-hide_banner",
129
+ "-nostats",
130
+ "-y",
131
+ "-f",
132
+ "concat",
133
+ "-safe",
134
+ "0",
135
+ "-i",
136
+ str(list_file),
137
+ *extra,
138
+ str(output.absolute()),
139
+ ],
140
+ what,
141
+ )
142
+ finally:
143
+ list_file.unlink(missing_ok=True)
144
+
145
+
146
+ # ---------------------------------------------------------------------------
147
+ # Silence trimming
148
+ # ---------------------------------------------------------------------------
149
+
150
+
151
+ def trim_silence(wav: np.ndarray, sample_rate: int) -> np.ndarray:
152
+ """Trim leading/trailing silence using peak-relative frame RMS."""
153
+ if wav.ndim != 1 or len(wav) == 0:
154
+ return wav
155
+ frame = max(1, int(0.025 * sample_rate))
156
+ n_frames = len(wav) // frame
157
+ if n_frames < 3:
158
+ return wav
159
+ frames = wav[: n_frames * frame].reshape(n_frames, frame)
160
+ rms = np.sqrt(np.mean(frames * frames, axis=1))
161
+ peak = float(rms.max())
162
+ if peak <= 0:
163
+ return wav
164
+ threshold = peak * (10.0 ** (_TRIM_THRESHOLD_DB / 20.0))
165
+ active = np.flatnonzero(rms >= threshold)
166
+ if len(active) == 0:
167
+ return wav
168
+ pad = int(_TRIM_PAD_S * sample_rate)
169
+ start = max(0, active[0] * frame - pad)
170
+ end = min(len(wav), (active[-1] + 1) * frame + pad)
171
+ return wav[start:end]
172
+
173
+
174
+ # ---------------------------------------------------------------------------
175
+ # ffmetadata
176
+ # ---------------------------------------------------------------------------
177
+
178
+
179
+ def _escape_meta(value: str) -> str:
180
+ return re.sub(r"([=;#\\\n])", lambda m: "\\" + m.group(1), value)
181
+
182
+
183
+ def _ffmetadata(meta: BookMeta, chapters: list[tuple[str, float]]) -> str:
184
+ out = [";FFMETADATA1"]
185
+ # iTunes/ilst-style keys; ffmpeg maps them to the classic atoms
186
+ # (composer = author, artist = narrator: Cozy's author/reader fields).
187
+ tags = {
188
+ "title": meta.title,
189
+ "album": meta.title, # book-grouping key in players (e.g. Cozy)
190
+ "artist": meta.narrator,
191
+ "composer": meta.author,
192
+ "album_artist": meta.author,
193
+ "date": meta.date,
194
+ "genre": meta.genre,
195
+ "description": meta.description,
196
+ }
197
+ for key, value in tags.items():
198
+ value = value.strip()
199
+ if value:
200
+ out.append(f"{key}={_escape_meta(value)}")
201
+ start_ms = 0
202
+ for chapter_title, duration in chapters:
203
+ end_ms = start_ms + round(duration * 1000)
204
+ out.append("[CHAPTER]")
205
+ out.append("TIMEBASE=1/1000")
206
+ out.append(f"START={start_ms}")
207
+ out.append(f"END={end_ms}")
208
+ out.append(f"title={_escape_meta(chapter_title)}")
209
+ start_ms = end_ms
210
+ return "\n".join(out) + "\n"
211
+
212
+
213
+ # ---------------------------------------------------------------------------
214
+ # Loudnorm (two-pass, per chapter)
215
+ # ---------------------------------------------------------------------------
216
+
217
+
218
+ def _loudnorm_measure(source: Path) -> dict[str, str]:
219
+ proc = _run(
220
+ [
221
+ "ffmpeg",
222
+ "-hide_banner",
223
+ "-nostats",
224
+ "-i",
225
+ str(source),
226
+ "-af",
227
+ f"loudnorm={_LOUDNORM}:print_format=json",
228
+ "-f",
229
+ "null",
230
+ "-",
231
+ ],
232
+ "loudnorm measurement pass",
233
+ )
234
+ match = re.search(r"\{[^{}]*\"input_i\"[^{}]*\}", proc.stderr, re.DOTALL)
235
+ if not match:
236
+ raise AssemblyError("could not parse loudnorm measurement output")
237
+ measured = json.loads(match.group(0))
238
+ return {
239
+ "measured_I": measured["input_i"],
240
+ "measured_TP": measured["input_tp"],
241
+ "measured_LRA": measured["input_lra"],
242
+ "measured_thresh": measured["input_thresh"],
243
+ "offset": measured["target_offset"],
244
+ }
245
+
246
+
247
+ def _loudnorm_applied(measured: dict[str, str]) -> str:
248
+ return (
249
+ f"loudnorm={_LOUDNORM}:"
250
+ + ":".join(f"{k}={v}" for k, v in measured.items())
251
+ + ":linear=true:print_format=summary"
252
+ )
253
+
254
+
255
+ # ---------------------------------------------------------------------------
256
+ # Per-chapter pipeline stages (run in parallel across chapters)
257
+ # ---------------------------------------------------------------------------
258
+
259
+
260
+ def _build_chapter(
261
+ chapter_index: int,
262
+ entries: list[tuple[str, str]], # (chunk_id, boundary_after value)
263
+ cache_dir: Path,
264
+ chapter_dir: Path,
265
+ pause_seconds: dict[str, float], # boundary value -> seconds
266
+ target_rate: int,
267
+ ) -> tuple[int, Path]:
268
+ """Trim one chapter's chunks, interleave boundary silences, write FLAC."""
269
+ parts: list[np.ndarray] = []
270
+ for chunk_id, boundary in entries:
271
+ wav, rate = sf.read(
272
+ cache_dir / f"{chunk_id}.wav", dtype="float32", always_2d=False
273
+ )
274
+ wav = np.asarray(wav, dtype=np.float32).reshape(-1)
275
+ wav = trim_silence(wav, rate)
276
+ if rate != target_rate:
277
+ raw = chapter_dir / f"{chunk_id}.raw.wav"
278
+ resampled = chapter_dir / f"{chunk_id}.rs.wav"
279
+ try:
280
+ sf.write(raw, wav, rate, format="WAV", subtype="FLOAT")
281
+ _run(
282
+ [
283
+ "ffmpeg",
284
+ "-hide_banner",
285
+ "-nostats",
286
+ "-y",
287
+ "-i",
288
+ str(raw),
289
+ "-af",
290
+ f"aresample={target_rate}",
291
+ "-ar",
292
+ str(target_rate),
293
+ str(resampled),
294
+ ],
295
+ f"resample chunk {chunk_id}",
296
+ )
297
+ wav = sf.read(resampled, dtype="float32", always_2d=False)[0]
298
+ wav = np.asarray(wav, dtype=np.float32).reshape(-1)
299
+ finally:
300
+ raw.unlink(missing_ok=True)
301
+ resampled.unlink(missing_ok=True)
302
+ parts.append(wav)
303
+ silence_s = pause_seconds.get(boundary, 0.0)
304
+ if silence_s > 0:
305
+ parts.append(np.zeros(int(silence_s * target_rate), dtype=np.float32))
306
+ audio = np.concatenate(parts) if parts else np.zeros(target_rate, dtype=np.float32)
307
+ chapter_path = chapter_dir / f"chapter_{chapter_index:03d}.flac"
308
+ # 24-bit FLAC: lossless, matching the previous float32 -> s32 pipeline.
309
+ sf.write(chapter_path, audio, target_rate, format="FLAC", subtype="PCM_24")
310
+ return chapter_index, chapter_path
311
+
312
+
313
+ def _measure_encode_chapter(
314
+ chapter_index: int,
315
+ chapter_path: Path,
316
+ segment_path: Path,
317
+ codec: list[str],
318
+ ) -> tuple[int, Path]:
319
+ """Two-pass loudnorm measured and applied per chapter, then encode."""
320
+ measured = _loudnorm_measure(chapter_path)
321
+ _run(
322
+ [
323
+ "ffmpeg",
324
+ "-hide_banner",
325
+ "-nostats",
326
+ "-y",
327
+ "-i",
328
+ str(chapter_path.absolute()),
329
+ "-af",
330
+ _loudnorm_applied(measured),
331
+ *codec,
332
+ str(segment_path.absolute()),
333
+ ],
334
+ f"encode chapter {chapter_index + 1}",
335
+ )
336
+ return chapter_index, segment_path
337
+
338
+
339
+ def _mux_extra(
340
+ meta_file: Path, cover: Path | None, fmt: str, rate: int | None = None
341
+ ) -> list[str]:
342
+ """Final-mux args placed after the concat input.
343
+
344
+ Input 0 is the concatenated encoded audio (stream-copied, no
345
+ re-encode); the optional cover image is input 1 (mapped as an
346
+ attached_pic video stream -> covr/APIC/PICTURE); the ffmetadata file
347
+ is the last input, so ``-map_metadata`` must track the input count.
348
+ ``rate`` pins the flac re-encode's sample rate (book-modal).
349
+ """
350
+ args: list[str] = []
351
+ if cover is not None:
352
+ args += ["-i", str(cover.absolute())]
353
+ args += ["-f", "ffmetadata", "-i", str(meta_file.absolute())]
354
+ meta_index = 2 if cover is not None else 1
355
+ args += ["-map", "0:a"]
356
+ if cover is not None:
357
+ cover_codec = (
358
+ "copy" if cover.suffix.lower() in (".png", ".jpg", ".jpeg") else "mjpeg"
359
+ )
360
+ args += ["-map", "1:v", "-c:v", cover_codec, "-disposition:v:0", "attached_pic"]
361
+ args += ["-map_metadata", str(meta_index)]
362
+ if fmt == "flac":
363
+ # Stream copy into a raw FLAC container writes a STREAMINFO header
364
+ # the demuxer trusts (total samples = first segment only), making
365
+ # everything past the first segment unreachable. Re-encode instead:
366
+ # lossless -> lossless is bit-exact, and FLAC encode is fast.
367
+ args += ["-c:a", "flac", "-compression_level", "5"]
368
+ if rate is not None:
369
+ args += ["-ar", str(rate)]
370
+ else:
371
+ args += ["-c:a", "copy"]
372
+ if fmt == "m4b":
373
+ args += ["-movflags", "+faststart"]
374
+ return args
375
+
376
+
377
+ #: (codec args, segment suffix) per output format; same encoders/bitrates
378
+ #: as the previous single-pass pipeline — only the segmentation changed.
379
+ def _codec_args(fmt: str, target_rate: int) -> tuple[list[str], str]:
380
+ if fmt == "m4b":
381
+ return ["-c:a", "aac", "-b:a", "128k", "-ar", "44100"], ".m4a"
382
+ if fmt == "mp3":
383
+ return ["-c:a", "libmp3lame", "-b:a", "192k", "-ar", "44100"], ".mp3"
384
+ if fmt == "flac":
385
+ return [
386
+ "-c:a",
387
+ "flac",
388
+ "-compression_level",
389
+ "5",
390
+ "-ar",
391
+ str(target_rate),
392
+ ], ".flac"
393
+ raise AssemblyError(f"unsupported output format {fmt!r} (use .m4b, .mp3 or .flac)")
394
+
395
+
396
+ def _worker_count(n_tasks: int) -> int:
397
+ return max(1, min(n_tasks, os.cpu_count() or 4))
398
+
399
+
400
+ # ---------------------------------------------------------------------------
401
+ # Overlappable per-chapter pipeline
402
+ # ---------------------------------------------------------------------------
403
+
404
+
405
+ class ChapterPipeline:
406
+ """Per-chapter assembly that can overlap synthesis.
407
+
408
+ ``chunk_done`` is the ``on_chunk_complete`` hook of
409
+ :func:`recito.synthesize.synthesize_all`: as each chunk's audio
410
+ becomes final in the cache, the chapter's pending set shrinks, and
411
+ once a chapter's unique chunk ids are all cached, its build ->
412
+ loudnorm -> encode task is submitted to a thread pool — chapters
413
+ assemble while later chapters are still being synthesized.
414
+ ``finalize`` waits for the stragglers and muxes the final file.
415
+
416
+ The output is identical to non-overlapped assembly: the gate is
417
+ cache completeness (writes are atomic), chapter intermediates are
418
+ rebuilt unconditionally on every run, and each chapter's output
419
+ depends only on the cache contents, the manifest, and the pauses.
420
+
421
+ Contract: ``chunk_done`` must be called serially, from one thread
422
+ (true for every ``synthesize_all`` completion path). Chapter task
423
+ failures surface at ``finalize``, not mid-synthesis: they are
424
+ deterministic and re-fail on resume, and letting synthesis finish
425
+ leaves a warm cache for that resume.
426
+ """
427
+
428
+ def __init__(
429
+ self,
430
+ manifest: Manifest,
431
+ cache_dir: Path,
432
+ workdir: Path,
433
+ pauses: Pauses,
434
+ fmt: str,
435
+ ) -> None:
436
+ _, self._seg_ext = _codec_args(fmt, 44100) # validates fmt early
437
+ self._fmt = fmt
438
+ self._manifest = manifest
439
+ self._cache_dir = cache_dir
440
+ self._workdir = workdir
441
+ self._chapter_dir = workdir / "chapters"
442
+ self._segment_dir = workdir / "segments"
443
+ self._chapter_dir.mkdir(parents=True, exist_ok=True)
444
+ self._segment_dir.mkdir(parents=True, exist_ok=True)
445
+ # Every Boundary value maps to a Pauses field of the same name, so
446
+ # a new boundary is wired automatically; a Boundary value without
447
+ # a matching Pauses field fails loudly here instead of silently
448
+ # assembling 0.0 s pauses.
449
+ self._pause_seconds = {b.value: getattr(pauses, b.value) for b in Boundary}
450
+ self._entries: dict[int, list[tuple[str, str]]] = {}
451
+ for chunk in manifest.chunks:
452
+ self._entries.setdefault(chunk.chapter_index, []).append(
453
+ (chunk.id, chunk.boundary_after.value)
454
+ )
455
+ # Duplicate chunk ids (identical text) share one cache entry and
456
+ # may span chapters; synthesis queues only the FIRST occurrence
457
+ # of an uncached id, so its completion callback carries just one
458
+ # chapter_index. Completion is therefore tracked per id: one
459
+ # callback clears the id in EVERY chapter that contains it.
460
+ self._pending: dict[int, set[str]] = {
461
+ ci: {cid for cid, _ in entries} for ci, entries in self._entries.items()
462
+ }
463
+ self._id_chapters: dict[str, set[int]] = {}
464
+ for ci, ids in self._pending.items():
465
+ for cid in ids:
466
+ self._id_chapters.setdefault(cid, set()).add(ci)
467
+ workers = _worker_count(len(self._entries))
468
+ self._pool = ThreadPoolExecutor(max_workers=workers)
469
+ self._futures: dict[int, Future] = {}
470
+
471
+ def __enter__(self) -> Self:
472
+ return self
473
+
474
+ def __exit__(self, exc_type: object, *_: object) -> None:
475
+ # On abort, cancel not-yet-started tasks but always wait for
476
+ # running ones: their ffmpeg/soundfile writes into the workdir
477
+ # must quiesce before the caller's workdir lock
478
+ # (main._locked_workdir) is released, or a promptly re-run
479
+ # `speak` could interleave writes with the orphaned tasks.
480
+ # Running tasks are chapter-sized ffmpeg passes (seconds each),
481
+ # so the abort delay stays bounded; chapter intermediates are
482
+ # rebuilt unconditionally on the resumed run.
483
+ self._pool.shutdown(wait=True, cancel_futures=exc_type is not None)
484
+
485
+ def chunk_done(self, chunk: Chunk) -> None:
486
+ """Record one cached chunk; fire each chapter whose ids are all in."""
487
+ for ci in self._id_chapters.get(chunk.id, ()):
488
+ pending = self._pending[ci]
489
+ if chunk.id not in pending:
490
+ continue # already cleared by an earlier occurrence
491
+ pending.discard(chunk.id)
492
+ if not pending:
493
+ self._futures[ci] = self._pool.submit(
494
+ self._assemble_chapter, ci, self._entries[ci]
495
+ )
496
+
497
+ def _assemble_chapter(
498
+ self, chapter_index: int, entries: list[tuple[str, str]]
499
+ ) -> int:
500
+ """Build, loudnorm and encode one chapter. Returns its sample rate."""
501
+ try:
502
+ # Chapter-modal rate from headers (chunks are all cached by
503
+ # construction). Equals the book-modal rate whenever chapters
504
+ # are internally uniform — the only case seen in practice.
505
+ rates: dict[int, int] = {}
506
+ for chunk_id, _ in entries:
507
+ rate = sf.info(self._cache_dir / f"{chunk_id}.wav").samplerate
508
+ rates[rate] = rates.get(rate, 0) + 1
509
+ rate = max(rates, key=rates.get)
510
+ codec, _ = _codec_args(self._fmt, rate)
511
+ _, chapter_path = _build_chapter(
512
+ chapter_index,
513
+ entries,
514
+ self._cache_dir,
515
+ self._chapter_dir,
516
+ self._pause_seconds,
517
+ rate,
518
+ )
519
+ _measure_encode_chapter(
520
+ chapter_index,
521
+ chapter_path,
522
+ self._segment_dir / f"chapter_{chapter_index:03d}{self._seg_ext}",
523
+ codec,
524
+ )
525
+ return rate
526
+ except Exception as exc:
527
+ raise AssemblyError(
528
+ f"chapter {chapter_index + 1} assembly failed: {exc}"
529
+ ) from exc
530
+
531
+ def finalize(self, output: Path, meta: BookMeta) -> AssemblyResult:
532
+ """Wait for all chapter tasks, then mux the final audiobook."""
533
+ unfinished = sorted(set(self._pending) - set(self._futures))
534
+ if unfinished:
535
+ raise AssemblyError(
536
+ f"{len(unfinished)} chapter(s) lack completed audio; "
537
+ "run the synthesis stage first (re-run `speak` to resume)"
538
+ )
539
+ chapter_indexes = sorted(self._futures)
540
+ # In chapter order: the first task failure propagates.
541
+ chapter_rates = {ci: self._futures[ci].result() for ci in chapter_indexes}
542
+ segment_paths = [
543
+ self._segment_dir / f"chapter_{i:03d}{self._seg_ext}"
544
+ for i in chapter_indexes
545
+ ]
546
+
547
+ # Chapter marks follow the encoded segments' real durations (which
548
+ # include per-segment encoder priming, so marks stay on the joins).
549
+ with ThreadPoolExecutor(
550
+ max_workers=_worker_count(len(chapter_indexes))
551
+ ) as pool:
552
+ segment_durations = list(pool.map(_probe_duration, segment_paths))
553
+ chapters_meta = list(
554
+ zip(
555
+ (self._manifest.chapters[i] for i in chapter_indexes), segment_durations
556
+ )
557
+ )
558
+
559
+ meta_file = self._workdir / "ffmetadata.txt"
560
+ meta_file.write_text(_ffmetadata(meta, chapters_meta))
561
+
562
+ # Book-modal rate for the flac mux (chunk-weighted, matching the
563
+ # historical whole-manifest modal rate). Computed now — on an
564
+ # overlapped run the cache was incomplete at __init__ time.
565
+ tally: dict[int, int] = {}
566
+ for ci, rate in chapter_rates.items():
567
+ tally[rate] = tally.get(rate, 0) + len(self._entries[ci])
568
+ mux_rate = max(tally, key=tally.get)
569
+
570
+ # Final mux: stream copy, metadata + chapter marks + cover, faststart.
571
+ _concat_demuxer(
572
+ segment_paths,
573
+ output,
574
+ _mux_extra(meta_file, meta.cover, self._fmt, mux_rate),
575
+ f"mux {output.name}",
576
+ )
577
+ duration = _probe_duration(output)
578
+ return AssemblyResult(
579
+ output=output, duration_s=duration, chapters=chapters_meta
580
+ )
581
+
582
+
583
+ # ---------------------------------------------------------------------------
584
+ # Entry point
585
+ # ---------------------------------------------------------------------------
586
+
587
+
588
+ def assemble(
589
+ manifest: Manifest,
590
+ cache_dir: Path,
591
+ output: Path,
592
+ workdir: Path,
593
+ pauses: Pauses,
594
+ meta: BookMeta,
595
+ ) -> AssemblyResult:
596
+ """Assemble the final audiobook from cached chunk audio."""
597
+ require_ffmpeg()
598
+ missing = [c for c in manifest.chunks if not (cache_dir / f"{c.id}.wav").exists()]
599
+ if missing:
600
+ raise AssemblyError(
601
+ f"{len(missing)} of {len(manifest.chunks)} chunks lack audio; "
602
+ "run the synthesis stage first (re-run `speak` to resume)"
603
+ )
604
+ fmt = output.suffix.lstrip(".").lower()
605
+ with ChapterPipeline(manifest, cache_dir, workdir, pauses, fmt) as pipe:
606
+ for chunk in manifest.chunks:
607
+ pipe.chunk_done(chunk)
608
+ return pipe.finalize(output, meta)