recito 0.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. recito-0.7.2/.env.example +35 -0
  2. recito-0.7.2/.gitignore +97 -0
  3. recito-0.7.2/.python-version +1 -0
  4. recito-0.7.2/AGENTS.md +122 -0
  5. recito-0.7.2/CHANGELOG.txt +24 -0
  6. recito-0.7.2/LICENSE.txt +661 -0
  7. recito-0.7.2/PKG-INFO +123 -0
  8. recito-0.7.2/README.md +79 -0
  9. recito-0.7.2/docs/ALTERNATIVES.md +407 -0
  10. recito-0.7.2/docs/DEVELOP.md +3244 -0
  11. recito-0.7.2/docs/FUTURE.md +323 -0
  12. recito-0.7.2/docs/INSTALL.md +53 -0
  13. recito-0.7.2/docs/ISSUES.md +39 -0
  14. recito-0.7.2/docs/LIBRE-EBOOKS-EVALUATION.md +570 -0
  15. recito-0.7.2/docs/LIBRE-EBOOKS.md +100 -0
  16. recito-0.7.2/docs/REJECT.md +50 -0
  17. recito-0.7.2/docs/SGLANG.md +169 -0
  18. recito-0.7.2/docs/TTS.md +339 -0
  19. recito-0.7.2/docs/USAGE.md +325 -0
  20. recito-0.7.2/pyproject.toml +93 -0
  21. recito-0.7.2/setup.cfg +4 -0
  22. recito-0.7.2/sglang/README.md +55 -0
  23. recito-0.7.2/sglang/qwen3_tts_base.yaml +6 -0
  24. recito-0.7.2/sglang/qwen3_tts_customvoice.yaml +6 -0
  25. recito-0.7.2/sglang/sglang-omni-qwen3-tts-base.service +30 -0
  26. recito-0.7.2/sglang/sglang-omni-qwen3-tts.service +30 -0
  27. recito-0.7.2/src/recito/__init__.py +5 -0
  28. recito-0.7.2/src/recito/_version.py +24 -0
  29. recito-0.7.2/src/recito/assemble.py +608 -0
  30. recito-0.7.2/src/recito/chunk.py +119 -0
  31. recito-0.7.2/src/recito/clean.py +2497 -0
  32. recito-0.7.2/src/recito/config.py +282 -0
  33. recito-0.7.2/src/recito/cover.py +71 -0
  34. recito-0.7.2/src/recito/engines/__init__.py +31 -0
  35. recito-0.7.2/src/recito/engines/base.py +245 -0
  36. recito-0.7.2/src/recito/engines/qwen3.py +194 -0
  37. recito-0.7.2/src/recito/engines/tone.py +50 -0
  38. recito-0.7.2/src/recito/engines/vllm_omni.py +435 -0
  39. recito-0.7.2/src/recito/extract.py +71 -0
  40. recito-0.7.2/src/recito/extract_epub.py +799 -0
  41. recito-0.7.2/src/recito/extract_pdf.py +1989 -0
  42. recito-0.7.2/src/recito/main.py +661 -0
  43. recito-0.7.2/src/recito/models.py +463 -0
  44. recito-0.7.2/src/recito/synthesize.py +3710 -0
  45. recito-0.7.2/src/recito/textnorm/__init__.py +243 -0
  46. recito-0.7.2/src/recito/textnorm/backends/__init__.py +1 -0
  47. recito-0.7.2/src/recito/textnorm/backends/basic.py +50 -0
  48. recito-0.7.2/src/recito/textnorm/backends/own_en.py +1130 -0
  49. recito-0.7.2/src/recito/textnorm/data/__init__.py +63 -0
  50. recito-0.7.2/src/recito/textnorm/data/en/acronyms.tsv +39 -0
  51. recito-0.7.2/src/recito/textnorm/data/en/common.tsv +286607 -0
  52. recito-0.7.2/src/recito/textnorm/data/en/currency.tsv +6 -0
  53. recito-0.7.2/src/recito/textnorm/data/en/glue.tsv +11 -0
  54. recito-0.7.2/src/recito/textnorm/data/en/hotkeys.tsv +35 -0
  55. recito-0.7.2/src/recito/textnorm/data/en/months.tsv +18 -0
  56. recito-0.7.2/src/recito/textnorm/data/en/seconds_context.tsv +11 -0
  57. recito-0.7.2/src/recito/textnorm/data/en/symbols.tsv +11 -0
  58. recito-0.7.2/src/recito/textnorm/data/en/units.tsv +40 -0
  59. recito-0.7.2/src/recito/textnorm/data/en/whitelist.tsv +58 -0
  60. recito-0.7.2/src/recito/textnorm/data/en/year_context.tsv +23 -0
  61. recito-0.7.2/src/recito/textnorm/overlay.py +116 -0
  62. recito-0.7.2/src/recito/textnorm/prepare.py +879 -0
  63. recito-0.7.2/src/recito/textnorm/profile.py +795 -0
  64. recito-0.7.2/src/recito/textnorm/rules_en.py +46 -0
  65. recito-0.7.2/src/recito/textnorm/segment.py +265 -0
  66. recito-0.7.2/src/recito/textnorm/urls.py +366 -0
  67. recito-0.7.2/src/recito/textnorm/verbalize.py +89 -0
  68. recito-0.7.2/src/recito.egg-info/PKG-INFO +123 -0
  69. recito-0.7.2/src/recito.egg-info/SOURCES.txt +107 -0
  70. recito-0.7.2/src/recito.egg-info/dependency_links.txt +1 -0
  71. recito-0.7.2/src/recito.egg-info/entry_points.txt +2 -0
  72. recito-0.7.2/src/recito.egg-info/requires.txt +20 -0
  73. recito-0.7.2/src/recito.egg-info/scm_file_list.json +102 -0
  74. recito-0.7.2/src/recito.egg-info/scm_version.json +8 -0
  75. recito-0.7.2/src/recito.egg-info/top_level.txt +1 -0
  76. recito-0.7.2/tests/conftest.py +246 -0
  77. recito-0.7.2/tests/fixtures/verify_pairs.json +137 -0
  78. recito-0.7.2/tests/test_assemble.py +522 -0
  79. recito-0.7.2/tests/test_chunk.py +183 -0
  80. recito-0.7.2/tests/test_clean.py +3341 -0
  81. recito-0.7.2/tests/test_cli.py +428 -0
  82. recito-0.7.2/tests/test_config.py +164 -0
  83. recito-0.7.2/tests/test_cover.py +50 -0
  84. recito-0.7.2/tests/test_engines.py +183 -0
  85. recito-0.7.2/tests/test_extract.py +146 -0
  86. recito-0.7.2/tests/test_extract_epub.py +775 -0
  87. recito-0.7.2/tests/test_layout.py +2499 -0
  88. recito-0.7.2/tests/test_models.py +122 -0
  89. recito-0.7.2/tests/test_synthesize.py +3724 -0
  90. recito-0.7.2/tests/test_vllm_omni.py +568 -0
  91. recito-0.7.2/tests/textnorm/golden_en.tsv +39 -0
  92. recito-0.7.2/tests/textnorm/test_golden_en.py +31 -0
  93. recito-0.7.2/tests/textnorm/test_overlay.py +191 -0
  94. recito-0.7.2/tests/textnorm/test_pipeline.py +241 -0
  95. recito-0.7.2/tests/textnorm/test_prepare.py +793 -0
  96. recito-0.7.2/tests/textnorm/test_profile.py +195 -0
  97. recito-0.7.2/tests/textnorm/test_segment.py +332 -0
  98. recito-0.7.2/tests/textnorm/test_verbalize_en.py +1441 -0
  99. recito-0.7.2/tools/audio_probe.py +205 -0
  100. recito-0.7.2/tools/common_words.py +124 -0
  101. recito-0.7.2/tools/delivery_probe.py +369 -0
  102. recito-0.7.2/tools/norm_probe.py +356 -0
  103. recito-0.7.2/tools/norm_spike.py +347 -0
  104. recito-0.7.2/tools/probe/en.txt +162 -0
  105. recito-0.7.2/tools/probe/nemo-1.2.0-en.txt +83 -0
  106. recito-0.7.2/vllm/README.md +25 -0
  107. recito-0.7.2/vllm/qwen3_tts.yaml +155 -0
  108. recito-0.7.2/vllm/vllm-omni-qwen3-tts-base.service +37 -0
  109. recito-0.7.2/vllm/vllm-omni-qwen3-tts.service +37 -0
@@ -0,0 +1,35 @@
1
+ # recito server-backend settings.
2
+ #
3
+ # Used only when synthesizing via a vLLM-Omni server instead of the
4
+ # default in-process engine:
5
+ #
6
+ # recito speak mybook.pdf --engine vllm-omni
7
+ #
8
+ # Copy this file to .env and edit; real environment variables win over
9
+ # .env values. Only VLLM_OMNI_*-prefixed keys are read from .env — unrelated
10
+ # variables in the file never affect a run. (.env is gitignored; this
11
+ # example file is not.)
12
+
13
+ # Base URL of the vLLM-Omni OpenAI-compatible server.
14
+ # Default: http://127.0.0.1:44001
15
+ #VLLM_OMNI_URL=http://127.0.0.1:44001
16
+
17
+ # The server's --served-model-name.
18
+ # Default: qwen3-tts
19
+ #VLLM_OMNI_MODEL=qwen3-tts
20
+
21
+ # Voice cloning (--voice clone:ref.wav) needs a server serving the Base
22
+ # checkpoint instead — one server serves one checkpoint at a time, so
23
+ # run a second instance and point this file at it, e.g.:
24
+ #VLLM_OMNI_URL=http://127.0.0.1:45001
25
+ #VLLM_OMNI_MODEL=qwen3-tts-base
26
+
27
+ # Maximum concurrent synthesis requests. The server's own scheduler depth
28
+ # (per-replica max_num_seqs x replica count) is the ceiling; going past it
29
+ # only queues requests client-side.
30
+ # Default: 64 (128 measured fastest on an 8xB200 fleet)
31
+ #VLLM_OMNI_CONCURRENCY=128
32
+
33
+ # Per-request timeout in seconds.
34
+ # Default: 600
35
+ #VLLM_OMNI_TIMEOUT=600
@@ -0,0 +1,97 @@
1
+ # Python
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+ *.so
6
+ .Python
7
+ build/
8
+ develop-eggs/
9
+ dist/
10
+ downloads/
11
+ eggs/
12
+ .eggs/
13
+ # lib/ # Commented out - needed for frontend/src/lib
14
+ # lib64/ # Commented out - this was blocking frontend lib files
15
+ parts/
16
+ sdist/
17
+ var/
18
+ wheels/
19
+ *.egg-info/
20
+ .installed.cfg
21
+ *.egg
22
+ MANIFEST
23
+ .pytest_cache/
24
+ .coverage
25
+ htmlcov/
26
+ .tox/
27
+ .nox/
28
+ .hypothesis/
29
+
30
+ # Virtual environments
31
+ venv/
32
+ ENV/
33
+ env/
34
+ .venv
35
+
36
+ # IDE
37
+ .vscode/
38
+ .idea/
39
+ *.swp
40
+ *.swo
41
+ *~
42
+
43
+ # Frontend
44
+ frontend/node_modules/
45
+ frontend/dist/
46
+ frontend/.svelte-kit/
47
+ frontend/build/
48
+ frontend/.env.local
49
+ frontend/.env.*.local
50
+ frontend/npm-debug.log*
51
+ frontend/yarn-debug.log*
52
+ frontend/yarn-error.log*
53
+
54
+ # Logs
55
+ logs/
56
+ *.log
57
+
58
+ # Environment files
59
+ .env
60
+ .env.*
61
+ !.env.example
62
+ !deploy/production.env
63
+
64
+ # Test results
65
+ test_results_*.json
66
+ integration_test_results_*.json
67
+
68
+ # Temporary files
69
+ *.tmp
70
+ *.temp
71
+ .cache/
72
+
73
+ # Database
74
+ *.db
75
+ *.sqlite
76
+ *.sqlite3
77
+
78
+ # Secrets
79
+ *.pem
80
+ *.key
81
+ *.crt
82
+ *.csr
83
+
84
+ # OS files
85
+ Thumbs.db
86
+ Desktop.ini
87
+
88
+ # Backup files
89
+ *.bak
90
+ *.backup
91
+ *~
92
+
93
+ config.toml.local
94
+
95
+ _version.py
96
+
97
+ tmp
@@ -0,0 +1 @@
1
+ 3.12
recito-0.7.2/AGENTS.md ADDED
@@ -0,0 +1,122 @@
1
+ # AGENTS.md
2
+
3
+ Guidance for coding agents working on recito.
4
+
5
+ ## What this is
6
+
7
+ recito turns a PDF or EPUB ebook into a chaptered `.m4b` audiobook
8
+ with local, license-clean TTS (Qwen3-TTS, Apache-2.0 weights). Five
9
+ pipeline stages: EXTRACT → CLEAN → CHUNK → SYNTHESIZE → ASSEMBLE.
10
+ Project page: <https://recito.org>; source:
11
+ <https://spacecruft.org/books/recito>. User
12
+ docs: [README.md](README.md) (brief overview; links on it are absolute
13
+ because PyPI renders it), [docs/INSTALL.md](docs/INSTALL.md) (detailed
14
+ install), [docs/USAGE.md](docs/USAGE.md) (the user guide). Architecture,
15
+ design decisions, measured performance: [docs/DEVELOP.md](docs/DEVELOP.md).
16
+
17
+ ## Layout
18
+
19
+ - `src/recito/` — the package. One module per pipeline stage:
20
+ `extract.py` (dispatch) + `extract_pdf.py` / `extract_epub.py`,
21
+ `clean.py` + `textnorm/`, `chunk.py`, `synthesize.py`,
22
+ `assemble.py`. Plus `models.py` (JSON-round-trippable data model),
23
+ `config.py` (book.toml), `cover.py`, `main.py` (typer CLI), and
24
+ `engines/` (TTS adapters behind the protocol in `engines/base.py`;
25
+ registry in `engines/__init__.py`).
26
+ - `src/recito/textnorm/` — speakability normalization as three
27
+ phases: `prepare.py` (typography, boundary-neutral) → `segment.py`
28
+ (sentence boundaries, the one authority; pysbd + data corrections) →
29
+ `verbalize.py` (one sentence at a time, terminator held aside).
30
+ `profile.py` holds every per-language fact as a `LanguageProfile`;
31
+ `backends/own_en.py` is the English token classifier, `backends/basic.py`
32
+ the digits-only fallback; `overlay.py` + `data/<lang>/*.tsv` +
33
+ `rules_<lang>.py` are the whitelist layers.
34
+ - `tests/` — pytest suite; synthetic PDF/epub fixtures are built in
35
+ `tests/conftest.py`. `tests/textnorm/` is one file per normalization
36
+ phase plus `golden_en.tsv`.
37
+ - `tools/` — dev-only scripts, not shipped: `norm_probe.py`
38
+ (regenerates `tools/probe/en.txt`, the classifier's readings for the
39
+ maintainer's audio check), `norm_spike.py` (the NeMo measurement
40
+ spike; `norm_probe.py` imports its probe rows), `audio_probe.py`
41
+ (renders the probe readings with the real engine and transcribes
42
+ them with the take gate's whisper stack — the audio check itself),
43
+ `delivery_probe.py` (the take gate's replay instrument: re-decodes
44
+ cached/pending takes through `_Verifier.transcribe` and scores them
45
+ with the working tree's `_similarity_raw` via its `spans_out`
46
+ diagnostics — the set gate for gate/verifier rule changes),
47
+ `common_words.py` (regenerates `data/en/common.tsv`, the common-words
48
+ table shared by the verifier and prepare's diacritic fold, from the
49
+ SCOWL-huge US+UK wordlists).
50
+ - `docs/` — INSTALL.md (detailed install), USAGE.md (user guide:
51
+ voices, cloning, server backend, recipes, per-book config,
52
+ troubleshooting), DEVELOP.md (internals), TTS.md (TTS engine
53
+ landscape + license audits), FUTURE.md (planned features),
54
+ ISSUES.md (known bugs), ALTERNATIVES.md (prior art), REJECT.md
55
+ (rejected models/licenses), SGLANG.md (SGLang-Omni server-backend evaluation).
56
+ - `vllm/` — vLLM-Omni server deploy files (systemd units, YAML).
57
+ - `sglang/` — SGLang-Omni server deploy files (systemd units, YAML);
58
+ backend evaluation in docs/SGLANG.md.
59
+ - `tmp/` — gitignored real test corpus. **Never commit ebooks,
60
+ extracted text, or generated audio.**
61
+
62
+ ## Commands
63
+
64
+ ```
65
+ python3.12 -m venv venv
66
+ venv/bin/pip install -e '.[dev]'
67
+ venv/bin/pytest # full suite
68
+ venv/bin/ruff check
69
+ venv/bin/ruff format
70
+ ```
71
+
72
+ GPU-less pipeline smoke test (renders sine tones, not speech — the
73
+ `tone` engine exists for exactly this):
74
+
75
+ ```
76
+ recito speak book.pdf -o out.m4b --engine tone
77
+ ```
78
+
79
+ ## Hard rules
80
+
81
+ - **Quality bar:** the goal is the highest-quality output and the best
82
+ code — no quick fixes, no minimal patches that trade correctness for
83
+ size. Confirm every claim against the code, docs, tests, or corpus
84
+ before making it; if a claim can't be confirmed, say so instead of
85
+ stating it.
86
+ - **License policy:** model weights must be under an OSI-approved
87
+ license (or public domain). NC/ND/custom-restricted weights are
88
+ rejected — check [docs/REJECT.md](docs/REJECT.md) before evaluating
89
+ any model, and add new rejections there. ffmpeg/ffprobe are invoked
90
+ as external binaries only, never linked.
91
+ - **Chunk-hash contract:** an engine's `generation_params()` must cover
92
+ everything that changes the audio and nothing that doesn't — chunk
93
+ ids drive the resume cache across books and runs.
94
+ - **Language is data.** Pipeline code never branches on a language
95
+ code (`if lang == "en"` is a bug); every per-language fact lives in a
96
+ `LanguageProfile` (`textnorm/profile.py`) or a `data/<lang>/*.tsv`
97
+ table. The English classifier and rules apply only to the `en`
98
+ profile; other profiles use the basic verbalizer until they get their
99
+ own (docs/FUTURE.md).
100
+ - **Boundary discipline in textnorm.** `prepare` never reads a number
101
+ or abbreviation; `segment` is the only place a sentence boundary is
102
+ decided; a verbalizer rule never consults sentence context and must
103
+ accept the dotless end-of-body form of any abbreviation it anchors on
104
+ a period (the driver holds the terminator aside). Every table row is
105
+ asserted with the tail byte-identical and no terminator gained.
106
+ Normalizer changes are reviewed on the corpus before the expectation
107
+ tables are re-baselined: `recito extract <book> --workdir DIR`
108
+ on the unmodified tree and again after the change, then diff the two
109
+ workdirs (`.txt` for readings, `clean.json` `sentences` for
110
+ boundaries); every changed paragraph must be explained by the change.
111
+ - **Doc discipline:** new planned work → docs/FUTURE.md; bugs and
112
+ operational issues → docs/ISSUES.md; rejected models/licenses →
113
+ docs/REJECT.md; TTS engine evaluations and license audits →
114
+ docs/TTS.md. Keep README.md user-facing; internals go in
115
+ docs/DEVELOP.md.
116
+ - Python 3.12, ruff is the linter/formatter (config in pyproject.toml).
117
+ - Tests run on CPU via the `tone` engine; don't add GPU requirements to
118
+ the test suite.
119
+ - **Never touch `CHANGELOG.txt`** — release notes are maintained by the
120
+ maintainer.
121
+ - **Never make git commits** (or any other git mutations) — commit and
122
+ changelog decisions belong to the maintainer.
@@ -0,0 +1,24 @@
1
+ v0.7.2 Add build deps.
2
+ v0.7.1 Clearer docs. Prep for pypi.
3
+ v0.7.0 Rename to recito.
4
+ v0.6.10 More fixes, working well despite many quirky source document issues.
5
+ v0.6.9 Yet another 'nother round of bug fixes, lots of text normalization fixes.
6
+ v0.6.8 Yet another round of bug fixes, lots of text normalization fixes.
7
+ v0.6.7 Another round of bug fixes, lots of text normalization fixes.
8
+ v0.6.6 Fix all current known issues/bugs.
9
+ v0.6.5 Endless bug fixes, normalization, overfitting to a few docs.
10
+ v0.6.4 Major text normalization rework. File locking, many bugfixes.
11
+ v0.6.3 Support for cloned voices with vLLM.
12
+ v0.6.2 Fix verifier false positives, speed up, non-blocking.
13
+ v0.6.1 Significant speedups to ffmpeg processing.
14
+ v0.6.0 Support vLLM Omni server. MUCH faster.
15
+ v0.5.2 Add support for epub and other formats.
16
+ v0.5.1 Better metadata. Add cover art.
17
+ v0.5.0 Better doc parsing. Auto output filename. Verify by default.
18
+ v0.4.1 Add feature to use cloned voices. Add developer docs.
19
+ v0.4.0 Multi-GPU support. Confirmed working with 70+ page, 4 hour audio.
20
+ v0.3.0 Initial code implementation.
21
+ v0.2.0 Final plan, ready to code. Updated dependencies.
22
+ v0.1.2 Update plan, future, rejects. License as AGPLv3+.
23
+ v0.1.1 Create plan, evaluate alternatives, reject non-libre.
24
+ v0.1.0 Initial project setup.