narrapy 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- narrapy-0.1.0/.gitignore +30 -0
- narrapy-0.1.0/LICENSE +21 -0
- narrapy-0.1.0/PKG-INFO +176 -0
- narrapy-0.1.0/README.md +149 -0
- narrapy-0.1.0/pyproject.toml +42 -0
- narrapy-0.1.0/src/narrapy/__init__.py +3 -0
- narrapy-0.1.0/src/narrapy/__main__.py +3 -0
- narrapy-0.1.0/src/narrapy/cli.py +450 -0
- narrapy-0.1.0/src/narrapy/espeak_fix.py +56 -0
- narrapy-0.1.0/src/narrapy/voices.py +174 -0
narrapy-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Python virtual environment
|
|
2
|
+
.venv/
|
|
3
|
+
__pycache__/
|
|
4
|
+
|
|
5
|
+
# Local notes
|
|
6
|
+
command.txt
|
|
7
|
+
|
|
8
|
+
# Books
|
|
9
|
+
*.pdf
|
|
10
|
+
|
|
11
|
+
# Generated audio and work folders
|
|
12
|
+
*.m4b
|
|
13
|
+
*.mp3
|
|
14
|
+
*.wav
|
|
15
|
+
*_audiobook_work/
|
|
16
|
+
*.cleaned.txt
|
|
17
|
+
|
|
18
|
+
# Downloaded Piper voice models (large; re-download with piper.download_voices)
|
|
19
|
+
voices/
|
|
20
|
+
|
|
21
|
+
# Voice samples
|
|
22
|
+
voice_samples/
|
|
23
|
+
|
|
24
|
+
# Build output
|
|
25
|
+
dist/
|
|
26
|
+
build/
|
|
27
|
+
*.egg-info/
|
|
28
|
+
|
|
29
|
+
# Backup of the pre-package scripts (kept locally, not pushed)
|
|
30
|
+
previous_version/
|
narrapy-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Rakesh Sharma
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
narrapy-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: narrapy
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Turn PDFs into audiobooks with chapters, using local neural text-to-speech (Kokoro or Piper).
|
|
5
|
+
Project-URL: Homepage, https://github.com/rakishere/narrapy
|
|
6
|
+
Project-URL: Issues, https://github.com/rakishere/narrapy/issues
|
|
7
|
+
Author: Rakesh Sharma
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: audiobook,kokoro,m4b,pdf,piper,text-to-speech,tts
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
20
|
+
Requires-Python: <3.13,>=3.10
|
|
21
|
+
Requires-Dist: kokoro
|
|
22
|
+
Requires-Dist: numpy
|
|
23
|
+
Requires-Dist: piper-tts
|
|
24
|
+
Requires-Dist: pymupdf
|
|
25
|
+
Requires-Dist: soundfile
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# Narrapy
|
|
29
|
+
|
|
30
|
+
Turn any text-based PDF into an audiobook on your own computer. Narrapy reads the PDF, cleans up the text, detects chapters, narrates each chapter with a local neural voice ([Kokoro](https://huggingface.co/hexgrad/Kokoro-82M) or [Piper](https://github.com/rhasspy/piper)), and packages everything into a single `.m4b` audiobook with chapter markers, title, and author.
|
|
31
|
+
|
|
32
|
+
No cloud services, no API keys. After the first run it works fully offline.
|
|
33
|
+
|
|
34
|
+
## Features
|
|
35
|
+
|
|
36
|
+
- **Chapters, automatically.** Uses the PDF's built-in table of contents first. If there isn't one, it looks for "Chapter 3" or "Part Two" style headings. As a last resort it splits the book into 15-page parts so you can still skip around.
|
|
37
|
+
- **Clean narration.** Removes running headers and footers, page numbers, and citation markers like `[12]`. Rejoins hyphenated words and replaces URLs with the word "link" so they aren't read out letter by letter.
|
|
38
|
+
- **Resumable.** A long book can take hours on CPU. If a run stops, run the same command again and finished chapters are skipped.
|
|
39
|
+
- **Voice sampler.** Listen to a short sample of every voice before choosing one.
|
|
40
|
+
- **Two engines.** Kokoro (default, higher quality) or Piper (faster, lighter).
|
|
41
|
+
- **M4B or MP3.** One `.m4b` with chapters for audiobook players, or one MP3 per chapter.
|
|
42
|
+
|
|
43
|
+
## Requirements
|
|
44
|
+
|
|
45
|
+
- Python 3.10, 3.11, or 3.12 (Kokoro doesn't support 3.13+ yet)
|
|
46
|
+
- [ffmpeg](https://ffmpeg.org/) on your PATH
|
|
47
|
+
- [espeak-ng](https://github.com/espeak-ng/espeak-ng/releases) (Kokoro uses it for unusual words)
|
|
48
|
+
- A PDF with a real text layer. Scanned PDFs need OCR first, for example with [ocrmypdf](https://github.com/ocrmypdf/OCRmyPDF).
|
|
49
|
+
|
|
50
|
+
Installing pulls in PyTorch for Kokoro, so expect a download of several hundred MB. On the first Kokoro run the voice model (about 330 MB) downloads once from Hugging Face.
|
|
51
|
+
|
|
52
|
+
## Install
|
|
53
|
+
|
|
54
|
+
Install the tools first:
|
|
55
|
+
|
|
56
|
+
```powershell
|
|
57
|
+
# Windows
|
|
58
|
+
winget install Gyan.FFmpeg
|
|
59
|
+
winget install eSpeak-NG.eSpeak-NG
|
|
60
|
+
# then open a new terminal so ffmpeg is on your PATH
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
On macOS use `brew install ffmpeg espeak-ng`; on Debian/Ubuntu use `sudo apt install ffmpeg espeak-ng`.
|
|
64
|
+
|
|
65
|
+
Then install Narrapy into a Python 3.10-3.12 virtual environment:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
python -m venv .venv
|
|
69
|
+
# Windows: .venv\Scripts\activate macOS/Linux: source .venv/bin/activate
|
|
70
|
+
pip install narrapy
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### From source
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
git clone https://github.com/rakishere/narrapy.git
|
|
77
|
+
cd narrapy
|
|
78
|
+
python -m venv .venv
|
|
79
|
+
pip install -e .
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
On Windows, `run.ps1` runs Narrapy with the project's `.venv` without activating it, e.g. `.\run.ps1 "book.pdf" --preview`.
|
|
83
|
+
|
|
84
|
+
## Usage
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
narrapy --help # all options and the list of voices
|
|
88
|
+
narrapy "book.pdf" --list-chapters # check the chapters look right
|
|
89
|
+
narrapy "book.pdf" --dump-text # review the cleaned text
|
|
90
|
+
narrapy "book.pdf" --preview --voice am_michael # 1-minute voice sample from the book
|
|
91
|
+
narrapy "book.pdf" --voice am_michael --speed 1.1 # make the audiobook
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
The result is saved next to the PDF as `book.m4b`.
|
|
95
|
+
|
|
96
|
+
Skip front or back matter (preface, notes, index) with `--start-page` and `--end-page`:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
narrapy "book.pdf" --voice bf_emma --start-page 9 --end-page 212
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Use Piper instead of Kokoro (download a voice into `./voices` first):
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
python -m piper.download_voices --download-dir voices en_US-lessac-medium
|
|
106
|
+
narrapy "book.pdf" --engine piper --piper-model voices/en_US-lessac-medium.onnx
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
### Choosing a voice
|
|
110
|
+
|
|
111
|
+
Play a short sample of each voice (Ctrl+C stops):
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
narrapy voices --group best # the 6 best voices
|
|
115
|
+
narrapy voices # all English Kokoro voices + Piper voices in ./voices
|
|
116
|
+
narrapy voices --group british # also: american, piper
|
|
117
|
+
narrapy voices am_michael bm_george # only these voices
|
|
118
|
+
narrapy voices --text "Your own sentence" --speed 1.1
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Samples are saved in `~/.cache/narrapy/voice_samples` and reused, so replaying is instant. Playback is built in on Windows; on other systems the WAV files are saved for you to open.
|
|
122
|
+
|
|
123
|
+
| Group | Kokoro voices |
|
|
124
|
+
|---|---|
|
|
125
|
+
| American female | af_heart (default), af_bella, af_nicole, af_aoede, af_kore, af_sarah, af_nova, af_sky, af_alloy, af_jessica, af_river |
|
|
126
|
+
| American male | am_michael, am_fenrir, am_puck, am_echo, am_eric, am_liam, am_onyx, am_adam, am_santa |
|
|
127
|
+
| British female | bf_emma, bf_isabella, bf_alice, bf_lily |
|
|
128
|
+
| British male | bm_george, bm_fable, bm_lewis, bm_daniel |
|
|
129
|
+
|
|
130
|
+
Good starting points: af_heart, af_bella, am_michael, am_fenrir, bf_emma, bm_george.
|
|
131
|
+
|
|
132
|
+
### All options
|
|
133
|
+
|
|
134
|
+
| Option | Description |
|
|
135
|
+
|---|---|
|
|
136
|
+
| `--engine {kokoro,piper}` | Text-to-speech engine (default: kokoro) |
|
|
137
|
+
| `--voice VOICE` | Kokoro voice (default: af_heart) |
|
|
138
|
+
| `--piper-model PATH` | Piper `.onnx` voice file (required with `--engine piper`) |
|
|
139
|
+
| `--speed SPEED` | Reading speed, e.g. 0.9 or 1.15 |
|
|
140
|
+
| `--format {m4b,mp3}` | One `.m4b` with chapters, or one MP3 per chapter |
|
|
141
|
+
| `--bitrate BITRATE` | Audio bitrate (default: 64k, plenty for speech) |
|
|
142
|
+
| `--output PATH` | Output file (m4b) or folder (mp3) |
|
|
143
|
+
| `--title`, `--author` | Override the PDF metadata |
|
|
144
|
+
| `--start-page`, `--end-page` | Page range to read (1-based) |
|
|
145
|
+
| `--pages-per-part N` | Part size when the PDF has no chapters (default: 15) |
|
|
146
|
+
| `--no-announce` | Don't read the chapter title at the start of each chapter |
|
|
147
|
+
| `--list-chapters` | Show detected chapters and exit |
|
|
148
|
+
| `--dump-text` | Save the cleaned text to `book.cleaned.txt` and exit |
|
|
149
|
+
| `--preview` | Make a ~1 minute sample from the first chapter and exit |
|
|
150
|
+
| `--keep-work` | Keep the per-chapter WAV files after finishing |
|
|
151
|
+
| `--version` | Show the version |
|
|
152
|
+
|
|
153
|
+
## How long does it take?
|
|
154
|
+
|
|
155
|
+
Kokoro runs on the CPU. On a 10-core laptop it produces about 1.5 minutes of audio per minute, so a 285-page book (about 8.5 hours of audio) takes roughly 6 hours. Piper is several times faster. Runs are resumable, so you can stop and continue later - just keep the same `--voice` and `--speed`.
|
|
156
|
+
|
|
157
|
+
## Troubleshooting
|
|
158
|
+
|
|
159
|
+
- **`No module named 'soundfile'` (or similar)** - you ran the system Python. Activate the virtual environment first, or use `run.ps1` from a source checkout.
|
|
160
|
+
- **"An Application Control policy has blocked this file"** - Windows Smart App Control blocked a new, unsigned spaCy DLL. Install an older build: `pip install "spacy==3.8.7"`.
|
|
161
|
+
- **"No module named pip" on the first Kokoro run** - Kokoro downloads the spaCy English model with pip. Environments made by `uv venv` have no pip; run `python -m ensurepip` or `uv pip install pip`, then retry.
|
|
162
|
+
- **"Cleanup: removed ... espeak-ng.dll temp folder(s)"** - harmless and Windows-only. Kokoro's phonemizer copies espeak-ng.dll to a temp folder and can't delete it at exit while it is still loaded, so Narrapy removes those folders on the next run.
|
|
163
|
+
- **"Almost no text found"** - the PDF is scanned images. Run OCR on it first.
|
|
164
|
+
|
|
165
|
+
## Project layout
|
|
166
|
+
|
|
167
|
+
| File | Purpose |
|
|
168
|
+
|---|---|
|
|
169
|
+
| `src/narrapy/cli.py` | The converter: text extraction, cleanup, chapters, TTS, packaging |
|
|
170
|
+
| `src/narrapy/voices.py` | `narrapy voices`: make and play short samples of each voice |
|
|
171
|
+
| `src/narrapy/espeak_fix.py` | Quiets a harmless Windows espeak-ng cleanup error |
|
|
172
|
+
| `run.ps1` | Windows launcher for a source checkout |
|
|
173
|
+
|
|
174
|
+
## License
|
|
175
|
+
|
|
176
|
+
MIT. Kokoro-82M is Apache 2.0; Piper voices have their own licenses listed on their model pages. Only convert books you have the right to use.
|
narrapy-0.1.0/README.md
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# Narrapy
|
|
2
|
+
|
|
3
|
+
Turn any text-based PDF into an audiobook on your own computer. Narrapy reads the PDF, cleans up the text, detects chapters, narrates each chapter with a local neural voice ([Kokoro](https://huggingface.co/hexgrad/Kokoro-82M) or [Piper](https://github.com/rhasspy/piper)), and packages everything into a single `.m4b` audiobook with chapter markers, title, and author.
|
|
4
|
+
|
|
5
|
+
No cloud services, no API keys. After the first run it works fully offline.
|
|
6
|
+
|
|
7
|
+
## Features
|
|
8
|
+
|
|
9
|
+
- **Chapters, automatically.** Uses the PDF's built-in table of contents first. If there isn't one, it looks for "Chapter 3" or "Part Two" style headings. As a last resort it splits the book into 15-page parts so you can still skip around.
|
|
10
|
+
- **Clean narration.** Removes running headers and footers, page numbers, and citation markers like `[12]`. Rejoins hyphenated words and replaces URLs with the word "link" so they aren't read out letter by letter.
|
|
11
|
+
- **Resumable.** A long book can take hours on CPU. If a run stops, run the same command again and finished chapters are skipped.
|
|
12
|
+
- **Voice sampler.** Listen to a short sample of every voice before choosing one.
|
|
13
|
+
- **Two engines.** Kokoro (default, higher quality) or Piper (faster, lighter).
|
|
14
|
+
- **M4B or MP3.** One `.m4b` with chapters for audiobook players, or one MP3 per chapter.
|
|
15
|
+
|
|
16
|
+
## Requirements
|
|
17
|
+
|
|
18
|
+
- Python 3.10, 3.11, or 3.12 (Kokoro doesn't support 3.13+ yet)
|
|
19
|
+
- [ffmpeg](https://ffmpeg.org/) on your PATH
|
|
20
|
+
- [espeak-ng](https://github.com/espeak-ng/espeak-ng/releases) (Kokoro uses it for unusual words)
|
|
21
|
+
- A PDF with a real text layer. Scanned PDFs need OCR first, for example with [ocrmypdf](https://github.com/ocrmypdf/OCRmyPDF).
|
|
22
|
+
|
|
23
|
+
Installing pulls in PyTorch for Kokoro, so expect a download of several hundred MB. On the first Kokoro run the voice model (about 330 MB) downloads once from Hugging Face.
|
|
24
|
+
|
|
25
|
+
## Install
|
|
26
|
+
|
|
27
|
+
Install the tools first:
|
|
28
|
+
|
|
29
|
+
```powershell
|
|
30
|
+
# Windows
|
|
31
|
+
winget install Gyan.FFmpeg
|
|
32
|
+
winget install eSpeak-NG.eSpeak-NG
|
|
33
|
+
# then open a new terminal so ffmpeg is on your PATH
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
On macOS use `brew install ffmpeg espeak-ng`; on Debian/Ubuntu use `sudo apt install ffmpeg espeak-ng`.
|
|
37
|
+
|
|
38
|
+
Then install Narrapy into a Python 3.10-3.12 virtual environment:
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
python -m venv .venv
|
|
42
|
+
# Windows: .venv\Scripts\activate macOS/Linux: source .venv/bin/activate
|
|
43
|
+
pip install narrapy
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
### From source
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
git clone https://github.com/rakishere/narrapy.git
|
|
50
|
+
cd narrapy
|
|
51
|
+
python -m venv .venv
|
|
52
|
+
pip install -e .
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
On Windows, `run.ps1` runs Narrapy with the project's `.venv` without activating it, e.g. `.\run.ps1 "book.pdf" --preview`.
|
|
56
|
+
|
|
57
|
+
## Usage
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
narrapy --help # all options and the list of voices
|
|
61
|
+
narrapy "book.pdf" --list-chapters # check the chapters look right
|
|
62
|
+
narrapy "book.pdf" --dump-text # review the cleaned text
|
|
63
|
+
narrapy "book.pdf" --preview --voice am_michael # 1-minute voice sample from the book
|
|
64
|
+
narrapy "book.pdf" --voice am_michael --speed 1.1 # make the audiobook
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
The result is saved next to the PDF as `book.m4b`.
|
|
68
|
+
|
|
69
|
+
Skip front or back matter (preface, notes, index) with `--start-page` and `--end-page`:
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
narrapy "book.pdf" --voice bf_emma --start-page 9 --end-page 212
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Use Piper instead of Kokoro (download a voice into `./voices` first):
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
python -m piper.download_voices --download-dir voices en_US-lessac-medium
|
|
79
|
+
narrapy "book.pdf" --engine piper --piper-model voices/en_US-lessac-medium.onnx
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
### Choosing a voice
|
|
83
|
+
|
|
84
|
+
Play a short sample of each voice (Ctrl+C stops):
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
narrapy voices --group best # the 6 best voices
|
|
88
|
+
narrapy voices # all English Kokoro voices + Piper voices in ./voices
|
|
89
|
+
narrapy voices --group british # also: american, piper
|
|
90
|
+
narrapy voices am_michael bm_george # only these voices
|
|
91
|
+
narrapy voices --text "Your own sentence" --speed 1.1
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Samples are saved in `~/.cache/narrapy/voice_samples` and reused, so replaying is instant. Playback is built in on Windows; on other systems the WAV files are saved for you to open.
|
|
95
|
+
|
|
96
|
+
| Group | Kokoro voices |
|
|
97
|
+
|---|---|
|
|
98
|
+
| American female | af_heart (default), af_bella, af_nicole, af_aoede, af_kore, af_sarah, af_nova, af_sky, af_alloy, af_jessica, af_river |
|
|
99
|
+
| American male | am_michael, am_fenrir, am_puck, am_echo, am_eric, am_liam, am_onyx, am_adam, am_santa |
|
|
100
|
+
| British female | bf_emma, bf_isabella, bf_alice, bf_lily |
|
|
101
|
+
| British male | bm_george, bm_fable, bm_lewis, bm_daniel |
|
|
102
|
+
|
|
103
|
+
Good starting points: af_heart, af_bella, am_michael, am_fenrir, bf_emma, bm_george.
|
|
104
|
+
|
|
105
|
+
### All options
|
|
106
|
+
|
|
107
|
+
| Option | Description |
|
|
108
|
+
|---|---|
|
|
109
|
+
| `--engine {kokoro,piper}` | Text-to-speech engine (default: kokoro) |
|
|
110
|
+
| `--voice VOICE` | Kokoro voice (default: af_heart) |
|
|
111
|
+
| `--piper-model PATH` | Piper `.onnx` voice file (required with `--engine piper`) |
|
|
112
|
+
| `--speed SPEED` | Reading speed, e.g. 0.9 or 1.15 |
|
|
113
|
+
| `--format {m4b,mp3}` | One `.m4b` with chapters, or one MP3 per chapter |
|
|
114
|
+
| `--bitrate BITRATE` | Audio bitrate (default: 64k, plenty for speech) |
|
|
115
|
+
| `--output PATH` | Output file (m4b) or folder (mp3) |
|
|
116
|
+
| `--title`, `--author` | Override the PDF metadata |
|
|
117
|
+
| `--start-page`, `--end-page` | Page range to read (1-based) |
|
|
118
|
+
| `--pages-per-part N` | Part size when the PDF has no chapters (default: 15) |
|
|
119
|
+
| `--no-announce` | Don't read the chapter title at the start of each chapter |
|
|
120
|
+
| `--list-chapters` | Show detected chapters and exit |
|
|
121
|
+
| `--dump-text` | Save the cleaned text to `book.cleaned.txt` and exit |
|
|
122
|
+
| `--preview` | Make a ~1 minute sample from the first chapter and exit |
|
|
123
|
+
| `--keep-work` | Keep the per-chapter WAV files after finishing |
|
|
124
|
+
| `--version` | Show the version |
|
|
125
|
+
|
|
126
|
+
## How long does it take?
|
|
127
|
+
|
|
128
|
+
Kokoro runs on the CPU. On a 10-core laptop it produces about 1.5 minutes of audio per minute, so a 285-page book (about 8.5 hours of audio) takes roughly 6 hours. Piper is several times faster. Runs are resumable, so you can stop and continue later - just keep the same `--voice` and `--speed`.
|
|
129
|
+
|
|
130
|
+
## Troubleshooting
|
|
131
|
+
|
|
132
|
+
- **`No module named 'soundfile'` (or similar)** - you ran the system Python. Activate the virtual environment first, or use `run.ps1` from a source checkout.
|
|
133
|
+
- **"An Application Control policy has blocked this file"** - Windows Smart App Control blocked a new, unsigned spaCy DLL. Install an older build: `pip install "spacy==3.8.7"`.
|
|
134
|
+
- **"No module named pip" on the first Kokoro run** - Kokoro downloads the spaCy English model with pip. Environments made by `uv venv` have no pip; run `python -m ensurepip` or `uv pip install pip`, then retry.
|
|
135
|
+
- **"Cleanup: removed ... espeak-ng.dll temp folder(s)"** - harmless and Windows-only. Kokoro's phonemizer copies espeak-ng.dll to a temp folder and can't delete it at exit while it is still loaded, so Narrapy removes those folders on the next run.
|
|
136
|
+
- **"Almost no text found"** - the PDF is scanned images. Run OCR on it first.
|
|
137
|
+
|
|
138
|
+
## Project layout
|
|
139
|
+
|
|
140
|
+
| File | Purpose |
|
|
141
|
+
|---|---|
|
|
142
|
+
| `src/narrapy/cli.py` | The converter: text extraction, cleanup, chapters, TTS, packaging |
|
|
143
|
+
| `src/narrapy/voices.py` | `narrapy voices`: make and play short samples of each voice |
|
|
144
|
+
| `src/narrapy/espeak_fix.py` | Quiets a harmless Windows espeak-ng cleanup error |
|
|
145
|
+
| `run.ps1` | Windows launcher for a source checkout |
|
|
146
|
+
|
|
147
|
+
## License
|
|
148
|
+
|
|
149
|
+
MIT. Kokoro-82M is Apache 2.0; Piper voices have their own licenses listed on their model pages. Only convert books you have the right to use.
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "narrapy"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Turn PDFs into audiobooks with chapters, using local neural text-to-speech (Kokoro or Piper)."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
authors = [{ name = "Rakesh Sharma" }]
|
|
13
|
+
requires-python = ">=3.10,<3.13"
|
|
14
|
+
keywords = ["pdf", "audiobook", "tts", "text-to-speech", "kokoro", "piper", "m4b"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"Environment :: Console",
|
|
18
|
+
"Intended Audience :: End Users/Desktop",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"pymupdf",
|
|
28
|
+
"soundfile",
|
|
29
|
+
"numpy",
|
|
30
|
+
"kokoro",
|
|
31
|
+
"piper-tts",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://github.com/rakishere/narrapy"
|
|
36
|
+
Issues = "https://github.com/rakishere/narrapy/issues"
|
|
37
|
+
|
|
38
|
+
[project.scripts]
|
|
39
|
+
narrapy = "narrapy.cli:main"
|
|
40
|
+
|
|
41
|
+
[tool.hatch.build.targets.sdist]
|
|
42
|
+
include = ["src/narrapy", "README.md", "LICENSE", "pyproject.toml"]
|
|
@@ -0,0 +1,450 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
narrapy - Convert a PDF into an audiobook (M4B with chapters, or MP3s)
|
|
4
|
+
using fully local text-to-speech: Kokoro or Piper.
|
|
5
|
+
|
|
6
|
+
Usage examples:
|
|
7
|
+
narrapy book.pdf
|
|
8
|
+
narrapy book.pdf --voice am_michael --speed 1.1
|
|
9
|
+
narrapy book.pdf --engine piper --piper-model voices/en_US-lessac-medium.onnx
|
|
10
|
+
narrapy book.pdf --list-chapters (preview chapters, no audio)
|
|
11
|
+
narrapy book.pdf --preview (short voice sample)
|
|
12
|
+
narrapy book.pdf --start-page 5 --end-page 250 --format mp3
|
|
13
|
+
narrapy voices --group best (listen to voices before choosing)
|
|
14
|
+
|
|
15
|
+
ffmpeg must be installed and on your PATH, and espeak-ng for Kokoro.
|
|
16
|
+
|
|
17
|
+
Long books take a while. If the run stops, run the same command again:
|
|
18
|
+
finished chapters are kept in the work folder and skipped.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import re
|
|
23
|
+
import shutil
|
|
24
|
+
import subprocess
|
|
25
|
+
import sys
|
|
26
|
+
import wave
|
|
27
|
+
from collections import Counter
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
import numpy as np
|
|
31
|
+
import pymupdf
|
|
32
|
+
import soundfile as sf
|
|
33
|
+
|
|
34
|
+
from . import __version__
|
|
35
|
+
from .espeak_fix import quiet_espeak_cleanup
|
|
36
|
+
from .voices import main as voices_main, voice_list_text
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# ---------------------------------------------------------------------------
|
|
40
|
+
# 1. Text extraction and cleanup
|
|
41
|
+
# ---------------------------------------------------------------------------
|
|
42
|
+
|
|
43
|
+
def normalize_for_compare(line):
|
|
44
|
+
"""Turn a line into a pattern so 'Page 12' and 'Page 13' look identical."""
|
|
45
|
+
return re.sub(r"\d+", "#", line.strip().lower())
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def page_blocks(page):
|
|
49
|
+
"""Return the text blocks of a page, in reading order, as plain strings."""
|
|
50
|
+
blocks = page.get_text("blocks", sort=True)
|
|
51
|
+
return [b[4].strip() for b in blocks if b[6] == 0 and b[4].strip()]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def find_repeated_edges(doc, first, last):
|
|
55
|
+
"""Detect running headers and footers: short blocks that repeat at the top
|
|
56
|
+
or bottom of many pages."""
|
|
57
|
+
counts = Counter()
|
|
58
|
+
pages = 0
|
|
59
|
+
for i in range(first, last + 1):
|
|
60
|
+
blocks = page_blocks(doc[i])
|
|
61
|
+
if not blocks:
|
|
62
|
+
continue
|
|
63
|
+
pages += 1
|
|
64
|
+
edges = set(blocks[:2] + blocks[-2:])
|
|
65
|
+
for b in edges:
|
|
66
|
+
if len(b) < 120:
|
|
67
|
+
counts[normalize_for_compare(b)] += 1
|
|
68
|
+
if pages < 4:
|
|
69
|
+
return set()
|
|
70
|
+
return {text for text, n in counts.items() if n >= max(3, pages * 0.3)}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
PAGE_NUMBER = re.compile(r"^(page\s*)?\d+(\s*(of|/)\s*\d+)?$", re.IGNORECASE)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def clean_block(text):
|
|
77
|
+
"""Make one block of PDF text speakable."""
|
|
78
|
+
text = re.sub(r"(\w)-\n(\w)", r"\1\2", text) # re-join hyphenated words
|
|
79
|
+
text = text.replace("\n", " ") # lines -> one paragraph
|
|
80
|
+
text = re.sub(r"https?://\S+|www\.\S+", "link", text) # don't read URLs aloud
|
|
81
|
+
text = re.sub(r"\[\d+(,\s*\d+)*\]", "", text) # citation markers [12]
|
|
82
|
+
text = text.replace("\u00ad", "") # soft hyphens
|
|
83
|
+
text = re.sub(r"[\u2013\u2014]", ", ", text) # long dashes -> pause
|
|
84
|
+
text = re.sub(r"\s+", " ", text)
|
|
85
|
+
text = re.sub(r"\s+([.,;:!?])", r"\1", text) # "link ." -> "link."
|
|
86
|
+
return text.strip()
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def extract_pages(doc, first, last):
|
|
90
|
+
"""Return {page_index: [clean paragraphs]} for the page range."""
|
|
91
|
+
repeated = find_repeated_edges(doc, first, last)
|
|
92
|
+
pages = {}
|
|
93
|
+
for i in range(first, last + 1):
|
|
94
|
+
paragraphs = []
|
|
95
|
+
for block in page_blocks(doc[i]):
|
|
96
|
+
if PAGE_NUMBER.match(block.strip()):
|
|
97
|
+
continue
|
|
98
|
+
if normalize_for_compare(block) in repeated:
|
|
99
|
+
continue
|
|
100
|
+
cleaned = clean_block(block)
|
|
101
|
+
if len(re.sub(r"\W", "", cleaned)) >= 2:
|
|
102
|
+
paragraphs.append(cleaned)
|
|
103
|
+
pages[i] = paragraphs
|
|
104
|
+
return pages
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
# ---------------------------------------------------------------------------
|
|
108
|
+
# 2. Chapter detection
|
|
109
|
+
# ---------------------------------------------------------------------------
|
|
110
|
+
|
|
111
|
+
HEADING = re.compile(
|
|
112
|
+
r"^(chapter|part|section|book)\s+([0-9ivxlcdm]+|one|two|three|four|five|six|"
|
|
113
|
+
r"seven|eight|nine|ten|eleven|twelve)\b",
|
|
114
|
+
re.IGNORECASE,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def chapters_from_toc(doc, first, last):
|
|
119
|
+
"""Use the PDF's built-in table of contents (bookmarks), top level only."""
|
|
120
|
+
toc = [t for t in doc.get_toc(simple=True) if t[2] > 0]
|
|
121
|
+
if not toc:
|
|
122
|
+
return []
|
|
123
|
+
top = min(t[0] for t in toc)
|
|
124
|
+
entries = [(title.strip(), page - 1) for lvl, title, page in toc if lvl == top]
|
|
125
|
+
entries = [(t, p) for t, p in entries if first <= p <= last]
|
|
126
|
+
if len(entries) < 2:
|
|
127
|
+
return []
|
|
128
|
+
chapters = []
|
|
129
|
+
if entries[0][1] > first:
|
|
130
|
+
chapters.append(("Opening", first, entries[0][1] - 1))
|
|
131
|
+
for n, (title, start) in enumerate(entries):
|
|
132
|
+
end = entries[n + 1][1] - 1 if n + 1 < len(entries) else last
|
|
133
|
+
if end < start: # two chapters starting on the same page
|
|
134
|
+
end = start
|
|
135
|
+
chapters.append((title, start, end))
|
|
136
|
+
return chapters
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def chapters_from_headings(pages):
|
|
140
|
+
"""Fallback: look for short blocks like 'Chapter 3' or 'PART TWO'."""
|
|
141
|
+
starts = []
|
|
142
|
+
for i, paragraphs in pages.items():
|
|
143
|
+
for p in paragraphs[:3]:
|
|
144
|
+
if len(p) < 80 and HEADING.match(p):
|
|
145
|
+
starts.append((p, i))
|
|
146
|
+
break
|
|
147
|
+
if len(starts) < 2:
|
|
148
|
+
return []
|
|
149
|
+
keys = sorted(pages)
|
|
150
|
+
first, last = keys[0], keys[-1]
|
|
151
|
+
chapters = []
|
|
152
|
+
if starts[0][1] > first:
|
|
153
|
+
chapters.append(("Opening", first, starts[0][1] - 1))
|
|
154
|
+
for n, (title, start) in enumerate(starts):
|
|
155
|
+
end = starts[n + 1][1] - 1 if n + 1 < len(starts) else last
|
|
156
|
+
chapters.append((title, start, end))
|
|
157
|
+
return chapters
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def chapters_by_page_count(first, last, size):
|
|
161
|
+
"""Last resort: fixed-size parts so the player still has navigation."""
|
|
162
|
+
chapters = []
|
|
163
|
+
for n, start in enumerate(range(first, last + 1, size), 1):
|
|
164
|
+
chapters.append((f"Part {n}", start, min(start + size - 1, last)))
|
|
165
|
+
return chapters
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def build_chapters(doc, pages, first, last, pages_per_part):
|
|
169
|
+
chapters = chapters_from_toc(doc, first, last)
|
|
170
|
+
source = "PDF table of contents"
|
|
171
|
+
if not chapters:
|
|
172
|
+
chapters = chapters_from_headings(pages)
|
|
173
|
+
source = "chapter headings in the text"
|
|
174
|
+
if not chapters:
|
|
175
|
+
chapters = chapters_by_page_count(first, last, pages_per_part)
|
|
176
|
+
source = f"fixed parts of {pages_per_part} pages"
|
|
177
|
+
|
|
178
|
+
result = []
|
|
179
|
+
for title, start, end in chapters:
|
|
180
|
+
text = "\n".join(p for i in range(start, end + 1) for p in pages.get(i, []))
|
|
181
|
+
if text.strip():
|
|
182
|
+
result.append({"title": title, "start": start, "end": end, "text": text})
|
|
183
|
+
return result, source
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
# ---------------------------------------------------------------------------
|
|
187
|
+
# 3. Text-to-speech engines (both run locally)
|
|
188
|
+
# ---------------------------------------------------------------------------
|
|
189
|
+
|
|
190
|
+
class KokoroEngine:
|
|
191
|
+
"""Kokoro-82M. Model downloads once from Hugging Face on first run,
|
|
192
|
+
then works offline from the local cache."""
|
|
193
|
+
|
|
194
|
+
sample_rate = 24000
|
|
195
|
+
|
|
196
|
+
def __init__(self, voice, speed):
|
|
197
|
+
from kokoro import KPipeline
|
|
198
|
+
lang = voice[0] if voice and voice[0] in "abefhijpz" else "a"
|
|
199
|
+
self.pipeline = KPipeline(lang_code=lang)
|
|
200
|
+
self.voice = voice
|
|
201
|
+
self.speed = speed
|
|
202
|
+
self.pause = np.zeros(int(self.sample_rate * 0.25), dtype=np.float32)
|
|
203
|
+
|
|
204
|
+
def synthesize(self, text, wav_path):
|
|
205
|
+
with sf.SoundFile(wav_path, "w", samplerate=self.sample_rate,
|
|
206
|
+
channels=1, subtype="PCM_16") as out:
|
|
207
|
+
for result in self.pipeline(text, voice=self.voice, speed=self.speed,
|
|
208
|
+
split_pattern=r"\n+"):
|
|
209
|
+
audio = result[2] if isinstance(result, tuple) else result.audio
|
|
210
|
+
if audio is None:
|
|
211
|
+
continue
|
|
212
|
+
if hasattr(audio, "cpu"):
|
|
213
|
+
audio = audio.cpu().numpy()
|
|
214
|
+
out.write(np.asarray(audio, dtype=np.float32))
|
|
215
|
+
out.write(self.pause)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
class PiperEngine:
|
|
219
|
+
"""Piper. Needs a voice file (.onnx) with its .onnx.json next to it."""
|
|
220
|
+
|
|
221
|
+
def __init__(self, model_path, speed):
|
|
222
|
+
from piper import PiperVoice
|
|
223
|
+
model = Path(model_path)
|
|
224
|
+
if not model.exists():
|
|
225
|
+
sys.exit(f"Piper voice not found: {model}")
|
|
226
|
+
self.voice = PiperVoice.load(str(model))
|
|
227
|
+
self.length_scale = 1.0 / speed
|
|
228
|
+
self.sample_rate = self.voice.config.sample_rate
|
|
229
|
+
|
|
230
|
+
def synthesize(self, text, wav_path):
|
|
231
|
+
with wave.open(str(wav_path), "wb") as wav_file:
|
|
232
|
+
if hasattr(self.voice, "synthesize_wav"): # piper-tts 1.3+
|
|
233
|
+
from piper import SynthesisConfig
|
|
234
|
+
cfg = SynthesisConfig(length_scale=self.length_scale)
|
|
235
|
+
self.voice.synthesize_wav(text, wav_file, syn_config=cfg)
|
|
236
|
+
else: # piper-tts 1.2
|
|
237
|
+
self.voice.synthesize(text, wav_file, length_scale=self.length_scale,
|
|
238
|
+
sentence_silence=0.25)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
# ---------------------------------------------------------------------------
|
|
242
|
+
# 4. Packaging with ffmpeg
|
|
243
|
+
# ---------------------------------------------------------------------------
|
|
244
|
+
|
|
245
|
+
def ffmeta_escape(value):
|
|
246
|
+
return re.sub(r"([=;#\\\n])", r"\\\1", str(value))
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def safe_name(text, limit=60):
|
|
250
|
+
text = re.sub(r'[<>:"/\\|?*\x00-\x1f]', "", text).strip().rstrip(".")
|
|
251
|
+
return (text or "untitled")[:limit]
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def run_ffmpeg(args):
|
|
255
|
+
cmd = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y"] + args
|
|
256
|
+
result = subprocess.run(cmd, capture_output=True, text=True)
|
|
257
|
+
if result.returncode != 0:
|
|
258
|
+
sys.exit(f"ffmpeg failed:\n{result.stderr}")
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def build_m4b(chapter_wavs, titles, out_path, work_dir, book_title, author, bitrate):
|
|
262
|
+
list_file = work_dir / "concat.txt"
|
|
263
|
+
list_file.write_text(
|
|
264
|
+
"".join(f"file '{w.resolve().as_posix()}'\n" for w in chapter_wavs),
|
|
265
|
+
encoding="utf-8")
|
|
266
|
+
|
|
267
|
+
lines = [";FFMETADATA1",
|
|
268
|
+
f"title={ffmeta_escape(book_title)}",
|
|
269
|
+
f"album={ffmeta_escape(book_title)}",
|
|
270
|
+
f"artist={ffmeta_escape(author)}",
|
|
271
|
+
"genre=Audiobook", ""]
|
|
272
|
+
position = 0
|
|
273
|
+
for wav, title in zip(chapter_wavs, titles):
|
|
274
|
+
length = int(sf.info(str(wav)).duration * 1000)
|
|
275
|
+
lines += ["[CHAPTER]", "TIMEBASE=1/1000",
|
|
276
|
+
f"START={position}", f"END={position + length}",
|
|
277
|
+
f"title={ffmeta_escape(title)}", ""]
|
|
278
|
+
position += length
|
|
279
|
+
meta_file = work_dir / "chapters.txt"
|
|
280
|
+
meta_file.write_text("\n".join(lines), encoding="utf-8")
|
|
281
|
+
|
|
282
|
+
run_ffmpeg(["-f", "concat", "-safe", "0", "-i", str(list_file),
|
|
283
|
+
"-i", str(meta_file), "-map", "0:a",
|
|
284
|
+
"-map_metadata", "1", "-map_chapters", "1",
|
|
285
|
+
"-c:a", "aac", "-b:a", bitrate, "-ac", "1",
|
|
286
|
+
"-movflags", "+faststart", "-f", "mp4", str(out_path)])
|
|
287
|
+
return position / 1000
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def build_mp3s(chapter_wavs, titles, out_dir, book_title, author, bitrate):
|
|
291
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
292
|
+
total = 0.0
|
|
293
|
+
for n, (wav, title) in enumerate(zip(chapter_wavs, titles), 1):
|
|
294
|
+
target = out_dir / f"{n:02d} - {safe_name(title)}.mp3"
|
|
295
|
+
run_ffmpeg(["-i", str(wav), "-c:a", "libmp3lame", "-b:a", bitrate,
|
|
296
|
+
"-metadata", f"title={title}", "-metadata", f"album={book_title}",
|
|
297
|
+
"-metadata", f"artist={author}", "-metadata", f"track={n}",
|
|
298
|
+
"-metadata", "genre=Audiobook", str(target)])
|
|
299
|
+
total += sf.info(str(wav)).duration
|
|
300
|
+
return total
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
# ---------------------------------------------------------------------------
|
|
304
|
+
# Main
|
|
305
|
+
# ---------------------------------------------------------------------------
|
|
306
|
+
|
|
307
|
+
EXAMPLES = """examples:
|
|
308
|
+
narrapy book.pdf --list-chapters check the detected chapters
|
|
309
|
+
narrapy book.pdf --dump-text review the cleaned text
|
|
310
|
+
narrapy book.pdf --preview --voice bf_emma 1-minute voice sample
|
|
311
|
+
narrapy book.pdf --voice am_michael --speed 1.1 --end-page 212
|
|
312
|
+
narrapy book.pdf --engine piper --piper-model voices/en_US-lessac-medium.onnx
|
|
313
|
+
|
|
314
|
+
listen to voices before choosing one (plays a ~10 second sample of each):
|
|
315
|
+
narrapy voices --group best the 6 best (also: all, american, british, piper)
|
|
316
|
+
narrapy voices af_heart bm_george only these voices
|
|
317
|
+
narrapy voices --help all sample options
|
|
318
|
+
"""
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def parse_args(argv=None):
|
|
322
|
+
p = argparse.ArgumentParser(
|
|
323
|
+
prog="narrapy",
|
|
324
|
+
description="Convert a PDF into an audiobook with local TTS.",
|
|
325
|
+
epilog=EXAMPLES + "\n" + voice_list_text(),
|
|
326
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
327
|
+
p.add_argument("pdf", help="Path to the PDF file")
|
|
328
|
+
p.add_argument("--engine", choices=["kokoro", "piper"], default="kokoro")
|
|
329
|
+
p.add_argument("--voice", default="af_heart",
|
|
330
|
+
help="Kokoro voice, e.g. af_heart, af_bella, am_michael, am_fenrir, "
|
|
331
|
+
"bf_emma, bm_george (default: af_heart)")
|
|
332
|
+
p.add_argument("--piper-model", help="Path to a Piper .onnx voice file")
|
|
333
|
+
p.add_argument("--speed", type=float, default=1.0, help="Reading speed, e.g. 0.9 or 1.15")
|
|
334
|
+
p.add_argument("--format", choices=["m4b", "mp3"], default="m4b")
|
|
335
|
+
p.add_argument("--bitrate", default="64k", help="Audio bitrate (64k is plenty for speech)")
|
|
336
|
+
p.add_argument("--output", help="Output file (m4b) or folder (mp3)")
|
|
337
|
+
p.add_argument("--title", help="Book title (default: PDF metadata or file name)")
|
|
338
|
+
p.add_argument("--author", help="Author (default: PDF metadata)")
|
|
339
|
+
p.add_argument("--start-page", type=int, default=1, help="First page to read (1-based)")
|
|
340
|
+
p.add_argument("--end-page", type=int, help="Last page to read (1-based)")
|
|
341
|
+
p.add_argument("--pages-per-part", type=int, default=15,
|
|
342
|
+
help="Part size when the PDF has no chapters (default: 15)")
|
|
343
|
+
p.add_argument("--no-announce", action="store_true",
|
|
344
|
+
help="Don't read the chapter title at the start of each chapter")
|
|
345
|
+
p.add_argument("--list-chapters", action="store_true",
|
|
346
|
+
help="Show detected chapters and exit (no audio)")
|
|
347
|
+
p.add_argument("--dump-text", action="store_true",
|
|
348
|
+
help="Save the cleaned text to a .txt file to review, then exit")
|
|
349
|
+
p.add_argument("--preview", action="store_true",
|
|
350
|
+
help="Make a ~1 minute sample from the first chapter and exit")
|
|
351
|
+
p.add_argument("--keep-work", action="store_true",
|
|
352
|
+
help="Keep the per-chapter WAV files after finishing")
|
|
353
|
+
p.add_argument("--version", action="version", version=f"narrapy {__version__}")
|
|
354
|
+
return p.parse_args(argv)
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def main(argv=None):
|
|
358
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
359
|
+
if argv and argv[0] == "voices":
|
|
360
|
+
return voices_main(argv[1:])
|
|
361
|
+
quiet_espeak_cleanup()
|
|
362
|
+
args = parse_args(argv)
|
|
363
|
+
pdf_path = Path(args.pdf)
|
|
364
|
+
if not pdf_path.exists():
|
|
365
|
+
sys.exit(f"File not found: {pdf_path}")
|
|
366
|
+
if shutil.which("ffmpeg") is None and not (args.list_chapters or args.dump_text):
|
|
367
|
+
sys.exit("ffmpeg was not found. Install it and make sure it is on your PATH.")
|
|
368
|
+
if args.engine == "piper" and not args.piper_model:
|
|
369
|
+
sys.exit("--piper-model is required with --engine piper")
|
|
370
|
+
|
|
371
|
+
doc = pymupdf.open(str(pdf_path))
|
|
372
|
+
first = max(args.start_page, 1) - 1
|
|
373
|
+
last = min(args.end_page or doc.page_count, doc.page_count) - 1
|
|
374
|
+
meta = doc.metadata or {}
|
|
375
|
+
book_title = args.title or (meta.get("title") or "").strip() or pdf_path.stem
|
|
376
|
+
author = args.author or (meta.get("author") or "").strip() or "Unknown"
|
|
377
|
+
|
|
378
|
+
print(f"Reading '{pdf_path.name}' pages {first + 1} to {last + 1}...")
|
|
379
|
+
pages = extract_pages(doc, first, last)
|
|
380
|
+
total_chars = sum(len(p) for ps in pages.values() for p in ps)
|
|
381
|
+
if total_chars < 200:
|
|
382
|
+
sys.exit("Almost no text found. This PDF is probably scanned images and "
|
|
383
|
+
"needs OCR first (for example with ocrmypdf).")
|
|
384
|
+
|
|
385
|
+
chapters, source = build_chapters(doc, pages, first, last, args.pages_per_part)
|
|
386
|
+
print(f"Found {len(chapters)} chapters using {source}.")
|
|
387
|
+
print(f"About {total_chars:,} characters, roughly {total_chars / 900 / 60:.1f} hours of audio.\n")
|
|
388
|
+
|
|
389
|
+
if args.list_chapters:
|
|
390
|
+
for n, ch in enumerate(chapters, 1):
|
|
391
|
+
print(f"{n:3d}. {ch['title'][:60]:<60} pages {ch['start'] + 1}-{ch['end'] + 1}"
|
|
392
|
+
f" ({len(ch['text']):,} chars)")
|
|
393
|
+
return
|
|
394
|
+
|
|
395
|
+
if args.dump_text:
|
|
396
|
+
txt_path = pdf_path.with_suffix(".cleaned.txt")
|
|
397
|
+
with open(txt_path, "w", encoding="utf-8") as f:
|
|
398
|
+
for ch in chapters:
|
|
399
|
+
f.write(f"===== {ch['title']} =====\n\n{ch['text']}\n\n")
|
|
400
|
+
print(f"Cleaned text saved to {txt_path}")
|
|
401
|
+
return
|
|
402
|
+
|
|
403
|
+
if args.engine == "kokoro":
|
|
404
|
+
engine = KokoroEngine(args.voice, args.speed)
|
|
405
|
+
else:
|
|
406
|
+
engine = PiperEngine(args.piper_model, args.speed)
|
|
407
|
+
|
|
408
|
+
if args.preview:
|
|
409
|
+
sample = chapters[0]["text"][:1000].rsplit(" ", 1)[0]
|
|
410
|
+
out = pdf_path.with_name(f"{pdf_path.stem}_preview.wav")
|
|
411
|
+
engine.synthesize(sample, out)
|
|
412
|
+
print(f"Preview saved to {out}")
|
|
413
|
+
return
|
|
414
|
+
|
|
415
|
+
work_dir = pdf_path.with_name(f"{pdf_path.stem}_audiobook_work")
|
|
416
|
+
work_dir.mkdir(exist_ok=True)
|
|
417
|
+
|
|
418
|
+
chapter_wavs, titles = [], []
|
|
419
|
+
for n, ch in enumerate(chapters, 1):
|
|
420
|
+
wav = work_dir / f"{n:03d}.wav"
|
|
421
|
+
titles.append(ch["title"])
|
|
422
|
+
chapter_wavs.append(wav)
|
|
423
|
+
if wav.exists():
|
|
424
|
+
print(f"[{n}/{len(chapters)}] {ch['title'][:50]} (already done, skipping)")
|
|
425
|
+
continue
|
|
426
|
+
print(f"[{n}/{len(chapters)}] {ch['title'][:50]}...", flush=True)
|
|
427
|
+
text = ch["text"] if args.no_announce else f"{ch['title']}.\n{ch['text']}"
|
|
428
|
+
tmp = work_dir / f"{n:03d}.partial.wav"
|
|
429
|
+
engine.synthesize(text, tmp)
|
|
430
|
+
tmp.replace(wav) # only mark done once the chapter is complete
|
|
431
|
+
|
|
432
|
+
print("\nPackaging audiobook...")
|
|
433
|
+
if args.format == "m4b":
|
|
434
|
+
out_path = Path(args.output) if args.output else pdf_path.with_suffix(".m4b")
|
|
435
|
+
seconds = build_m4b(chapter_wavs, titles, out_path, work_dir,
|
|
436
|
+
book_title, author, args.bitrate)
|
|
437
|
+
else:
|
|
438
|
+
out_path = Path(args.output) if args.output else pdf_path.with_name(
|
|
439
|
+
f"{safe_name(book_title)} - MP3")
|
|
440
|
+
seconds = build_mp3s(chapter_wavs, titles, out_path, book_title, author, args.bitrate)
|
|
441
|
+
|
|
442
|
+
if not args.keep_work:
|
|
443
|
+
shutil.rmtree(work_dir, ignore_errors=True)
|
|
444
|
+
|
|
445
|
+
h, m = divmod(int(seconds) // 60, 60)
|
|
446
|
+
print(f"Done! {out_path} ({h}h {m}m, {len(chapters)} chapters)")
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
if __name__ == "__main__":
|
|
450
|
+
main()
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Quiet a harmless Windows-only cleanup error from phonemizer (used by Kokoro).
|
|
3
|
+
|
|
4
|
+
phonemizer copies espeak-ng.dll into a temp folder and tries to delete it at
|
|
5
|
+
exit while the DLL is still loaded, which prints "Access is denied" tracebacks
|
|
6
|
+
and leaves the folder behind. We silence that cleanup and, on the next run,
|
|
7
|
+
remove the folders earlier runs left behind (their DLLs are unloaded by then).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import shutil
|
|
11
|
+
import sys
|
|
12
|
+
import tempfile
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _patch_phonemizer():
|
|
17
|
+
try:
|
|
18
|
+
from phonemizer.backend.espeak.api import EspeakAPI
|
|
19
|
+
except ImportError:
|
|
20
|
+
return
|
|
21
|
+
delete = EspeakAPI._delete
|
|
22
|
+
|
|
23
|
+
def quiet_delete_win32(self):
|
|
24
|
+
try:
|
|
25
|
+
delete(self._library, self._tempdir)
|
|
26
|
+
except OSError:
|
|
27
|
+
pass
|
|
28
|
+
|
|
29
|
+
EspeakAPI._delete_win32 = quiet_delete_win32
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _remove_leftovers():
|
|
33
|
+
removed = 0
|
|
34
|
+
for folder in Path(tempfile.gettempdir()).glob("tmp*"):
|
|
35
|
+
try:
|
|
36
|
+
files = list(folder.iterdir()) if folder.is_dir() else []
|
|
37
|
+
except OSError:
|
|
38
|
+
continue
|
|
39
|
+
if len(files) == 1 and files[0].name == "espeak-ng.dll":
|
|
40
|
+
try:
|
|
41
|
+
shutil.rmtree(folder) # fails if another run still has it loaded
|
|
42
|
+
removed += 1
|
|
43
|
+
except OSError:
|
|
44
|
+
pass
|
|
45
|
+
if removed:
|
|
46
|
+
print(f"Cleanup: removed {removed} leftover espeak-ng.dll temp folder(s) from earlier runs.\n"
|
|
47
|
+
" Kokoro's phonemizer copies espeak-ng.dll to a temp folder while it runs and\n"
|
|
48
|
+
" cannot delete it at exit because Windows keeps a loaded DLL locked.\n"
|
|
49
|
+
" This is harmless and does not affect the audio.\n")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def quiet_espeak_cleanup():
|
|
53
|
+
if sys.platform != "win32":
|
|
54
|
+
return
|
|
55
|
+
_remove_leftovers()
|
|
56
|
+
_patch_phonemizer()
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
narrapy voices - Make and play a short sample of each voice so you can pick one.
|
|
4
|
+
|
|
5
|
+
Usage examples:
|
|
6
|
+
narrapy voices (all English Kokoro voices + Piper voices)
|
|
7
|
+
narrapy voices af_heart bm_george (only these voices)
|
|
8
|
+
narrapy voices --group british (american, british, best, piper)
|
|
9
|
+
narrapy voices --text "Chapter one. It was a bright cold day in April."
|
|
10
|
+
narrapy voices --no-play (just save the WAV files)
|
|
11
|
+
narrapy voices --list (print the voice list and exit)
|
|
12
|
+
|
|
13
|
+
Samples are saved in ~/.cache/narrapy/voice_samples and reused, so replaying is instant.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import hashlib
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
SAMPLE_DIR = Path.home() / ".cache" / "narrapy" / "voice_samples"
|
|
22
|
+
PIPER_DIR = Path("voices") # Piper voices are looked up in ./voices of the current folder
|
|
23
|
+
|
|
24
|
+
KOKORO_VOICES = {
|
|
25
|
+
"American female": ["af_heart", "af_bella", "af_nicole", "af_aoede", "af_kore", "af_sarah",
|
|
26
|
+
"af_nova", "af_sky", "af_alloy", "af_jessica", "af_river"],
|
|
27
|
+
"American male": ["am_michael", "am_fenrir", "am_puck", "am_echo", "am_eric", "am_liam",
|
|
28
|
+
"am_onyx", "am_adam", "am_santa"],
|
|
29
|
+
"British female": ["bf_emma", "bf_isabella", "bf_alice", "bf_lily"],
|
|
30
|
+
"British male": ["bm_george", "bm_fable", "bm_lewis", "bm_daniel"],
|
|
31
|
+
}
|
|
32
|
+
BEST = ["af_heart", "af_bella", "am_michael", "am_fenrir", "bf_emma", "bm_george"]
|
|
33
|
+
|
|
34
|
+
DEFAULT_TEXT = ("It was a quiet evening when the letter finally arrived. She read it twice, "
|
|
35
|
+
"smiled, and put the kettle on. Some news deserves a cup of tea.")
|
|
36
|
+
|
|
37
|
+
KOKORO_CACHE = Path.home() / ".cache/huggingface/hub/models--hexgrad--Kokoro-82M/snapshots"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def all_kokoro():
|
|
41
|
+
return [v for group in KOKORO_VOICES.values() for v in group]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def piper_models():
|
|
45
|
+
return sorted(PIPER_DIR.glob("*.onnx"))
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def voice_list_text():
|
|
49
|
+
downloaded = {p.stem for p in KOKORO_CACHE.glob("*/voices/*.pt")}
|
|
50
|
+
lines = ["Kokoro voices for --voice (default engine; af_heart is the default voice):"]
|
|
51
|
+
for label, names in KOKORO_VOICES.items():
|
|
52
|
+
shown = [f"{n}*" if n in downloaded else n for n in names]
|
|
53
|
+
lines.append(f" {label + ':':<19}{', '.join(shown)}")
|
|
54
|
+
lines += [
|
|
55
|
+
f" Best quality: {', '.join(BEST)}.",
|
|
56
|
+
" * = already downloaded. Others download once (~0.5 MB) on first use.",
|
|
57
|
+
" The first letter sets the language (a/b = English); use those for English books.",
|
|
58
|
+
" Other languages exist (e.g. ef_dora Spanish, ff_siwis French, hf_alpha Hindi,",
|
|
59
|
+
" if_sara Italian, pf_dora Portuguese); Japanese/Chinese voices need extra packages.",
|
|
60
|
+
"",
|
|
61
|
+
"Piper voices for --piper-model (in ./voices):",
|
|
62
|
+
]
|
|
63
|
+
models = piper_models()
|
|
64
|
+
lines += [f" {m.as_posix()}" for m in models] or [" (none downloaded)"]
|
|
65
|
+
lines += [
|
|
66
|
+
" Get more: python -m piper.download_voices --download-dir voices <name>",
|
|
67
|
+
" Browse names at https://huggingface.co/rhasspy/piper-voices",
|
|
68
|
+
]
|
|
69
|
+
return "\n".join(lines)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def pick_voices(args):
|
|
73
|
+
"""Return a list of (name, kind) where kind is 'kokoro' or a Piper model path."""
|
|
74
|
+
piper = {m.stem: m for m in piper_models()}
|
|
75
|
+
if args.voices:
|
|
76
|
+
picked = []
|
|
77
|
+
for v in args.voices:
|
|
78
|
+
if v in piper:
|
|
79
|
+
picked.append((v, piper[v]))
|
|
80
|
+
elif Path(v).suffix == ".onnx" and Path(v).exists():
|
|
81
|
+
picked.append((Path(v).stem, Path(v)))
|
|
82
|
+
elif v in all_kokoro() or (len(v) > 3 and v[2] == "_"):
|
|
83
|
+
picked.append((v, "kokoro"))
|
|
84
|
+
else:
|
|
85
|
+
sys.exit(f"Unknown voice: {v}. Run with --list to see the voices.")
|
|
86
|
+
return picked
|
|
87
|
+
group = args.group
|
|
88
|
+
if group == "american":
|
|
89
|
+
names = KOKORO_VOICES["American female"] + KOKORO_VOICES["American male"]
|
|
90
|
+
elif group == "british":
|
|
91
|
+
names = KOKORO_VOICES["British female"] + KOKORO_VOICES["British male"]
|
|
92
|
+
elif group == "best":
|
|
93
|
+
names = BEST
|
|
94
|
+
elif group == "piper":
|
|
95
|
+
names = []
|
|
96
|
+
else:
|
|
97
|
+
names = all_kokoro()
|
|
98
|
+
picked = [(n, "kokoro") for n in names]
|
|
99
|
+
if group in ("all", "piper"):
|
|
100
|
+
picked += [(m.stem, m) for m in piper_models()]
|
|
101
|
+
return picked
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def play(wav_path):
|
|
105
|
+
try:
|
|
106
|
+
import winsound
|
|
107
|
+
winsound.PlaySound(str(wav_path), winsound.SND_FILENAME)
|
|
108
|
+
except ImportError:
|
|
109
|
+
print(" (playback is only built in on Windows; open the WAV file to listen)")
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def main(argv=None):
|
|
113
|
+
try:
|
|
114
|
+
run(argv)
|
|
115
|
+
except KeyboardInterrupt:
|
|
116
|
+
print("\nStopped.")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def run(argv=None):
|
|
120
|
+
p = argparse.ArgumentParser(prog="narrapy voices",
|
|
121
|
+
description="Make and play a short sample of each voice.")
|
|
122
|
+
p.add_argument("voices", nargs="*",
|
|
123
|
+
help="Voices to sample, e.g. af_heart bm_george en_US-lessac-medium "
|
|
124
|
+
"(default: every voice in --group)")
|
|
125
|
+
p.add_argument("--group", choices=["all", "american", "british", "best", "piper"],
|
|
126
|
+
default="all", help="Which voices to sample when none are named (default: all)")
|
|
127
|
+
p.add_argument("--text", default=DEFAULT_TEXT, help="Sentence to read in each sample")
|
|
128
|
+
p.add_argument("--speed", type=float, default=1.0, help="Reading speed, e.g. 0.9 or 1.15")
|
|
129
|
+
p.add_argument("--no-play", action="store_true", help="Only save the samples, don't play them")
|
|
130
|
+
p.add_argument("--regenerate", action="store_true", help="Remake samples even if saved ones exist")
|
|
131
|
+
p.add_argument("--list", action="store_true", help="Print the available voices and exit")
|
|
132
|
+
args = p.parse_args(argv)
|
|
133
|
+
|
|
134
|
+
if args.list:
|
|
135
|
+
print(voice_list_text())
|
|
136
|
+
return
|
|
137
|
+
|
|
138
|
+
picked = pick_voices(args)
|
|
139
|
+
if not picked:
|
|
140
|
+
sys.exit("No voices to sample.")
|
|
141
|
+
|
|
142
|
+
# Import the engines only when needed; they pull in torch and friends.
|
|
143
|
+
from .cli import KokoroEngine, PiperEngine
|
|
144
|
+
from .espeak_fix import quiet_espeak_cleanup
|
|
145
|
+
quiet_espeak_cleanup()
|
|
146
|
+
|
|
147
|
+
SAMPLE_DIR.mkdir(parents=True, exist_ok=True)
|
|
148
|
+
# Different text or speed gets its own files, so saved samples always match.
|
|
149
|
+
tag = hashlib.sha1(f"{args.text}|{args.speed}".encode()).hexdigest()[:8]
|
|
150
|
+
kokoro_by_lang = {} # one Kokoro pipeline per language, voice switched per sample
|
|
151
|
+
|
|
152
|
+
print(f"Sampling {len(picked)} voice(s). Press Ctrl+C to stop.\n")
|
|
153
|
+
for i, (name, kind) in enumerate(picked, 1):
|
|
154
|
+
wav = SAMPLE_DIR / f"{name}_{tag}.wav"
|
|
155
|
+
if args.regenerate or not wav.exists():
|
|
156
|
+
if kind == "kokoro":
|
|
157
|
+
lang = name[0]
|
|
158
|
+
if lang not in kokoro_by_lang:
|
|
159
|
+
kokoro_by_lang[lang] = KokoroEngine(name, args.speed)
|
|
160
|
+
engine = kokoro_by_lang[lang]
|
|
161
|
+
engine.voice = name
|
|
162
|
+
else:
|
|
163
|
+
engine = PiperEngine(kind, args.speed)
|
|
164
|
+
tmp = wav.with_suffix(".partial.wav")
|
|
165
|
+
engine.synthesize(args.text, tmp)
|
|
166
|
+
tmp.replace(wav)
|
|
167
|
+
label = "Piper" if kind != "kokoro" else "Kokoro"
|
|
168
|
+
print(f"[{i}/{len(picked)}] {name} ({label})", flush=True)
|
|
169
|
+
if not args.no_play:
|
|
170
|
+
play(wav)
|
|
171
|
+
|
|
172
|
+
print(f"\nSamples saved in {SAMPLE_DIR}")
|
|
173
|
+
print('Use one with: narrapy "book.pdf" --voice <name>')
|
|
174
|
+
print(' (Piper: narrapy "book.pdf" --engine piper --piper-model voices/<name>.onnx)')
|