kawi-tts 1.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kawi_tts-1.1.2/LICENSE +21 -0
- kawi_tts-1.1.2/PKG-INFO +164 -0
- kawi_tts-1.1.2/README.md +131 -0
- kawi_tts-1.1.2/kawi_tts/__init__.py +3 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/__init__.py +28 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/espeak_backend.py +213 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/mapper.py +152 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/pipeline.py +119 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/strategies/__init__.py +10 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/strategies/base.py +35 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/strategies/profile_a.py +123 -0
- kawi_tts-1.1.2/kawi_tts/acoustic/strategies/profile_b.py +67 -0
- kawi_tts-1.1.2/kawi_tts/cli/__init__.py +0 -0
- kawi_tts-1.1.2/kawi_tts/cli/coverage_report.py +55 -0
- kawi_tts-1.1.2/kawi_tts/cli/kawi_trace.py +90 -0
- kawi_tts-1.1.2/kawi_tts/evaluation/__init__.py +5 -0
- kawi_tts-1.1.2/kawi_tts/evaluation/evaluate_ojw.py +343 -0
- kawi_tts-1.1.2/kawi_tts/g2p/__init__.py +10 -0
- kawi_tts-1.1.2/kawi_tts/g2p/engine.py +143 -0
- kawi_tts-1.1.2/kawi_tts/normalization/__init__.py +31 -0
- kawi_tts-1.1.2/kawi_tts/normalization/normalizer.py +246 -0
- kawi_tts-1.1.2/kawi_tts/normalization/tokenizer.py +210 -0
- kawi_tts-1.1.2/kawi_tts/tts/__init__.py +5 -0
- kawi_tts-1.1.2/kawi_tts/tts/experimental_piper.py +55 -0
- kawi_tts-1.1.2/kawi_tts.egg-info/PKG-INFO +164 -0
- kawi_tts-1.1.2/kawi_tts.egg-info/SOURCES.txt +44 -0
- kawi_tts-1.1.2/kawi_tts.egg-info/dependency_links.txt +1 -0
- kawi_tts-1.1.2/kawi_tts.egg-info/entry_points.txt +2 -0
- kawi_tts-1.1.2/kawi_tts.egg-info/requires.txt +4 -0
- kawi_tts-1.1.2/kawi_tts.egg-info/top_level.txt +1 -0
- kawi_tts-1.1.2/pyproject.toml +62 -0
- kawi_tts-1.1.2/setup.cfg +4 -0
- kawi_tts-1.1.2/tests/test_acoustic_mapper.py +115 -0
- kawi_tts-1.1.2/tests/test_acoustic_mapper_profile_a.py +74 -0
- kawi_tts-1.1.2/tests/test_espeak_backend.py +30 -0
- kawi_tts-1.1.2/tests/test_evaluation.py +71 -0
- kawi_tts-1.1.2/tests/test_g2p.py +109 -0
- kawi_tts-1.1.2/tests/test_g2p_bugfix.py +33 -0
- kawi_tts-1.1.2/tests/test_integration.py +220 -0
- kawi_tts-1.1.2/tests/test_normalization.py +241 -0
- kawi_tts-1.1.2/tests/test_profile_strategy.py +88 -0
- kawi_tts-1.1.2/tests/test_regression_corpus.py +73 -0
- kawi_tts-1.1.2/tests/test_scaffold.py +19 -0
- kawi_tts-1.1.2/tests/test_synthesis_pipeline.py +117 -0
- kawi_tts-1.1.2/tests/test_tokenizer.py +213 -0
- kawi_tts-1.1.2/tests/test_validation.py +216 -0
kawi_tts-1.1.2/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Kawi-TTS Contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
kawi_tts-1.1.2/PKG-INFO
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: kawi-tts
|
|
3
|
+
Version: 1.1.2
|
|
4
|
+
Summary: Deterministic pronunciation and reconstruction engine for Old Javanese/Kawi.
|
|
5
|
+
Author: Project-TTS-Kawi
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/krispypasta/kawi-tts
|
|
8
|
+
Project-URL: Repository, https://github.com/krispypasta/kawi-tts
|
|
9
|
+
Project-URL: Bug Tracker, https://github.com/krispypasta/kawi-tts/issues
|
|
10
|
+
Project-URL: Documentation, https://github.com/krispypasta/kawi-tts#readme
|
|
11
|
+
Keywords: kawi,old-javanese,linguistics,phonology,g2p,text-to-speech,reconstruction,austronesian
|
|
12
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
23
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
24
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
25
|
+
Classifier: Operating System :: OS Independent
|
|
26
|
+
Requires-Python: >=3.8
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: pytest; extra == "dev"
|
|
31
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
34
|
+
# Kawi-TTS
|
|
35
|
+
|
|
36
|
+
An open-source deterministic pronunciation and reconstruction engine for Old Javanese (Kawi).
|
|
37
|
+
|
|
38
|
+
Kawi-TTS is **not** a natural neural TTS system, it does **not** claim to generate historically authentic speech, and it is **not** a linguistic oracle. It is a strictly deterministic pipeline that implements a documented linguistic reconstruction policy and provides reproducible acoustic output through a selected backend.
|
|
39
|
+
|
|
40
|
+
## What It Does
|
|
41
|
+
|
|
42
|
+
Kawi-TTS transforms Old Javanese text into phonetic representations and synthesizes them into audio. The pipeline is strictly deterministic:
|
|
43
|
+
|
|
44
|
+
`Text → Normalization → G2P / Tokenization → Canonical Representation → Profile A / Profile B → Acoustic Mapping → eSpeak-ng → WAV`
|
|
45
|
+
|
|
46
|
+
The engine resolves linguistic features through two distinct profiles:
|
|
47
|
+
* **Profile A (Reconstructed Spoken):** Represents the historical spoken target. It is explicitly evidence-scoped, includes documented spoken mergers and adaptations, and strictly preserves unresolved items as unresolved where evidence is lacking.
|
|
48
|
+
* **Profile B (Scholarly/Orthographic):** A scholarly reading profile that artificially preserves the intended distinctions of the scholarly orthographic representation (such as unmerged Sanskrit distinctions).
|
|
49
|
+
|
|
50
|
+
## Important Epistemic Limitation
|
|
51
|
+
|
|
52
|
+
**The generated audio is NOT claimed to reproduce exactly how historical Old Javanese speakers sounded.**
|
|
53
|
+
|
|
54
|
+
The software simply operationalizes the project's documented linguistic reconstruction and backend policy. Users must clearly distinguish between:
|
|
55
|
+
1. **Evidence-backed linguistic rules** (e.g., historical spoken mergers)
|
|
56
|
+
2. **Provisional reconstructions** (e.g., adaptation of Sanskrit vocalic liquids to native Javanese phonotactics)
|
|
57
|
+
3. **Engineering policy** (e.g., deferring vowel duration resolution)
|
|
58
|
+
4. **Backend approximations** (e.g., dropping retroflex distinctions due to eSpeak limitations)
|
|
59
|
+
|
|
60
|
+
## Current Status
|
|
61
|
+
|
|
62
|
+
* **Core:** The deterministic core is accepted and historically frozen.
|
|
63
|
+
* **Tests:** The current test suite passes 97/97. The 100-form regression corpus passes fully.
|
|
64
|
+
* **Backend:** eSpeak-ng is the current acoustic backend. The `id` (Indonesian) voice is utilized with explicit backend-scoped approximations where native eSpeak support is lacking.
|
|
65
|
+
* **Infrastructure:** Neural TTS is explicitly deferred. The project remains strictly zero-budget compatible. No cloud service or proprietary data is required for core operation.
|
|
66
|
+
* **Distribution:** The package is not yet published to PyPI.
|
|
67
|
+
|
|
68
|
+
## Quick Start
|
|
69
|
+
|
|
70
|
+
You can run the engine locally without any cloud dependencies.
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
# 1. Clone the repository
|
|
74
|
+
git clone https://github.com/krispypasta/kawi-tts.git
|
|
75
|
+
cd kawi-tts
|
|
76
|
+
|
|
77
|
+
# 2. Create and activate a Python virtual environment
|
|
78
|
+
python -m venv venv
|
|
79
|
+
# On Windows: venv\Scripts\activate
|
|
80
|
+
# On Linux/macOS: source venv/bin/activate
|
|
81
|
+
|
|
82
|
+
# 3. Install the package locally in editable mode
|
|
83
|
+
pip install -e .
|
|
84
|
+
|
|
85
|
+
# 4. Verify installation by running tests
|
|
86
|
+
python -m unittest discover -s tests -p "test_*.py"
|
|
87
|
+
|
|
88
|
+
# 5. Use the trace diagnostics CLI
|
|
89
|
+
kawi-trace "sĕkar"
|
|
90
|
+
|
|
91
|
+
# 6. Generate demo audio (requires eSpeak-ng installed on your system)
|
|
92
|
+
python run_audio_demo.py
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## eSpeak Dependency
|
|
96
|
+
|
|
97
|
+
The deterministic linguistic processing (Normalization, G2P, Profiling) operates entirely independent of eSpeak. However, actual WAV synthesis requires `espeak-ng` to be installed on your system path.
|
|
98
|
+
|
|
99
|
+
If eSpeak-ng is missing, the pipeline produces an actionable error. The package will **not** silently pretend that dummy audio is real synthesis. Current acoustic mapping supports the eSpeak `id` voice, utilizing explicit `BACKEND_APPROXIMATION` rules to bypass unsupported IPA characters safely.
|
|
100
|
+
|
|
101
|
+
## Examples
|
|
102
|
+
|
|
103
|
+
* `sĕkar` — Standard Javanese lexical item. Both profiles map identically to Javanese schwa and native consonants.
|
|
104
|
+
* `BHAṬĀRA` — Sanskrit loan.
|
|
105
|
+
* *Profile B* preserves the aspirated `bʱ`, retroflex `ṭ`, and duration `aː`.
|
|
106
|
+
* *Profile A* merges the aspirate to `b`, maps the retroflex to `ṭ`, and leaves duration unresolved.
|
|
107
|
+
* *Acoustic Mapper* then approximates retroflex `ṭ` to dental `t` for eSpeak compatibility.
|
|
108
|
+
* `śānti` — Sanskrit loan.
|
|
109
|
+
* *Profile B* preserves palatal sibilant `ś`.
|
|
110
|
+
* *Profile A* merges it to native `s`.
|
|
111
|
+
* `kṝta` — Sanskrit loan with a long vocalic liquid.
|
|
112
|
+
* *Profile B* preserves canonical long `r̩ː`.
|
|
113
|
+
* *Profile A* adapts it to native phonotactics as a short Javanese schwa base (`rə`), deliberately discarding non-native duration as per reconstruction policy.
|
|
114
|
+
|
|
115
|
+
## Trace / Diagnostics
|
|
116
|
+
|
|
117
|
+
The `kawi-trace` CLI utility exposes the exact transformation of every string. It traces:
|
|
118
|
+
`Input → Normalized → Canonical → Profile Target → Backend Target`
|
|
119
|
+
|
|
120
|
+
Backend approximations (e.g., eSpeak's inability to pronounce retroflex consonants) are visibly distinguished in the terminal output from intentional linguistic targets.
|
|
121
|
+
|
|
122
|
+
## Project Structure
|
|
123
|
+
|
|
124
|
+
```text
|
|
125
|
+
kawi_tts/ # Core engine (normalization, g2p, acoustic mappers)
|
|
126
|
+
docs/ # Project documentation, policies, and research logs
|
|
127
|
+
tests/ # Test suite and regression corpora
|
|
128
|
+
artifacts/ # Generated WAVs and manifest outputs
|
|
129
|
+
pyproject.toml # Project build configuration
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## Open-Source & Contributing
|
|
133
|
+
|
|
134
|
+
Kawi-TTS is an open-source research tool.
|
|
135
|
+
* **Evidence Before Code:** Research and evidence must precede any linguistic rule changes. Contributors must not invent pronunciation rules.
|
|
136
|
+
* **Changes:** Modifying linguistic policy requires cited evidence and corresponding regression coverage.
|
|
137
|
+
* **Boundaries:** Backend approximations must remain strictly backend-scoped (within the `AcousticMapper`). The canonical representation must remain protected.
|
|
138
|
+
|
|
139
|
+
Please see the `docs/` folder for architectural decisions and governance.
|
|
140
|
+
|
|
141
|
+
## Kawi Learn Relationship
|
|
142
|
+
|
|
143
|
+
**Kawi Learn** is considered a future, external educational application that may consume Kawi-TTS.
|
|
144
|
+
* **Kawi-TTS:** A strict pronunciation and reconstruction engine.
|
|
145
|
+
* **Kawi Learn:** A future end-user application.
|
|
146
|
+
|
|
147
|
+
Kawi-TTS must remain entirely independent of Kawi Learn. Kawi Learn does not currently exist as a finished product.
|
|
148
|
+
|
|
149
|
+
## Future / Deferred Work
|
|
150
|
+
|
|
151
|
+
The following items are explicitly deferred to protect the zero-budget constraint:
|
|
152
|
+
* Neural TTS models
|
|
153
|
+
* Custom recorded speech corpora
|
|
154
|
+
* New acoustic backends requiring unavailable computational resources
|
|
155
|
+
|
|
156
|
+
These will only be reopened if specific, resource-compatible packaging requirements are met.
|
|
157
|
+
|
|
158
|
+
## Research & Citation
|
|
159
|
+
|
|
160
|
+
Linguistic decisions, open questions, and principal references are maintained in `docs/RESEARCH_LOG.md` and the adjacent policy documents (e.g., `docs/P5_002_ACOUSTIC_PROFILE_POLICY.md`). All linguistic claims within the codebase are tied to these cited policy rules.
|
|
161
|
+
|
|
162
|
+
## License
|
|
163
|
+
|
|
164
|
+
Kawi-TTS is licensed under the MIT License. See the `LICENSE` file for details.
|
kawi_tts-1.1.2/README.md
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# Kawi-TTS
|
|
2
|
+
|
|
3
|
+
An open-source deterministic pronunciation and reconstruction engine for Old Javanese (Kawi).
|
|
4
|
+
|
|
5
|
+
Kawi-TTS is **not** a natural neural TTS system, it does **not** claim to generate historically authentic speech, and it is **not** a linguistic oracle. It is a strictly deterministic pipeline that implements a documented linguistic reconstruction policy and provides reproducible acoustic output through a selected backend.
|
|
6
|
+
|
|
7
|
+
## What It Does
|
|
8
|
+
|
|
9
|
+
Kawi-TTS transforms Old Javanese text into phonetic representations and synthesizes them into audio. The pipeline is strictly deterministic:
|
|
10
|
+
|
|
11
|
+
`Text → Normalization → G2P / Tokenization → Canonical Representation → Profile A / Profile B → Acoustic Mapping → eSpeak-ng → WAV`
|
|
12
|
+
|
|
13
|
+
The engine resolves linguistic features through two distinct profiles:
|
|
14
|
+
* **Profile A (Reconstructed Spoken):** Represents the historical spoken target. It is explicitly evidence-scoped, includes documented spoken mergers and adaptations, and strictly preserves unresolved items as unresolved where evidence is lacking.
|
|
15
|
+
* **Profile B (Scholarly/Orthographic):** A scholarly reading profile that artificially preserves the intended distinctions of the scholarly orthographic representation (such as unmerged Sanskrit distinctions).
|
|
16
|
+
|
|
17
|
+
## Important Epistemic Limitation
|
|
18
|
+
|
|
19
|
+
**The generated audio is NOT claimed to reproduce exactly how historical Old Javanese speakers sounded.**
|
|
20
|
+
|
|
21
|
+
The software simply operationalizes the project's documented linguistic reconstruction and backend policy. Users must clearly distinguish between:
|
|
22
|
+
1. **Evidence-backed linguistic rules** (e.g., historical spoken mergers)
|
|
23
|
+
2. **Provisional reconstructions** (e.g., adaptation of Sanskrit vocalic liquids to native Javanese phonotactics)
|
|
24
|
+
3. **Engineering policy** (e.g., deferring vowel duration resolution)
|
|
25
|
+
4. **Backend approximations** (e.g., dropping retroflex distinctions due to eSpeak limitations)
|
|
26
|
+
|
|
27
|
+
## Current Status
|
|
28
|
+
|
|
29
|
+
* **Core:** The deterministic core is accepted and historically frozen.
|
|
30
|
+
* **Tests:** The current test suite passes 97/97. The 100-form regression corpus passes fully.
|
|
31
|
+
* **Backend:** eSpeak-ng is the current acoustic backend. The `id` (Indonesian) voice is utilized with explicit backend-scoped approximations where native eSpeak support is lacking.
|
|
32
|
+
* **Infrastructure:** Neural TTS is explicitly deferred. The project remains strictly zero-budget compatible. No cloud service or proprietary data is required for core operation.
|
|
33
|
+
* **Distribution:** The package is not yet published to PyPI.
|
|
34
|
+
|
|
35
|
+
## Quick Start
|
|
36
|
+
|
|
37
|
+
You can run the engine locally without any cloud dependencies.
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
# 1. Clone the repository
|
|
41
|
+
git clone https://github.com/krispypasta/kawi-tts.git
|
|
42
|
+
cd kawi-tts
|
|
43
|
+
|
|
44
|
+
# 2. Create and activate a Python virtual environment
|
|
45
|
+
python -m venv venv
|
|
46
|
+
# On Windows: venv\Scripts\activate
|
|
47
|
+
# On Linux/macOS: source venv/bin/activate
|
|
48
|
+
|
|
49
|
+
# 3. Install the package locally in editable mode
|
|
50
|
+
pip install -e .
|
|
51
|
+
|
|
52
|
+
# 4. Verify installation by running tests
|
|
53
|
+
python -m unittest discover -s tests -p "test_*.py"
|
|
54
|
+
|
|
55
|
+
# 5. Use the trace diagnostics CLI
|
|
56
|
+
kawi-trace "sĕkar"
|
|
57
|
+
|
|
58
|
+
# 6. Generate demo audio (requires eSpeak-ng installed on your system)
|
|
59
|
+
python run_audio_demo.py
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## eSpeak Dependency
|
|
63
|
+
|
|
64
|
+
The deterministic linguistic processing (Normalization, G2P, Profiling) operates entirely independent of eSpeak. However, actual WAV synthesis requires `espeak-ng` to be installed on your system path.
|
|
65
|
+
|
|
66
|
+
If eSpeak-ng is missing, the pipeline produces an actionable error. The package will **not** silently pretend that dummy audio is real synthesis. Current acoustic mapping supports the eSpeak `id` voice, utilizing explicit `BACKEND_APPROXIMATION` rules to bypass unsupported IPA characters safely.
|
|
67
|
+
|
|
68
|
+
## Examples
|
|
69
|
+
|
|
70
|
+
* `sĕkar` — Standard Javanese lexical item. Both profiles map identically to Javanese schwa and native consonants.
|
|
71
|
+
* `BHAṬĀRA` — Sanskrit loan.
|
|
72
|
+
* *Profile B* preserves the aspirated `bʱ`, retroflex `ṭ`, and duration `aː`.
|
|
73
|
+
* *Profile A* merges the aspirate to `b`, maps the retroflex to `ṭ`, and leaves duration unresolved.
|
|
74
|
+
* *Acoustic Mapper* then approximates retroflex `ṭ` to dental `t` for eSpeak compatibility.
|
|
75
|
+
* `śānti` — Sanskrit loan.
|
|
76
|
+
* *Profile B* preserves palatal sibilant `ś`.
|
|
77
|
+
* *Profile A* merges it to native `s`.
|
|
78
|
+
* `kṝta` — Sanskrit loan with a long vocalic liquid.
|
|
79
|
+
* *Profile B* preserves canonical long `r̩ː`.
|
|
80
|
+
* *Profile A* adapts it to native phonotactics as a short Javanese schwa base (`rə`), deliberately discarding non-native duration as per reconstruction policy.
|
|
81
|
+
|
|
82
|
+
## Trace / Diagnostics
|
|
83
|
+
|
|
84
|
+
The `kawi-trace` CLI utility exposes the exact transformation of every string. It traces:
|
|
85
|
+
`Input → Normalized → Canonical → Profile Target → Backend Target`
|
|
86
|
+
|
|
87
|
+
Backend approximations (e.g., eSpeak's inability to pronounce retroflex consonants) are visibly distinguished in the terminal output from intentional linguistic targets.
|
|
88
|
+
|
|
89
|
+
## Project Structure
|
|
90
|
+
|
|
91
|
+
```text
|
|
92
|
+
kawi_tts/ # Core engine (normalization, g2p, acoustic mappers)
|
|
93
|
+
docs/ # Project documentation, policies, and research logs
|
|
94
|
+
tests/ # Test suite and regression corpora
|
|
95
|
+
artifacts/ # Generated WAVs and manifest outputs
|
|
96
|
+
pyproject.toml # Project build configuration
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
## Open-Source & Contributing
|
|
100
|
+
|
|
101
|
+
Kawi-TTS is an open-source research tool.
|
|
102
|
+
* **Evidence Before Code:** Research and evidence must precede any linguistic rule changes. Contributors must not invent pronunciation rules.
|
|
103
|
+
* **Changes:** Modifying linguistic policy requires cited evidence and corresponding regression coverage.
|
|
104
|
+
* **Boundaries:** Backend approximations must remain strictly backend-scoped (within the `AcousticMapper`). The canonical representation must remain protected.
|
|
105
|
+
|
|
106
|
+
Please see the `docs/` folder for architectural decisions and governance.
|
|
107
|
+
|
|
108
|
+
## Kawi Learn Relationship
|
|
109
|
+
|
|
110
|
+
**Kawi Learn** is considered a future, external educational application that may consume Kawi-TTS.
|
|
111
|
+
* **Kawi-TTS:** A strict pronunciation and reconstruction engine.
|
|
112
|
+
* **Kawi Learn:** A future end-user application.
|
|
113
|
+
|
|
114
|
+
Kawi-TTS must remain entirely independent of Kawi Learn. Kawi Learn does not currently exist as a finished product.
|
|
115
|
+
|
|
116
|
+
## Future / Deferred Work
|
|
117
|
+
|
|
118
|
+
The following items are explicitly deferred to protect the zero-budget constraint:
|
|
119
|
+
* Neural TTS models
|
|
120
|
+
* Custom recorded speech corpora
|
|
121
|
+
* New acoustic backends requiring unavailable computational resources
|
|
122
|
+
|
|
123
|
+
These will only be reopened if specific, resource-compatible packaging requirements are met.
|
|
124
|
+
|
|
125
|
+
## Research & Citation
|
|
126
|
+
|
|
127
|
+
Linguistic decisions, open questions, and principal references are maintained in `docs/RESEARCH_LOG.md` and the adjacent policy documents (e.g., `docs/P5_002_ACOUSTIC_PROFILE_POLICY.md`). All linguistic claims within the codebase are tied to these cited policy rules.
|
|
128
|
+
|
|
129
|
+
## License
|
|
130
|
+
|
|
131
|
+
Kawi-TTS is licensed under the MIT License. See the `LICENSE` file for details.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Acoustic mapping and backend synthesis interfaces for Kawi-TTS."""
|
|
2
|
+
|
|
3
|
+
from kawi_tts.acoustic.espeak_backend import (
|
|
4
|
+
ESpeakBackend,
|
|
5
|
+
ESpeakNotFoundError,
|
|
6
|
+
SynthesisResult,
|
|
7
|
+
create_minimal_wav_header,
|
|
8
|
+
)
|
|
9
|
+
from kawi_tts.acoustic.mapper import (
|
|
10
|
+
AcousticMapper,
|
|
11
|
+
AcousticMappingResult,
|
|
12
|
+
MappedToken,
|
|
13
|
+
MappingStatus,
|
|
14
|
+
)
|
|
15
|
+
from kawi_tts.acoustic.pipeline import PipelineResult, synthesize
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"AcousticMapper",
|
|
19
|
+
"AcousticMappingResult",
|
|
20
|
+
"MappedToken",
|
|
21
|
+
"MappingStatus",
|
|
22
|
+
"ESpeakBackend",
|
|
23
|
+
"ESpeakNotFoundError",
|
|
24
|
+
"SynthesisResult",
|
|
25
|
+
"create_minimal_wav_header",
|
|
26
|
+
"PipelineResult",
|
|
27
|
+
"synthesize",
|
|
28
|
+
]
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""eSpeak-ng acoustic backend interface for Kawi-TTS.
|
|
2
|
+
|
|
3
|
+
This module invokes eSpeak-ng via its phoneme/IPA input interface to synthesize
|
|
4
|
+
speech from mapped phonetic tokens. It provides mock/dry-run support when eSpeak-ng
|
|
5
|
+
is not installed on the host machine.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import shutil
|
|
11
|
+
import struct
|
|
12
|
+
import subprocess
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import List, Optional
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ESpeakNotFoundError(RuntimeError):
|
|
19
|
+
"""Raised when an attempt is made to synthesize real audio but eSpeak-ng is not installed."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class SynthesisResult:
|
|
24
|
+
"""Detailed result of an acoustic synthesis invocation.
|
|
25
|
+
|
|
26
|
+
Attributes:
|
|
27
|
+
phoneme_input: The phonetic/IPA string passed to the backend.
|
|
28
|
+
voice: The voice identifier requested (e.g. 'jv', 'id').
|
|
29
|
+
output_path: Target path for the output WAV file.
|
|
30
|
+
command: Full CLI argument list constructed for eSpeak-ng.
|
|
31
|
+
dry_run: Whether synthesis was executed in dry-run/mock mode.
|
|
32
|
+
audio_generated: Whether a valid WAV file was produced on disk or in memory.
|
|
33
|
+
bytes_written: Number of bytes written to output_path.
|
|
34
|
+
audio_bytes: Raw WAV bytes if output_path is None.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
phoneme_input: str
|
|
38
|
+
voice: str
|
|
39
|
+
output_path: Optional[str]
|
|
40
|
+
command: List[str]
|
|
41
|
+
dry_run: bool
|
|
42
|
+
audio_generated: bool
|
|
43
|
+
bytes_written: int = 0
|
|
44
|
+
audio_bytes: Optional[bytes] = None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def create_minimal_wav_header(sample_rate: int = 22050) -> bytes:
|
|
48
|
+
"""Generate a canonical 44-byte RIFF/WAVE header with 0 data samples for dry-run testing."""
|
|
49
|
+
num_channels = 1
|
|
50
|
+
bits_per_sample = 16
|
|
51
|
+
byte_rate = sample_rate * num_channels * bits_per_sample // 8
|
|
52
|
+
block_align = num_channels * bits_per_sample // 8
|
|
53
|
+
data_size = 0
|
|
54
|
+
riff_chunk_size = 36 + data_size
|
|
55
|
+
return struct.pack(
|
|
56
|
+
"<4sI4s4sIHHIIHH4sI",
|
|
57
|
+
b"RIFF",
|
|
58
|
+
riff_chunk_size,
|
|
59
|
+
b"WAVE",
|
|
60
|
+
b"fmt ",
|
|
61
|
+
16,
|
|
62
|
+
1, # PCM
|
|
63
|
+
num_channels,
|
|
64
|
+
sample_rate,
|
|
65
|
+
byte_rate,
|
|
66
|
+
block_align,
|
|
67
|
+
bits_per_sample,
|
|
68
|
+
b"data",
|
|
69
|
+
data_size,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class ESpeakBackend:
|
|
74
|
+
"""Wrapper around the eSpeak-ng command-line phoneme synthesis engine."""
|
|
75
|
+
|
|
76
|
+
_ESPEAK_ID_APPROXIMATION = {
|
|
77
|
+
"ə": "@", "əː": "@",
|
|
78
|
+
"ŋ": "N", "ɲ": "n^", "ɟ": "dZ",
|
|
79
|
+
"aː": "a", "iː": "i", "uː": "u",
|
|
80
|
+
"ʈ": "t", "ɖ": "d", "ɳ": "n",
|
|
81
|
+
"ʃ": "s", "ʂ": "s",
|
|
82
|
+
"tʰ": "th", "pʰ": "ph", "kʰ": "kh", "cʰ": "ch", "ʈʰ": "th",
|
|
83
|
+
"bʱ": "bh", "dʱ": "dh", "gʱ": "gh", "ɟʱ": "dZh", "ḍʱ": "dh",
|
|
84
|
+
"r̩": "r@", "l̩": "l@", "r̩ː": "r@", "l̩ː": "l@",
|
|
85
|
+
"rə": "r@", "lə": "l@",
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
def __init__(self, executable: Optional[str] = None, voice: str = "jv"):
|
|
89
|
+
"""Initialize the eSpeak backend interface.
|
|
90
|
+
|
|
91
|
+
Args:
|
|
92
|
+
executable: Custom path to espeak-ng/espeak binary. If None, discovers via PATH.
|
|
93
|
+
voice: Voice name or language code (default 'jv' for Javanese).
|
|
94
|
+
"""
|
|
95
|
+
self.voice = voice
|
|
96
|
+
self.executable_path = executable or self._find_executable()
|
|
97
|
+
|
|
98
|
+
@property
|
|
99
|
+
def is_available(self) -> bool:
|
|
100
|
+
"""Check whether a usable eSpeak-ng executable was located on the system."""
|
|
101
|
+
return self.executable_path is not None
|
|
102
|
+
|
|
103
|
+
@staticmethod
|
|
104
|
+
def _find_executable() -> Optional[str]:
|
|
105
|
+
"""Discover espeak-ng or espeak in system PATH."""
|
|
106
|
+
return shutil.which("espeak-ng") or shutil.which("espeak")
|
|
107
|
+
|
|
108
|
+
def synthesize(
|
|
109
|
+
self,
|
|
110
|
+
phoneme_str: str,
|
|
111
|
+
output_path: Optional[str] = None,
|
|
112
|
+
*,
|
|
113
|
+
dry_run: bool = False,
|
|
114
|
+
create_dummy_wav: bool = False,
|
|
115
|
+
) -> SynthesisResult:
|
|
116
|
+
"""Synthesize a phonetic/IPA string to audio.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
phoneme_str: Space-separated backend-ready phoneme string.
|
|
120
|
+
output_path: File path to save output WAV file. If None, audio is returned in memory.
|
|
121
|
+
dry_run: If True, simulates execution without invoking the real binary.
|
|
122
|
+
create_dummy_wav: If True in dry_run mode, writes a minimal 44-byte WAV header
|
|
123
|
+
to output_path to test file generation workflows.
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
SynthesisResult object documenting the invocation.
|
|
127
|
+
|
|
128
|
+
Raises:
|
|
129
|
+
ESpeakNotFoundError: If dry_run is False but eSpeak-ng is not installed.
|
|
130
|
+
subprocess.CalledProcessError: If eSpeak-ng execution fails.
|
|
131
|
+
"""
|
|
132
|
+
target_path = str(output_path) if output_path else None
|
|
133
|
+
|
|
134
|
+
# Apply backend-specific phonetic approximations if using Indonesian voice
|
|
135
|
+
final_phoneme_str = phoneme_str
|
|
136
|
+
if self.voice == "id":
|
|
137
|
+
# Sort replacements by length descending to prevent substring collisions
|
|
138
|
+
sorted_replacements = sorted(self._ESPEAK_ID_APPROXIMATION.items(), key=lambda x: len(x[0]), reverse=True)
|
|
139
|
+
for k, v in sorted_replacements:
|
|
140
|
+
final_phoneme_str = final_phoneme_str.replace(k, v)
|
|
141
|
+
|
|
142
|
+
# Format input phoneme string inside eSpeak [[...]] brackets
|
|
143
|
+
phoneme_bracketed = f"[[{final_phoneme_str}]]"
|
|
144
|
+
|
|
145
|
+
exec_cmd = [
|
|
146
|
+
self.executable_path or "espeak-ng",
|
|
147
|
+
"-v",
|
|
148
|
+
self.voice,
|
|
149
|
+
]
|
|
150
|
+
if target_path:
|
|
151
|
+
exec_cmd.extend(["-w", target_path])
|
|
152
|
+
else:
|
|
153
|
+
exec_cmd.append("--stdout")
|
|
154
|
+
|
|
155
|
+
exec_cmd.append(phoneme_bracketed)
|
|
156
|
+
|
|
157
|
+
if dry_run:
|
|
158
|
+
bytes_written = 0
|
|
159
|
+
wav_bytes = create_minimal_wav_header() if create_dummy_wav else None
|
|
160
|
+
if target_path and wav_bytes:
|
|
161
|
+
Path(target_path).parent.mkdir(parents=True, exist_ok=True)
|
|
162
|
+
with open(target_path, "wb") as f:
|
|
163
|
+
f.write(wav_bytes)
|
|
164
|
+
bytes_written = len(wav_bytes)
|
|
165
|
+
|
|
166
|
+
return SynthesisResult(
|
|
167
|
+
phoneme_input=final_phoneme_str,
|
|
168
|
+
voice=self.voice,
|
|
169
|
+
output_path=target_path,
|
|
170
|
+
command=exec_cmd,
|
|
171
|
+
dry_run=True,
|
|
172
|
+
audio_generated=(create_dummy_wav),
|
|
173
|
+
bytes_written=bytes_written,
|
|
174
|
+
audio_bytes=wav_bytes if not target_path else None,
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
# Real execution requested
|
|
178
|
+
if not self.is_available:
|
|
179
|
+
raise ESpeakNotFoundError(
|
|
180
|
+
"eSpeak-ng executable not found in system PATH. To generate real audio:\n"
|
|
181
|
+
" 1. Install eSpeak-ng (e.g., via 'winget install eSpeak-ng' on Windows, "
|
|
182
|
+
"'apt install espeak-ng' on Debian/Ubuntu, or from https://github.com/espeak-ng/espeak-ng/releases).\n"
|
|
183
|
+
" 2. Or instantiate ESpeakBackend(executable='/path/to/espeak-ng').\n"
|
|
184
|
+
"Alternatively, specify dry_run=True to verify pipeline data flow without generating audio."
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
if target_path:
|
|
188
|
+
Path(target_path).parent.mkdir(parents=True, exist_ok=True)
|
|
189
|
+
|
|
190
|
+
result = subprocess.run(
|
|
191
|
+
exec_cmd,
|
|
192
|
+
check=True,
|
|
193
|
+
capture_output=True,
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
audio_bytes = None
|
|
197
|
+
file_size = 0
|
|
198
|
+
if target_path:
|
|
199
|
+
file_size = Path(target_path).stat().st_size if Path(target_path).exists() else 0
|
|
200
|
+
else:
|
|
201
|
+
audio_bytes = result.stdout
|
|
202
|
+
file_size = len(audio_bytes)
|
|
203
|
+
|
|
204
|
+
return SynthesisResult(
|
|
205
|
+
phoneme_input=final_phoneme_str,
|
|
206
|
+
voice=self.voice,
|
|
207
|
+
output_path=target_path,
|
|
208
|
+
command=exec_cmd,
|
|
209
|
+
dry_run=False,
|
|
210
|
+
audio_generated=(file_size > 0),
|
|
211
|
+
bytes_written=file_size,
|
|
212
|
+
audio_bytes=audio_bytes,
|
|
213
|
+
)
|