kawi-tts 1.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. kawi_tts-1.1.2/LICENSE +21 -0
  2. kawi_tts-1.1.2/PKG-INFO +164 -0
  3. kawi_tts-1.1.2/README.md +131 -0
  4. kawi_tts-1.1.2/kawi_tts/__init__.py +3 -0
  5. kawi_tts-1.1.2/kawi_tts/acoustic/__init__.py +28 -0
  6. kawi_tts-1.1.2/kawi_tts/acoustic/espeak_backend.py +213 -0
  7. kawi_tts-1.1.2/kawi_tts/acoustic/mapper.py +152 -0
  8. kawi_tts-1.1.2/kawi_tts/acoustic/pipeline.py +119 -0
  9. kawi_tts-1.1.2/kawi_tts/acoustic/strategies/__init__.py +10 -0
  10. kawi_tts-1.1.2/kawi_tts/acoustic/strategies/base.py +35 -0
  11. kawi_tts-1.1.2/kawi_tts/acoustic/strategies/profile_a.py +123 -0
  12. kawi_tts-1.1.2/kawi_tts/acoustic/strategies/profile_b.py +67 -0
  13. kawi_tts-1.1.2/kawi_tts/cli/__init__.py +0 -0
  14. kawi_tts-1.1.2/kawi_tts/cli/coverage_report.py +55 -0
  15. kawi_tts-1.1.2/kawi_tts/cli/kawi_trace.py +90 -0
  16. kawi_tts-1.1.2/kawi_tts/evaluation/__init__.py +5 -0
  17. kawi_tts-1.1.2/kawi_tts/evaluation/evaluate_ojw.py +343 -0
  18. kawi_tts-1.1.2/kawi_tts/g2p/__init__.py +10 -0
  19. kawi_tts-1.1.2/kawi_tts/g2p/engine.py +143 -0
  20. kawi_tts-1.1.2/kawi_tts/normalization/__init__.py +31 -0
  21. kawi_tts-1.1.2/kawi_tts/normalization/normalizer.py +246 -0
  22. kawi_tts-1.1.2/kawi_tts/normalization/tokenizer.py +210 -0
  23. kawi_tts-1.1.2/kawi_tts/tts/__init__.py +5 -0
  24. kawi_tts-1.1.2/kawi_tts/tts/experimental_piper.py +55 -0
  25. kawi_tts-1.1.2/kawi_tts.egg-info/PKG-INFO +164 -0
  26. kawi_tts-1.1.2/kawi_tts.egg-info/SOURCES.txt +44 -0
  27. kawi_tts-1.1.2/kawi_tts.egg-info/dependency_links.txt +1 -0
  28. kawi_tts-1.1.2/kawi_tts.egg-info/entry_points.txt +2 -0
  29. kawi_tts-1.1.2/kawi_tts.egg-info/requires.txt +4 -0
  30. kawi_tts-1.1.2/kawi_tts.egg-info/top_level.txt +1 -0
  31. kawi_tts-1.1.2/pyproject.toml +62 -0
  32. kawi_tts-1.1.2/setup.cfg +4 -0
  33. kawi_tts-1.1.2/tests/test_acoustic_mapper.py +115 -0
  34. kawi_tts-1.1.2/tests/test_acoustic_mapper_profile_a.py +74 -0
  35. kawi_tts-1.1.2/tests/test_espeak_backend.py +30 -0
  36. kawi_tts-1.1.2/tests/test_evaluation.py +71 -0
  37. kawi_tts-1.1.2/tests/test_g2p.py +109 -0
  38. kawi_tts-1.1.2/tests/test_g2p_bugfix.py +33 -0
  39. kawi_tts-1.1.2/tests/test_integration.py +220 -0
  40. kawi_tts-1.1.2/tests/test_normalization.py +241 -0
  41. kawi_tts-1.1.2/tests/test_profile_strategy.py +88 -0
  42. kawi_tts-1.1.2/tests/test_regression_corpus.py +73 -0
  43. kawi_tts-1.1.2/tests/test_scaffold.py +19 -0
  44. kawi_tts-1.1.2/tests/test_synthesis_pipeline.py +117 -0
  45. kawi_tts-1.1.2/tests/test_tokenizer.py +213 -0
  46. kawi_tts-1.1.2/tests/test_validation.py +216 -0
kawi_tts-1.1.2/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Kawi-TTS Contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,164 @@
1
+ Metadata-Version: 2.4
2
+ Name: kawi-tts
3
+ Version: 1.1.2
4
+ Summary: Deterministic pronunciation and reconstruction engine for Old Javanese/Kawi.
5
+ Author: Project-TTS-Kawi
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/krispypasta/kawi-tts
8
+ Project-URL: Repository, https://github.com/krispypasta/kawi-tts
9
+ Project-URL: Bug Tracker, https://github.com/krispypasta/kawi-tts/issues
10
+ Project-URL: Documentation, https://github.com/krispypasta/kawi-tts#readme
11
+ Keywords: kawi,old-javanese,linguistics,phonology,g2p,text-to-speech,reconstruction,austronesian
12
+ Classifier: Development Status :: 5 - Production/Stable
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.8
17
+ Classifier: Programming Language :: Python :: 3.9
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
23
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
24
+ Classifier: Topic :: Text Processing :: Linguistic
25
+ Classifier: Operating System :: OS Independent
26
+ Requires-Python: >=3.8
27
+ Description-Content-Type: text/markdown
28
+ License-File: LICENSE
29
+ Provides-Extra: dev
30
+ Requires-Dist: pytest; extra == "dev"
31
+ Requires-Dist: pytest-cov; extra == "dev"
32
+ Dynamic: license-file
33
+
34
+ # Kawi-TTS
35
+
36
+ An open-source deterministic pronunciation and reconstruction engine for Old Javanese (Kawi).
37
+
38
+ Kawi-TTS is **not** a natural neural TTS system, it does **not** claim to generate historically authentic speech, and it is **not** a linguistic oracle. It is a strictly deterministic pipeline that implements a documented linguistic reconstruction policy and provides reproducible acoustic output through a selected backend.
39
+
40
+ ## What It Does
41
+
42
+ Kawi-TTS transforms Old Javanese text into phonetic representations and synthesizes them into audio. The pipeline is strictly deterministic:
43
+
44
+ `Text → Normalization → G2P / Tokenization → Canonical Representation → Profile A / Profile B → Acoustic Mapping → eSpeak-ng → WAV`
45
+
46
+ The engine resolves linguistic features through two distinct profiles:
47
+ * **Profile A (Reconstructed Spoken):** Represents the historical spoken target. It is explicitly evidence-scoped, includes documented spoken mergers and adaptations, and strictly preserves unresolved items as unresolved where evidence is lacking.
48
+ * **Profile B (Scholarly/Orthographic):** A scholarly reading profile that artificially preserves the intended distinctions of the scholarly orthographic representation (such as unmerged Sanskrit distinctions).
49
+
50
+ ## Important Epistemic Limitation
51
+
52
+ **The generated audio is NOT claimed to reproduce exactly how historical Old Javanese speakers sounded.**
53
+
54
+ The software simply operationalizes the project's documented linguistic reconstruction and backend policy. Users must clearly distinguish between:
55
+ 1. **Evidence-backed linguistic rules** (e.g., historical spoken mergers)
56
+ 2. **Provisional reconstructions** (e.g., adaptation of Sanskrit vocalic liquids to native Javanese phonotactics)
57
+ 3. **Engineering policy** (e.g., deferring vowel duration resolution)
58
+ 4. **Backend approximations** (e.g., dropping retroflex distinctions due to eSpeak limitations)
59
+
60
+ ## Current Status
61
+
62
+ * **Core:** The deterministic core is accepted and historically frozen.
63
+ * **Tests:** The current test suite passes 97/97. The 100-form regression corpus passes fully.
64
+ * **Backend:** eSpeak-ng is the current acoustic backend. The `id` (Indonesian) voice is utilized with explicit backend-scoped approximations where native eSpeak support is lacking.
65
+ * **Infrastructure:** Neural TTS is explicitly deferred. The project remains strictly zero-budget compatible. No cloud service or proprietary data is required for core operation.
66
+ * **Distribution:** The package is not yet published to PyPI.
67
+
68
+ ## Quick Start
69
+
70
+ You can run the engine locally without any cloud dependencies.
71
+
72
+ ```bash
73
+ # 1. Clone the repository
74
+ git clone https://github.com/krispypasta/kawi-tts.git
75
+ cd kawi-tts
76
+
77
+ # 2. Create and activate a Python virtual environment
78
+ python -m venv venv
79
+ # On Windows: venv\Scripts\activate
80
+ # On Linux/macOS: source venv/bin/activate
81
+
82
+ # 3. Install the package locally in editable mode
83
+ pip install -e .
84
+
85
+ # 4. Verify installation by running tests
86
+ python -m unittest discover -s tests -p "test_*.py"
87
+
88
+ # 5. Use the trace diagnostics CLI
89
+ kawi-trace "sĕkar"
90
+
91
+ # 6. Generate demo audio (requires eSpeak-ng installed on your system)
92
+ python run_audio_demo.py
93
+ ```
94
+
95
+ ## eSpeak Dependency
96
+
97
+ The deterministic linguistic processing (Normalization, G2P, Profiling) operates entirely independent of eSpeak. However, actual WAV synthesis requires `espeak-ng` to be installed on your system path.
98
+
99
+ If eSpeak-ng is missing, the pipeline produces an actionable error. The package will **not** silently pretend that dummy audio is real synthesis. Current acoustic mapping supports the eSpeak `id` voice, utilizing explicit `BACKEND_APPROXIMATION` rules to bypass unsupported IPA characters safely.
100
+
101
+ ## Examples
102
+
103
+ * `sĕkar` — Standard Javanese lexical item. Both profiles map identically to Javanese schwa and native consonants.
104
+ * `BHAṬĀRA` — Sanskrit loan.
105
+ * *Profile B* preserves the aspirated `bʱ`, retroflex `ṭ`, and duration `aː`.
106
+ * *Profile A* merges the aspirate to `b`, maps the retroflex to `ṭ`, and leaves duration unresolved.
107
+ * *Acoustic Mapper* then approximates retroflex `ṭ` to dental `t` for eSpeak compatibility.
108
+ * `śānti` — Sanskrit loan.
109
+ * *Profile B* preserves palatal sibilant `ś`.
110
+ * *Profile A* merges it to native `s`.
111
+ * `kṝta` — Sanskrit loan with a long vocalic liquid.
112
+ * *Profile B* preserves canonical long `r̩ː`.
113
+ * *Profile A* adapts it to native phonotactics as a short Javanese schwa base (`rə`), deliberately discarding non-native duration as per reconstruction policy.
114
+
115
+ ## Trace / Diagnostics
116
+
117
+ The `kawi-trace` CLI utility exposes the exact transformation of every string. It traces:
118
+ `Input → Normalized → Canonical → Profile Target → Backend Target`
119
+
120
+ Backend approximations (e.g., eSpeak's inability to pronounce retroflex consonants) are visibly distinguished in the terminal output from intentional linguistic targets.
121
+
122
+ ## Project Structure
123
+
124
+ ```text
125
+ kawi_tts/ # Core engine (normalization, g2p, acoustic mappers)
126
+ docs/ # Project documentation, policies, and research logs
127
+ tests/ # Test suite and regression corpora
128
+ artifacts/ # Generated WAVs and manifest outputs
129
+ pyproject.toml # Project build configuration
130
+ ```
131
+
132
+ ## Open-Source & Contributing
133
+
134
+ Kawi-TTS is an open-source research tool.
135
+ * **Evidence Before Code:** Research and evidence must precede any linguistic rule changes. Contributors must not invent pronunciation rules.
136
+ * **Changes:** Modifying linguistic policy requires cited evidence and corresponding regression coverage.
137
+ * **Boundaries:** Backend approximations must remain strictly backend-scoped (within the `AcousticMapper`). The canonical representation must remain protected.
138
+
139
+ Please see the `docs/` folder for architectural decisions and governance.
140
+
141
+ ## Kawi Learn Relationship
142
+
143
+ **Kawi Learn** is considered a future, external educational application that may consume Kawi-TTS.
144
+ * **Kawi-TTS:** A strict pronunciation and reconstruction engine.
145
+ * **Kawi Learn:** A future end-user application.
146
+
147
+ Kawi-TTS must remain entirely independent of Kawi Learn. Kawi Learn does not currently exist as a finished product.
148
+
149
+ ## Future / Deferred Work
150
+
151
+ The following items are explicitly deferred to protect the zero-budget constraint:
152
+ * Neural TTS models
153
+ * Custom recorded speech corpora
154
+ * New acoustic backends requiring unavailable computational resources
155
+
156
+ These will only be reopened if specific, resource-compatible packaging requirements are met.
157
+
158
+ ## Research & Citation
159
+
160
+ Linguistic decisions, open questions, and principal references are maintained in `docs/RESEARCH_LOG.md` and the adjacent policy documents (e.g., `docs/P5_002_ACOUSTIC_PROFILE_POLICY.md`). All linguistic claims within the codebase are tied to these cited policy rules.
161
+
162
+ ## License
163
+
164
+ Kawi-TTS is licensed under the MIT License. See the `LICENSE` file for details.
@@ -0,0 +1,131 @@
1
+ # Kawi-TTS
2
+
3
+ An open-source deterministic pronunciation and reconstruction engine for Old Javanese (Kawi).
4
+
5
+ Kawi-TTS is **not** a natural neural TTS system, it does **not** claim to generate historically authentic speech, and it is **not** a linguistic oracle. It is a strictly deterministic pipeline that implements a documented linguistic reconstruction policy and provides reproducible acoustic output through a selected backend.
6
+
7
+ ## What It Does
8
+
9
+ Kawi-TTS transforms Old Javanese text into phonetic representations and synthesizes them into audio. The pipeline is strictly deterministic:
10
+
11
+ `Text → Normalization → G2P / Tokenization → Canonical Representation → Profile A / Profile B → Acoustic Mapping → eSpeak-ng → WAV`
12
+
13
+ The engine resolves linguistic features through two distinct profiles:
14
+ * **Profile A (Reconstructed Spoken):** Represents the historical spoken target. It is explicitly evidence-scoped, includes documented spoken mergers and adaptations, and strictly preserves unresolved items as unresolved where evidence is lacking.
15
+ * **Profile B (Scholarly/Orthographic):** A scholarly reading profile that artificially preserves the intended distinctions of the scholarly orthographic representation (such as unmerged Sanskrit distinctions).
16
+
17
+ ## Important Epistemic Limitation
18
+
19
+ **The generated audio is NOT claimed to reproduce exactly how historical Old Javanese speakers sounded.**
20
+
21
+ The software simply operationalizes the project's documented linguistic reconstruction and backend policy. Users must clearly distinguish between:
22
+ 1. **Evidence-backed linguistic rules** (e.g., historical spoken mergers)
23
+ 2. **Provisional reconstructions** (e.g., adaptation of Sanskrit vocalic liquids to native Javanese phonotactics)
24
+ 3. **Engineering policy** (e.g., deferring vowel duration resolution)
25
+ 4. **Backend approximations** (e.g., dropping retroflex distinctions due to eSpeak limitations)
26
+
27
+ ## Current Status
28
+
29
+ * **Core:** The deterministic core is accepted and historically frozen.
30
+ * **Tests:** The current test suite passes 97/97. The 100-form regression corpus passes fully.
31
+ * **Backend:** eSpeak-ng is the current acoustic backend. The `id` (Indonesian) voice is utilized with explicit backend-scoped approximations where native eSpeak support is lacking.
32
+ * **Infrastructure:** Neural TTS is explicitly deferred. The project remains strictly zero-budget compatible. No cloud service or proprietary data is required for core operation.
33
+ * **Distribution:** The package is not yet published to PyPI.
34
+
35
+ ## Quick Start
36
+
37
+ You can run the engine locally without any cloud dependencies.
38
+
39
+ ```bash
40
+ # 1. Clone the repository
41
+ git clone https://github.com/krispypasta/kawi-tts.git
42
+ cd kawi-tts
43
+
44
+ # 2. Create and activate a Python virtual environment
45
+ python -m venv venv
46
+ # On Windows: venv\Scripts\activate
47
+ # On Linux/macOS: source venv/bin/activate
48
+
49
+ # 3. Install the package locally in editable mode
50
+ pip install -e .
51
+
52
+ # 4. Verify installation by running tests
53
+ python -m unittest discover -s tests -p "test_*.py"
54
+
55
+ # 5. Use the trace diagnostics CLI
56
+ kawi-trace "sĕkar"
57
+
58
+ # 6. Generate demo audio (requires eSpeak-ng installed on your system)
59
+ python run_audio_demo.py
60
+ ```
61
+
62
+ ## eSpeak Dependency
63
+
64
+ The deterministic linguistic processing (Normalization, G2P, Profiling) operates entirely independent of eSpeak. However, actual WAV synthesis requires `espeak-ng` to be installed on your system path.
65
+
66
+ If eSpeak-ng is missing, the pipeline produces an actionable error. The package will **not** silently pretend that dummy audio is real synthesis. Current acoustic mapping supports the eSpeak `id` voice, utilizing explicit `BACKEND_APPROXIMATION` rules to bypass unsupported IPA characters safely.
67
+
68
+ ## Examples
69
+
70
+ * `sĕkar` — Standard Javanese lexical item. Both profiles map identically to Javanese schwa and native consonants.
71
+ * `BHAṬĀRA` — Sanskrit loan.
72
+ * *Profile B* preserves the aspirated `bʱ`, retroflex `ṭ`, and duration `aː`.
73
+ * *Profile A* merges the aspirate to `b`, maps the retroflex to `ṭ`, and leaves duration unresolved.
74
+ * *Acoustic Mapper* then approximates retroflex `ṭ` to dental `t` for eSpeak compatibility.
75
+ * `śānti` — Sanskrit loan.
76
+ * *Profile B* preserves palatal sibilant `ś`.
77
+ * *Profile A* merges it to native `s`.
78
+ * `kṝta` — Sanskrit loan with a long vocalic liquid.
79
+ * *Profile B* preserves canonical long `r̩ː`.
80
+ * *Profile A* adapts it to native phonotactics as a short Javanese schwa base (`rə`), deliberately discarding non-native duration as per reconstruction policy.
81
+
82
+ ## Trace / Diagnostics
83
+
84
+ The `kawi-trace` CLI utility exposes the exact transformation of every string. It traces:
85
+ `Input → Normalized → Canonical → Profile Target → Backend Target`
86
+
87
+ Backend approximations (e.g., eSpeak's inability to pronounce retroflex consonants) are visibly distinguished in the terminal output from intentional linguistic targets.
88
+
89
+ ## Project Structure
90
+
91
+ ```text
92
+ kawi_tts/ # Core engine (normalization, g2p, acoustic mappers)
93
+ docs/ # Project documentation, policies, and research logs
94
+ tests/ # Test suite and regression corpora
95
+ artifacts/ # Generated WAVs and manifest outputs
96
+ pyproject.toml # Project build configuration
97
+ ```
98
+
99
+ ## Open-Source & Contributing
100
+
101
+ Kawi-TTS is an open-source research tool.
102
+ * **Evidence Before Code:** Research and evidence must precede any linguistic rule changes. Contributors must not invent pronunciation rules.
103
+ * **Changes:** Modifying linguistic policy requires cited evidence and corresponding regression coverage.
104
+ * **Boundaries:** Backend approximations must remain strictly backend-scoped (within the `AcousticMapper`). The canonical representation must remain protected.
105
+
106
+ Please see the `docs/` folder for architectural decisions and governance.
107
+
108
+ ## Kawi Learn Relationship
109
+
110
+ **Kawi Learn** is considered a future, external educational application that may consume Kawi-TTS.
111
+ * **Kawi-TTS:** A strict pronunciation and reconstruction engine.
112
+ * **Kawi Learn:** A future end-user application.
113
+
114
+ Kawi-TTS must remain entirely independent of Kawi Learn. Kawi Learn does not currently exist as a finished product.
115
+
116
+ ## Future / Deferred Work
117
+
118
+ The following items are explicitly deferred to protect the zero-budget constraint:
119
+ * Neural TTS models
120
+ * Custom recorded speech corpora
121
+ * New acoustic backends requiring unavailable computational resources
122
+
123
+ These will only be reopened if specific, resource-compatible packaging requirements are met.
124
+
125
+ ## Research & Citation
126
+
127
+ Linguistic decisions, open questions, and principal references are maintained in `docs/RESEARCH_LOG.md` and the adjacent policy documents (e.g., `docs/P5_002_ACOUSTIC_PROFILE_POLICY.md`). All linguistic claims within the codebase are tied to these cited policy rules.
128
+
129
+ ## License
130
+
131
+ Kawi-TTS is licensed under the MIT License. See the `LICENSE` file for details.
@@ -0,0 +1,3 @@
1
+ """Kawi-TTS: Research-grade Text-to-Speech system for Old Javanese."""
2
+
3
+ __version__ = "1.1.2"
@@ -0,0 +1,28 @@
1
+ """Acoustic mapping and backend synthesis interfaces for Kawi-TTS."""
2
+
3
+ from kawi_tts.acoustic.espeak_backend import (
4
+ ESpeakBackend,
5
+ ESpeakNotFoundError,
6
+ SynthesisResult,
7
+ create_minimal_wav_header,
8
+ )
9
+ from kawi_tts.acoustic.mapper import (
10
+ AcousticMapper,
11
+ AcousticMappingResult,
12
+ MappedToken,
13
+ MappingStatus,
14
+ )
15
+ from kawi_tts.acoustic.pipeline import PipelineResult, synthesize
16
+
17
+ __all__ = [
18
+ "AcousticMapper",
19
+ "AcousticMappingResult",
20
+ "MappedToken",
21
+ "MappingStatus",
22
+ "ESpeakBackend",
23
+ "ESpeakNotFoundError",
24
+ "SynthesisResult",
25
+ "create_minimal_wav_header",
26
+ "PipelineResult",
27
+ "synthesize",
28
+ ]
@@ -0,0 +1,213 @@
1
+ """eSpeak-ng acoustic backend interface for Kawi-TTS.
2
+
3
+ This module invokes eSpeak-ng via its phoneme/IPA input interface to synthesize
4
+ speech from mapped phonetic tokens. It provides mock/dry-run support when eSpeak-ng
5
+ is not installed on the host machine.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import shutil
11
+ import struct
12
+ import subprocess
13
+ from dataclasses import dataclass, field
14
+ from pathlib import Path
15
+ from typing import List, Optional
16
+
17
+
18
+ class ESpeakNotFoundError(RuntimeError):
19
+ """Raised when an attempt is made to synthesize real audio but eSpeak-ng is not installed."""
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class SynthesisResult:
24
+ """Detailed result of an acoustic synthesis invocation.
25
+
26
+ Attributes:
27
+ phoneme_input: The phonetic/IPA string passed to the backend.
28
+ voice: The voice identifier requested (e.g. 'jv', 'id').
29
+ output_path: Target path for the output WAV file.
30
+ command: Full CLI argument list constructed for eSpeak-ng.
31
+ dry_run: Whether synthesis was executed in dry-run/mock mode.
32
+ audio_generated: Whether a valid WAV file was produced on disk or in memory.
33
+ bytes_written: Number of bytes written to output_path.
34
+ audio_bytes: Raw WAV bytes if output_path is None.
35
+ """
36
+
37
+ phoneme_input: str
38
+ voice: str
39
+ output_path: Optional[str]
40
+ command: List[str]
41
+ dry_run: bool
42
+ audio_generated: bool
43
+ bytes_written: int = 0
44
+ audio_bytes: Optional[bytes] = None
45
+
46
+
47
+ def create_minimal_wav_header(sample_rate: int = 22050) -> bytes:
48
+ """Generate a canonical 44-byte RIFF/WAVE header with 0 data samples for dry-run testing."""
49
+ num_channels = 1
50
+ bits_per_sample = 16
51
+ byte_rate = sample_rate * num_channels * bits_per_sample // 8
52
+ block_align = num_channels * bits_per_sample // 8
53
+ data_size = 0
54
+ riff_chunk_size = 36 + data_size
55
+ return struct.pack(
56
+ "<4sI4s4sIHHIIHH4sI",
57
+ b"RIFF",
58
+ riff_chunk_size,
59
+ b"WAVE",
60
+ b"fmt ",
61
+ 16,
62
+ 1, # PCM
63
+ num_channels,
64
+ sample_rate,
65
+ byte_rate,
66
+ block_align,
67
+ bits_per_sample,
68
+ b"data",
69
+ data_size,
70
+ )
71
+
72
+
73
+ class ESpeakBackend:
74
+ """Wrapper around the eSpeak-ng command-line phoneme synthesis engine."""
75
+
76
+ _ESPEAK_ID_APPROXIMATION = {
77
+ "ə": "@", "əː": "@",
78
+ "ŋ": "N", "ɲ": "n^", "ɟ": "dZ",
79
+ "aː": "a", "iː": "i", "uː": "u",
80
+ "ʈ": "t", "ɖ": "d", "ɳ": "n",
81
+ "ʃ": "s", "ʂ": "s",
82
+ "tʰ": "th", "pʰ": "ph", "kʰ": "kh", "cʰ": "ch", "ʈʰ": "th",
83
+ "bʱ": "bh", "dʱ": "dh", "gʱ": "gh", "ɟʱ": "dZh", "ḍʱ": "dh",
84
+ "r̩": "r@", "l̩": "l@", "r̩ː": "r@", "l̩ː": "l@",
85
+ "rə": "r@", "lə": "l@",
86
+ }
87
+
88
+ def __init__(self, executable: Optional[str] = None, voice: str = "jv"):
89
+ """Initialize the eSpeak backend interface.
90
+
91
+ Args:
92
+ executable: Custom path to espeak-ng/espeak binary. If None, discovers via PATH.
93
+ voice: Voice name or language code (default 'jv' for Javanese).
94
+ """
95
+ self.voice = voice
96
+ self.executable_path = executable or self._find_executable()
97
+
98
+ @property
99
+ def is_available(self) -> bool:
100
+ """Check whether a usable eSpeak-ng executable was located on the system."""
101
+ return self.executable_path is not None
102
+
103
+ @staticmethod
104
+ def _find_executable() -> Optional[str]:
105
+ """Discover espeak-ng or espeak in system PATH."""
106
+ return shutil.which("espeak-ng") or shutil.which("espeak")
107
+
108
+ def synthesize(
109
+ self,
110
+ phoneme_str: str,
111
+ output_path: Optional[str] = None,
112
+ *,
113
+ dry_run: bool = False,
114
+ create_dummy_wav: bool = False,
115
+ ) -> SynthesisResult:
116
+ """Synthesize a phonetic/IPA string to audio.
117
+
118
+ Args:
119
+ phoneme_str: Space-separated backend-ready phoneme string.
120
+ output_path: File path to save output WAV file. If None, audio is returned in memory.
121
+ dry_run: If True, simulates execution without invoking the real binary.
122
+ create_dummy_wav: If True in dry_run mode, writes a minimal 44-byte WAV header
123
+ to output_path to test file generation workflows.
124
+
125
+ Returns:
126
+ SynthesisResult object documenting the invocation.
127
+
128
+ Raises:
129
+ ESpeakNotFoundError: If dry_run is False but eSpeak-ng is not installed.
130
+ subprocess.CalledProcessError: If eSpeak-ng execution fails.
131
+ """
132
+ target_path = str(output_path) if output_path else None
133
+
134
+ # Apply backend-specific phonetic approximations if using Indonesian voice
135
+ final_phoneme_str = phoneme_str
136
+ if self.voice == "id":
137
+ # Sort replacements by length descending to prevent substring collisions
138
+ sorted_replacements = sorted(self._ESPEAK_ID_APPROXIMATION.items(), key=lambda x: len(x[0]), reverse=True)
139
+ for k, v in sorted_replacements:
140
+ final_phoneme_str = final_phoneme_str.replace(k, v)
141
+
142
+ # Format input phoneme string inside eSpeak [[...]] brackets
143
+ phoneme_bracketed = f"[[{final_phoneme_str}]]"
144
+
145
+ exec_cmd = [
146
+ self.executable_path or "espeak-ng",
147
+ "-v",
148
+ self.voice,
149
+ ]
150
+ if target_path:
151
+ exec_cmd.extend(["-w", target_path])
152
+ else:
153
+ exec_cmd.append("--stdout")
154
+
155
+ exec_cmd.append(phoneme_bracketed)
156
+
157
+ if dry_run:
158
+ bytes_written = 0
159
+ wav_bytes = create_minimal_wav_header() if create_dummy_wav else None
160
+ if target_path and wav_bytes:
161
+ Path(target_path).parent.mkdir(parents=True, exist_ok=True)
162
+ with open(target_path, "wb") as f:
163
+ f.write(wav_bytes)
164
+ bytes_written = len(wav_bytes)
165
+
166
+ return SynthesisResult(
167
+ phoneme_input=final_phoneme_str,
168
+ voice=self.voice,
169
+ output_path=target_path,
170
+ command=exec_cmd,
171
+ dry_run=True,
172
+ audio_generated=(create_dummy_wav),
173
+ bytes_written=bytes_written,
174
+ audio_bytes=wav_bytes if not target_path else None,
175
+ )
176
+
177
+ # Real execution requested
178
+ if not self.is_available:
179
+ raise ESpeakNotFoundError(
180
+ "eSpeak-ng executable not found in system PATH. To generate real audio:\n"
181
+ " 1. Install eSpeak-ng (e.g., via 'winget install eSpeak-ng' on Windows, "
182
+ "'apt install espeak-ng' on Debian/Ubuntu, or from https://github.com/espeak-ng/espeak-ng/releases).\n"
183
+ " 2. Or instantiate ESpeakBackend(executable='/path/to/espeak-ng').\n"
184
+ "Alternatively, specify dry_run=True to verify pipeline data flow without generating audio."
185
+ )
186
+
187
+ if target_path:
188
+ Path(target_path).parent.mkdir(parents=True, exist_ok=True)
189
+
190
+ result = subprocess.run(
191
+ exec_cmd,
192
+ check=True,
193
+ capture_output=True,
194
+ )
195
+
196
+ audio_bytes = None
197
+ file_size = 0
198
+ if target_path:
199
+ file_size = Path(target_path).stat().st_size if Path(target_path).exists() else 0
200
+ else:
201
+ audio_bytes = result.stdout
202
+ file_size = len(audio_bytes)
203
+
204
+ return SynthesisResult(
205
+ phoneme_input=final_phoneme_str,
206
+ voice=self.voice,
207
+ output_path=target_path,
208
+ command=exec_cmd,
209
+ dry_run=False,
210
+ audio_generated=(file_size > 0),
211
+ bytes_written=file_size,
212
+ audio_bytes=audio_bytes,
213
+ )