scriptconv 0.0.4a10__tar.gz → 0.0.4a13__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scriptconv-0.0.4a10/scriptconv.egg-info → scriptconv-0.0.4a13}/PKG-INFO +167 -151
- scriptconv-0.0.4a13/README.md +390 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/base.py +19 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/ja.py +7 -5
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/ko.py +2 -2
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/ru.py +12 -9
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/zh.py +2 -2
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/version.py +1 -1
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13/scriptconv.egg-info}/PKG-INFO +167 -151
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_phonemizers_cjk_ar.py +33 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_phonemizers_ru.py +15 -0
- scriptconv-0.0.4a10/README.md +0 -374
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/LICENSE +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/pyproject.toml +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/requirements.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/__main__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/cangjie.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/conventions.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/data/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/data/cangjie5_tc.tsv.gz +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/diacritics.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/graph.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/notation.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/bw2ipa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/hangul2ipa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/aspiration.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/assimilation.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/double_coda.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/hanja.tsv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/ipa.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/neutralization.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/tensification.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/ko_tables/yale.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/codeswitch.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/diacritize.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/dialectal.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/english_g2p.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/espeak.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/frontend.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/levantine_g2p.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/normalize.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/shami/phoneme_inventory.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/vosk_g2p.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_thirdparty/zh_num.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/kog2p/LICENSE.md +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/kog2p/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/kog2p/rulebook.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/LICENSE.md +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/buck/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/buck/phonetise_buckwalter.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/buck/symbols.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/buck/tokenization.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/num2words.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/_vendored/mantoq/unicode_symbol2label.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/ar.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/en.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/enums.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/eu.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/fa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/gl.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/he.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/mul.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/mwl.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/o2ipa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/pt.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/registry.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/shami.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/phonemizers/vi.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/py.typed +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/readings.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/scripts.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv/translit.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv.egg-info/SOURCES.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv.egg-info/dependency_links.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv.egg-info/requires.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/scriptconv.egg-info/top_level.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/setup.cfg +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_arpa_stress.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_cangjie.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_cli.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_conventions.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_diacritics.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_diacritics_graph.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_errors_policy.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_examples.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_graph.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_notation.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_phonemizers_base.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_phonemizers_friendly_import_errors.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_readings.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_readings_zh.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_scripts.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_scripts_stressonnx_compat.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a13}/tests/test_translit.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scriptconv
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.4a13
|
|
4
4
|
Summary: Zero-dependency script & phoneme-notation core — ISO-15924 detection & metadata, IPA↔ARPABET/X-SAMPA/Lexique/Kirshenbaum/Cotovía/RFE, Buckwalter↔Arabic, Hangul→jamo, kana
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Project-URL: Homepage, https://github.com/TigreGotico/scriptconv
|
|
@@ -140,13 +140,15 @@ Dynamic: license-file
|
|
|
140
140
|
Text arrives in many representations: scripts (Cyrillic, Hangul, kana),
|
|
141
141
|
romanizations (pinyin, Buckwalter), phoneme notations (IPA, ARPABET, X-SAMPA),
|
|
142
142
|
input codes (Cangjie), and decorated or undecorated spellings (Arabic with or
|
|
143
|
-
without vowel marks, pinyin with tone marks or with digits).
|
|
144
|
-
identifies which representation a piece of text is in, converts between them,
|
|
145
|
-
and tells you *programmatically* how faithful each conversion is.
|
|
143
|
+
without vowel marks, pinyin with tone marks or with digits).
|
|
146
144
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
145
|
+
scriptconv identifies which representation a piece of text is in. It
|
|
146
|
+
converts between them, and it reports *programmatically* how faithful each
|
|
147
|
+
conversion is.
|
|
148
|
+
|
|
149
|
+
The core is pure Python with zero dependencies. Heavier capabilities, such as
|
|
150
|
+
dictionary-backed readings and phonemizer engines, live behind optional
|
|
151
|
+
extras and never load unless asked for.
|
|
150
152
|
|
|
151
153
|
```bash
|
|
152
154
|
pip install scriptconv
|
|
@@ -175,31 +177,34 @@ python -m scriptconv route mantoq x-sampa # the conversion path, hop by
|
|
|
175
177
|
|
|
176
178
|
## The mental model
|
|
177
179
|
|
|
178
|
-
scriptconv is organized around four ideas, layered from identity to sound
|
|
180
|
+
scriptconv is organized around four ideas, layered from identity to sound.
|
|
179
181
|
|
|
180
182
|
1. **Scripts are identity.** A character belongs to a writing system
|
|
181
|
-
(ISO 15924: `Latn`, `Cyrl`, `Arab`,
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
183
|
+
(ISO 15924: `Latn`, `Cyrl`, `Arab`, and so on), detectable from the text
|
|
184
|
+
itself, and identity never changes what the text *means*.
|
|
185
|
+
|
|
186
|
+
2. **Representations are nodes. Conversions are edges.** IPA, ARPABET,
|
|
187
|
+
hiragana, pinyin, and Cangjie codes are each a node in one conversion
|
|
188
|
+
graph, and asking for `mantoq → x-sampa` finds the cheapest path
|
|
186
189
|
(`mantoq → ipa → x-sampa`), preferring lossless edges by construction.
|
|
190
|
+
|
|
187
191
|
3. **Conventions are decorations, not identities.** Arabic vowel marks,
|
|
188
|
-
Hebrew points, Japanese word spacing, pinyin tone spelling
|
|
189
|
-
text can carry or omit
|
|
192
|
+
Hebrew points, Japanese word spacing, and pinyin tone spelling are
|
|
193
|
+
parameters a script's text can carry or omit (`strip`, `restyle`,
|
|
190
194
|
`apply`), never graph nodes.
|
|
195
|
+
|
|
191
196
|
4. **Fidelity is data.** Whether a conversion round-trips, and what happens
|
|
192
|
-
to a symbol a table
|
|
193
|
-
`errors=` policy)
|
|
197
|
+
to a symbol a table does not know, is queryable (`NOTATION_INFO`, the
|
|
198
|
+
`errors=` policy), not folklore buried in docstrings.
|
|
194
199
|
|
|
195
|
-
The
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
200
|
+
The core never invents pronunciation. Grapheme-to-phoneme inference needs
|
|
201
|
+
language knowledge beyond symbol tables. When you need it, the
|
|
202
|
+
`phonemizers` subpackage wraps real G2P engines behind extras: the same
|
|
203
|
+
graph, one more kind of edge, explicitly opted into.
|
|
199
204
|
|
|
200
205
|
## A guided tour
|
|
201
206
|
|
|
202
|
-
### What script is this?
|
|
207
|
+
### What script is this? The `scripts` module
|
|
203
208
|
|
|
204
209
|
```python
|
|
205
210
|
from scriptconv import detect_script, script_runs, lang_to_script, base_direction
|
|
@@ -207,19 +212,19 @@ from scriptconv import detect_script, script_runs, lang_to_script, base_directio
|
|
|
207
212
|
detect_script("Здравствуйте") # 'Cyrl'
|
|
208
213
|
script_runs("привет hello") # [('Cyrl', 'привет '), ('Latn', 'hello')]
|
|
209
214
|
base_direction("مرحبا hello") # 'mixed'
|
|
210
|
-
lang_to_script("uzb_cyr") # 'Cyrl'
|
|
215
|
+
lang_to_script("uzb_cyr") # 'Cyrl' (639-1/2/3 codes, BCP-47 tags,
|
|
211
216
|
# informal _cyr/_lat variants
|
|
212
217
|
```
|
|
213
218
|
|
|
214
|
-
The returned ISO 15924 tags are **stable API
|
|
215
|
-
against them directly. `script_runs` follows the UAX #24 model
|
|
216
|
-
marks and neutral characters attach to the run they qualify,
|
|
219
|
+
The returned ISO 15924 tags are **stable API**: downstream code compares
|
|
220
|
+
against them directly. `script_runs` follows the UAX #24 model, so combining
|
|
221
|
+
marks and neutral characters attach to the run they qualify, and accented
|
|
217
222
|
Cyrillic never splits. Details: [docs/scripts.md](docs/scripts.md).
|
|
218
223
|
|
|
219
|
-
### Same sounds, different symbols
|
|
224
|
+
### Same sounds, different symbols: the `notation` module
|
|
220
225
|
|
|
221
226
|
Nine phoneme notations transcode through an IPA hub: ARPABET, X-SAMPA,
|
|
222
|
-
Kirshenbaum, Lexique, Cotovía, RFE, Buckwalter
|
|
227
|
+
Kirshenbaum, Lexique, Cotovía, RFE, Buckwalter to Arabic, and mantoq.
|
|
223
228
|
|
|
224
229
|
```python
|
|
225
230
|
from scriptconv import convert, arpa_to_ipa, ipa_to_arpa
|
|
@@ -227,28 +232,30 @@ from scriptconv import convert, arpa_to_ipa, ipa_to_arpa
|
|
|
227
232
|
convert("HH AH0 L OW1", "arpa", "ipa") # 'həloʊ'
|
|
228
233
|
convert("kˈæt", "ipa", "x-sampa") # 'k"{t'
|
|
229
234
|
arpa_to_ipa("HH AH0 L OW1", stress=True) # 'həlˈoʊ'
|
|
230
|
-
ipa_to_arpa("həlˈoʊ", stress=True) # 'HH AH0 L OW1'
|
|
235
|
+
ipa_to_arpa("həlˈoʊ", stress=True) # 'HH AH0 L OW1' (exact round-trip)
|
|
231
236
|
```
|
|
232
237
|
|
|
233
238
|
Two contracts make this trustworthy:
|
|
234
239
|
|
|
235
240
|
- **`errors=`, codecs-style.** Every converter takes
|
|
236
241
|
`errors="pass" | "replace" | "strict" | "ignore"` for symbols outside its
|
|
237
|
-
table. `"strict"` raises `UnknownSymbolError(symbol, position, notation)
|
|
238
|
-
|
|
242
|
+
table. `"strict"` raises `UnknownSymbolError(symbol, position, notation)`.
|
|
243
|
+
The defaults preserve each converter's long-standing behavior.
|
|
244
|
+
|
|
239
245
|
- **Stress is preserved, not discarded.** With `stress=True`, ARPABET stress
|
|
240
|
-
digits become IPA `ˈ`/`ˌ` placed before the stressed vowel
|
|
241
|
-
construction, no syllabification involved. Round-trips
|
|
242
|
-
|
|
243
|
-
residue.
|
|
246
|
+
digits become IPA `ˈ`/`ˌ` placed before the stressed vowel. This is
|
|
247
|
+
reversible by construction, with no syllabification involved. Round-trips
|
|
248
|
+
are exact up to IPA-equivalence. The
|
|
249
|
+
[fidelity table](#fidelity-guarantees) states every residue.
|
|
244
250
|
|
|
245
251
|
Buckwalter is an independent implementation of the published transliteration
|
|
246
|
-
scheme, including alef wasla and the dagger alef (رحمٰن
|
|
247
|
-
|
|
248
|
-
|
|
252
|
+
scheme, including alef wasla and the dagger alef (رحمٰن transliterates as
|
|
253
|
+
*rHm`n*). Mantoq
|
|
254
|
+
is the phonetic alphabet of the Halabi Arabic-Phonetiser. Text in it
|
|
255
|
+
converts one way to IPA (`mantoq_to_ipa("mrHbaa")` → `'mrħbaː'`) and onward
|
|
249
256
|
through the graph. Details: [docs/notation.md](docs/notation.md).
|
|
250
257
|
|
|
251
|
-
### Rewriting the writing
|
|
258
|
+
### Rewriting the writing: `translit`, `readings`, `cangjie`
|
|
252
259
|
|
|
253
260
|
Deterministic, table-driven operations on the writing system itself:
|
|
254
261
|
|
|
@@ -259,9 +266,9 @@ decompose_hangul("한국") # 'ㅎㅏㄴㄱㅜㄱ'
|
|
|
259
266
|
hira_to_kana("こんにちは") # 'コンニチハ'
|
|
260
267
|
```
|
|
261
268
|
|
|
262
|
-
Where a respelling needs a *dictionary
|
|
263
|
-
mechanical
|
|
264
|
-
the dictionary
|
|
269
|
+
Where a respelling needs a *dictionary*, because the answer is lexical and
|
|
270
|
+
not mechanical, it lives behind an extra and raises a clear `ImportError`
|
|
271
|
+
when the dictionary is not installed:
|
|
265
272
|
|
|
266
273
|
```python
|
|
267
274
|
from scriptconv import to_hiragana, to_katakana, to_pinyin, to_bopomofo, to_cangjie
|
|
@@ -273,16 +280,16 @@ to_bopomofo("中国") # 'ㄓㄨㄥ ㄍㄨㄛˊ'
|
|
|
273
280
|
to_cangjie("倉頡") # 'oiar grmbc' vendored table, no extra
|
|
274
281
|
```
|
|
275
282
|
|
|
276
|
-
`readings.tokens()` exposes the per-token stream underneath for consumers
|
|
277
|
-
that need word boundaries
|
|
283
|
+
`readings.tokens()` exposes the per-token stream underneath, for consumers
|
|
284
|
+
that need word boundaries. The Cangjie table (103,601 glyphs) ships inside
|
|
278
285
|
the wheel, so shape-code conversion needs no download and no dependency.
|
|
279
286
|
Details: [docs/translit.md](docs/translit.md),
|
|
280
287
|
[docs/readings.md](docs/readings.md).
|
|
281
288
|
|
|
282
|
-
### Marked or unmarked
|
|
289
|
+
### Marked or unmarked: the `conventions` module
|
|
283
290
|
|
|
284
291
|
A convention is a decoration a script's orthography can carry or omit. Each
|
|
285
|
-
one declares its *styles* and the transitions between them
|
|
292
|
+
one declares its *styles* and the transitions between them. `strip`/`apply`
|
|
286
293
|
are sugar for transitions to and from `"none"`.
|
|
287
294
|
|
|
288
295
|
```python
|
|
@@ -296,30 +303,31 @@ detect_convention("مُحَمَّد", "tashkeel") # 'marked'
|
|
|
296
303
|
[c.id for c in conventions_for("Arab")] # ['tashkeel', 'kashida', 'quranic-marks']
|
|
297
304
|
```
|
|
298
305
|
|
|
299
|
-
The registered set
|
|
300
|
-
`niqqud` and `teamim` (vocalization and cantillation are separate
|
|
301
|
-
Japanese `wakachigaki` (spacing is stripped only between Japanese
|
|
302
|
-
"きょうは good day" keeps its space); `pinyin-tone` (mark
|
|
303
|
-
deterministic both ways for standard apostrophized
|
|
304
|
-
(compatibility
|
|
305
|
-
|
|
306
|
-
Which transitions exist follows one criterion
|
|
307
|
-
re-spelling are always available
|
|
308
|
-
(applying wakachigaki
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
306
|
+
The registered set covers Arabic `tashkeel`, `kashida`, and `quranic-marks`;
|
|
307
|
+
Hebrew `niqqud` and `teamim` (vocalization and cantillation are separate
|
|
308
|
+
layers); Japanese `wakachigaki` (spacing is stripped only between Japanese
|
|
309
|
+
characters, so "きょうは good day" keeps its space); `pinyin-tone` (mark,
|
|
310
|
+
number, or none, deterministic both ways for standard apostrophized
|
|
311
|
+
pinyin); and `jamo-form` (compatibility or conjoining repertoires).
|
|
312
|
+
|
|
313
|
+
Which transitions exist follows one criterion. Stripping and deterministic
|
|
314
|
+
re-spelling are always available. Transitions that need a dictionary lookup
|
|
315
|
+
(applying wakachigaki means word segmentation) sit behind an extra.
|
|
316
|
+
|
|
317
|
+
Transitions that need contextual *prediction* (restoring tashkeel) are out
|
|
318
|
+
of scope entirely, because that is diacritization, a modelling problem. Their
|
|
319
|
+
absence is queryable data rather than a runtime surprise.
|
|
320
|
+
|
|
321
|
+
The codepoint sets are curated, not naive. Stripping tashkeel **excludes**
|
|
322
|
+
U+0653 to U+0655, because in decomposed text those combining marks *are* the
|
|
323
|
+
letters آ/أ/إ. A blanket strip would corrupt the consonantal skeleton.
|
|
316
324
|
Details: [docs/conventions.md](docs/conventions.md).
|
|
317
325
|
|
|
318
|
-
### One graph over everything
|
|
326
|
+
### One graph over everything: the `graph` module
|
|
319
327
|
|
|
320
|
-
Every notation and orthography is a node
|
|
321
|
-
finds the cheapest path and prefers lossless edges, so a lossless
|
|
322
|
-
beats a lossy shortcut:
|
|
328
|
+
Every notation and orthography is a node. Every converter is an edge.
|
|
329
|
+
Routing finds the cheapest path and prefers lossless edges, so a lossless
|
|
330
|
+
two-hop route beats a lossy shortcut:
|
|
323
331
|
|
|
324
332
|
```python
|
|
325
333
|
from scriptconv import DEFAULT_GRAPH
|
|
@@ -329,17 +337,19 @@ DEFAULT_GRAPH.convert("こんにちは", "hira", "kana") # 'コンニチ
|
|
|
329
337
|
# ['arpa->ipa', 'ipa->x-sampa']
|
|
330
338
|
```
|
|
331
339
|
|
|
332
|
-
Edges are `fn(text, **context)
|
|
333
|
-
opaquely.
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
340
|
+
Edges are `fn(text, **context)`. Routing context such as `lang=` passes
|
|
341
|
+
through opaquely.
|
|
342
|
+
|
|
343
|
+
Extension is explicit: `graph.extend(register_fn)` returns an extended copy,
|
|
344
|
+
and a graph's contents never depend on what happens to be installed.
|
|
345
|
+
`DEFAULT_GRAPH` itself contains only orthographic and notation edges.
|
|
346
|
+
Details: [docs/graph.md](docs/graph.md).
|
|
337
347
|
|
|
338
|
-
### From spelling to sound
|
|
348
|
+
### From spelling to sound: the `phonemizers` module
|
|
339
349
|
|
|
340
|
-
Wrappers over real G2P engines
|
|
341
|
-
specialized per-language engines
|
|
342
|
-
default per language and full override:
|
|
350
|
+
Wrappers over real G2P engines, such as espeak, gruut, epitran, and ByT5,
|
|
351
|
+
and specialized per-language engines, sit behind per-capability extras,
|
|
352
|
+
with a default per language and a full override:
|
|
343
353
|
|
|
344
354
|
```python
|
|
345
355
|
from scriptconv.phonemizers import phonemize, Phonemizer
|
|
@@ -349,13 +359,14 @@ phonemize("hello", "en", override=Phonemizer.GRUUT)
|
|
|
349
359
|
```
|
|
350
360
|
|
|
351
361
|
Defaults resolve in-house engines first: an explicit per-language chain
|
|
352
|
-
(Arabic
|
|
353
|
-
Hebrew
|
|
354
|
-
notations), then
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
362
|
+
(Arabic to arbtok, Basque to euskaphone, Mirandese, Portuguese to
|
|
363
|
+
tugaphone, Hebrew to phonikud, Galician to Cotovía, and Russian to vosk for
|
|
364
|
+
their own notations), then orthography2ipa wherever it has a language spec,
|
|
365
|
+
then espeak as the last resort.
|
|
366
|
+
|
|
367
|
+
Arabic never falls back past arbtok. A missing engine raises an error
|
|
368
|
+
rather than degrading silently. Every backend resolves lazily, and a
|
|
369
|
+
missing package raises an `ImportError` naming the extra to install.
|
|
359
370
|
|
|
360
371
|
Phonemization joins the graph only on request:
|
|
361
372
|
|
|
@@ -365,29 +376,31 @@ from scriptconv import phonemizers
|
|
|
365
376
|
|
|
366
377
|
g = DEFAULT_GRAPH.extend(phonemizers.register)
|
|
367
378
|
g.convert("bom dia", "text", "ipa", lang="pt") # 'ˈbõ ˈdʒiɐ'
|
|
368
|
-
g.can_convert("text", "arpa") # True
|
|
379
|
+
g.can_convert("text", "arpa") # True (chains through IPA)
|
|
369
380
|
```
|
|
370
381
|
|
|
371
382
|
Two design points worth knowing:
|
|
372
383
|
|
|
373
|
-
- **Normalization is injectable.** TTS stacks expand numbers and dates
|
|
374
|
-
phonemizing
|
|
375
|
-
`normalizer=` (a `(text, lang) -> str` callable) to run
|
|
376
|
-
pipeline
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
384
|
+
- **Normalization is injectable.** TTS stacks expand numbers and dates
|
|
385
|
+
before phonemizing, and that needs language resources scriptconv does
|
|
386
|
+
not ship. Pass `normalizer=` (a `(text, lang) -> str` callable) to run
|
|
387
|
+
yours inside the pipeline. Without it, text is phonemized as-is.
|
|
388
|
+
|
|
389
|
+
- **Large or licensed model-backed engines never download on their own.**
|
|
390
|
+
ByT5/Charsiu require an explicit local `model=` path, and caching those
|
|
391
|
+
model files is the caller's concern. Small, known-good, unencumbered
|
|
392
|
+
models are the exception: the Hebrew phonikud diacritizer auto-provisions
|
|
393
|
+
its ONNX model to a cache dir on first use (`phonikud_model=` still
|
|
394
|
+
overrides it with a path or callable; the cache location is set via
|
|
395
|
+
`SCRIPTCONV_CACHE` or `XDG_CACHE_HOME`).
|
|
396
|
+
|
|
397
|
+
**Pre-G2P disambiguation.** `add_diacritics(text, lang, model=None)`
|
|
398
|
+
restores information ordinary orthography omits but a G2P needs, before
|
|
399
|
+
phonemization. It covers Hebrew niqqud (phonikud), Arabic tashkeel
|
|
400
|
+
(text2tashkeel), word stress across 26 stressonnx language tags (East
|
|
401
|
+
Slavic; Bulgarian, Macedonian, and Slovene; Latvian; Armenian; Georgian;
|
|
402
|
+
and Turkic/Caucasian languages, via `scriptconv[stress]`), and
|
|
403
|
+
European-Portuguese heterophonic-homograph sense diacritics (bifonia, via
|
|
391
404
|
`scriptconv[pt]`, never applied to `pt-BR`, whose vowel system differs):
|
|
392
405
|
|
|
393
406
|
```python
|
|
@@ -397,8 +410,8 @@ p = GraphemePhonemizer()
|
|
|
397
410
|
p.add_diacritics("Tenho muita sede hoje.", "pt") # 'Tenho muita sêde hoje.' (thirst)
|
|
398
411
|
```
|
|
399
412
|
|
|
400
|
-
Diacritization also joins the graph, like phonemization,
|
|
401
|
-
`scriptconv.diacritics.register
|
|
413
|
+
Diacritization also joins the graph, like phonemization does, through
|
|
414
|
+
`scriptconv.diacritics.register`: a `text -> text-diacritized` edge, opt-in,
|
|
402
415
|
with `text -> ipa` routing unchanged:
|
|
403
416
|
|
|
404
417
|
```python
|
|
@@ -408,41 +421,41 @@ g.convert("Tenho muita sede hoje.", "text", "text-diacritized", lang="pt")
|
|
|
408
421
|
# 'Tenho muita sêde hoje.'
|
|
409
422
|
```
|
|
410
423
|
|
|
411
|
-
Stress is unwritten or under-marked in all 26 covered languages
|
|
412
|
-
is the clearest case, where unstressed vowels also reduce (
|
|
413
|
-
|
|
414
|
-
optional (not yet on PyPI) and emits the
|
|
415
|
-
after the stressed vowel.
|
|
424
|
+
Stress is unwritten or under-marked in all 26 covered languages. East
|
|
425
|
+
Slavic is the clearest case, where unstressed vowels also reduce (for
|
|
426
|
+
example, Russian о becomes [ɐ] or [ə]), so a missing mark corrupts more
|
|
427
|
+
than prosody there. stressonnx is optional (not yet on PyPI) and emits the
|
|
428
|
+
standard combining acute (U+0301) after the stressed vowel.
|
|
416
429
|
|
|
417
430
|
Details: [docs/phonemizers.md](docs/phonemizers.md).
|
|
418
431
|
|
|
419
432
|
## Fidelity guarantees
|
|
420
433
|
|
|
421
434
|
Transcoding faithfulness depends on the target notation's inventory. IPA is
|
|
422
|
-
the hub
|
|
423
|
-
notation, whether a round-trip is exact and what
|
|
424
|
-
table does not know.
|
|
425
|
-
|
|
426
|
-
Every converter (and `convert()`) accepts a codecs-style `errors=` policy
|
|
427
|
-
symbols outside its table
|
|
428
|
-
except `ipa_to_arpa`)
|
|
429
|
-
(
|
|
430
|
-
`"ignore"` drops
|
|
431
|
-
symbol, its position, and the notation. The
|
|
432
|
-
below describes the per-converter default.
|
|
435
|
+
the hub, so notation-to-notation conversion goes through IPA. The table
|
|
436
|
+
below states, for each notation, whether a round-trip is exact and what
|
|
437
|
+
happens to a symbol the table does not know.
|
|
438
|
+
|
|
439
|
+
Every converter (and `convert()`) accepts a codecs-style `errors=` policy
|
|
440
|
+
for symbols outside its table. `"pass"` keeps the symbol (the default
|
|
441
|
+
everywhere except `ipa_to_arpa`). `"replace"` substitutes the notation's
|
|
442
|
+
placeholder (`?`, the historical `ipa_to_arpa` default, tunable via
|
|
443
|
+
`unknown=`). `"ignore"` drops the symbol. `"strict"` raises
|
|
444
|
+
`UnknownSymbolError` naming the symbol, its position, and the notation. The
|
|
445
|
+
"Unknown-token behaviour" column below describes the per-converter default.
|
|
433
446
|
|
|
434
447
|
| Notation | `to_ipa` → `from_ipa` round-trip | `from_ipa` → `to_ipa` round-trip | Unknown-token behaviour |
|
|
435
448
|
|----------|----------------------------------|----------------------------------|-------------------------|
|
|
436
|
-
| **ARPABET** | **Lossless with `stress=True`** (digits
|
|
449
|
+
| **ARPABET** | **Lossless with `stress=True`** (digits map to IPA `ˈ`/`ˌ` before the stressed vowel; residues: extended-ARPABET `AX` normalises to CMUdict's `AH0` spelling, and `AH0 R` fuses to the r-colored `AXR0`, stable from the IPA side). Default `stress=False` drops digits and merges `AH0` with `AX` | **Lossy** (ARPABET is an English-only inventory, so any IPA symbol outside it becomes the *unknown* placeholder) | `arpa_to_ipa`: passed through unchanged. `ipa_to_arpa`: diacritics and suprasegmentals are dropped. Other out-of-inventory symbols follow `errors=` (default `"replace"` → `?`) |
|
|
437
450
|
| **X-SAMPA** | Exact for all canonical symbols | Exact except aliases (`f\`→ɸ, `&`→æ) normalise to their canonical spelling | Passed through unchanged |
|
|
438
|
-
| **Buckwalter ↔ Arabic** | Exact (precomposed lam-alef ligatures decompose to two
|
|
439
|
-
| **Mantoq → IPA** | One-directional (gemination
|
|
440
|
-
| **Lexique ↔ IPA** | Exact except the `°`/`3` schwa pair (both
|
|
441
|
-
| **Kirshenbaum ↔ IPA** | Exact | **Lossy**
|
|
442
|
-
| **Cotovía ↔ IPA** | Exact except the three `L`/`Z`/`jj` symbols for `ʎ` normalise to `L` | **Lossy**
|
|
443
|
-
| **RFE ↔ IPA** | Exact except `ñ`/`n̮` for `ɲ` normalise to `ñ` | **Lossy**
|
|
444
|
-
|
|
445
|
-
Every row is backed by a test
|
|
451
|
+
| **Buckwalter ↔ Arabic** | Exact (precomposed lam-alef ligatures decompose to two characters, visually identical) | Exact | Follows `errors=` (default: passed through) |
|
|
452
|
+
| **Mantoq → IPA** | One-directional (gemination and word markers are consumed; there is no IPA to Mantoq) | Not applicable | Follows `errors=` (default: passed through) |
|
|
453
|
+
| **Lexique ↔ IPA** | Exact except the `°`/`3` schwa pair (both map to `ə`; the reverse always produces `°`) | Exact | Passed through unchanged |
|
|
454
|
+
| **Kirshenbaum ↔ IPA** | Exact | **Lossy** (restricted ASCII inventory; IPA outside it passes through) | Passed through unchanged |
|
|
455
|
+
| **Cotovía ↔ IPA** | Exact except the three `L`/`Z`/`jj` symbols for `ʎ` normalise to `L` | **Lossy** (Galician/Spanish inventory; IPA outside it passes through) | Passed through unchanged |
|
|
456
|
+
| **RFE ↔ IPA** | Exact except `ñ`/`n̮` for `ɲ` normalise to `ñ` | **Lossy** (core Spanish/Romance inventory; IPA outside it passes through) | Passed through unchanged |
|
|
457
|
+
|
|
458
|
+
Every row is backed by a test. `NOTATION_INFO` exposes the same facts to
|
|
446
459
|
code, so a program can branch on whether a conversion is safe before making
|
|
447
460
|
it.
|
|
448
461
|
|
|
@@ -452,22 +465,22 @@ The core installs with zero dependencies. Capabilities opt in:
|
|
|
452
465
|
|
|
453
466
|
| Extra | Enables |
|
|
454
467
|
|---|---|
|
|
455
|
-
| `ja` / `zh` | Dictionary readings: kanji
|
|
468
|
+
| `ja` / `zh` | Dictionary readings: kanji to kana (pykakasi), hanzi to pinyin/bopomofo (pypinyin) |
|
|
456
469
|
| `phonemizers` | The phonemizer base layer (sentence chunking, language matching) |
|
|
457
470
|
| `espeak`, `gruut`, `goruut`, `epitran`, `transphone`, `misaki`, `byt5` | Multilingual phonemizer backends |
|
|
458
471
|
| `en-phonemizers`, `ja-phonemizers`, `zh-phonemizers`, `ko`, `ar-phonemizers`, `eu`, `pt-phonemizers`, `gl`, `he`, `fa`, `vi`, `mwl`, `shami`, `o2i` | Per-language phonemizer backends |
|
|
459
472
|
| `tashkeel` | Arabic diacritization for the phonemizer pipeline (text2tashkeel) |
|
|
460
|
-
| `stress` | Word-stress restoration for 26 stressonnx language tags
|
|
473
|
+
| `stress` | Word-stress restoration for 26 stressonnx language tags: East Slavic; Bulgarian, Macedonian, and Slovene; Latvian; Armenian; Georgian; and Turkic/Caucasian, for the phonemizer pipeline (stressonnx; not yet on PyPI) |
|
|
461
474
|
| `pt` | European-Portuguese heterophonic-homograph sense diacritics for the phonemizer pipeline (bifonia) |
|
|
462
475
|
|
|
463
476
|
## Licensing
|
|
464
477
|
|
|
465
|
-
scriptconv is Apache-2.0, with one deliberate, clearly
|
|
478
|
+
scriptconv is Apache-2.0, with one deliberate, clearly bounded exception:
|
|
466
479
|
`scriptconv/phonemizers/_vendored/` quarantines two unpublished third-party
|
|
467
|
-
G2P implementations under **their own licenses
|
|
468
|
-
core (CC BY-NC 4.0, non-commercial) and KoG2P (GPL-3.0)
|
|
469
|
-
LICENSE.md in the directory and in the wheel. Nothing imports them at
|
|
470
|
-
import time
|
|
480
|
+
G2P implementations under **their own licenses**: mantoq's phonetisation
|
|
481
|
+
core (CC BY-NC 4.0, non-commercial) and KoG2P (GPL-3.0), each with its
|
|
482
|
+
LICENSE.md in the directory and in the wheel. Nothing imports them at
|
|
483
|
+
package import time. They load only when a caller explicitly requests those
|
|
471
484
|
phonemizers, and unencumbered defaults (arbtok for Arabic, g2pk for Korean)
|
|
472
485
|
exist for both.
|
|
473
486
|
|
|
@@ -476,21 +489,23 @@ exist for both.
|
|
|
476
489
|
scriptconv sits at the bottom of a family of text and speech libraries that
|
|
477
490
|
build on it:
|
|
478
491
|
|
|
479
|
-
- [phoonnx](https://github.com/TigreGotico/phoonnx)
|
|
480
|
-
text-to-speech; consumes scriptconv for scripts, notation, conventions
|
|
481
|
-
the whole phonemizer layer.
|
|
482
|
-
- [orthography2ipa](https://github.com/TigreGotico/orthography2ipa)
|
|
483
|
-
data-driven orthography
|
|
484
|
-
|
|
485
|
-
- [
|
|
486
|
-
- [
|
|
487
|
-
- [
|
|
488
|
-
- [
|
|
489
|
-
- [
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
492
|
+
- [phoonnx](https://github.com/TigreGotico/phoonnx): multilingual ONNX
|
|
493
|
+
text-to-speech; consumes scriptconv for scripts, notation, conventions,
|
|
494
|
+
and the whole phonemizer layer.
|
|
495
|
+
- [orthography2ipa](https://github.com/TigreGotico/orthography2ipa): a
|
|
496
|
+
data-driven orthography-to-IPA engine, and the usual per-language
|
|
497
|
+
phonemizer default.
|
|
498
|
+
- [arbtok](https://github.com/TigreGotico/arbtok): Arabic phonemizer (dialect-aware).
|
|
499
|
+
- [euskaphone](https://github.com/TigreGotico/euskaphone): Basque phonemizer (dialect-aware).
|
|
500
|
+
- [tugaphone](https://github.com/TigreGotico/tugaphone): Portuguese phonemizer (dialect-aware).
|
|
501
|
+
- [mwl_phonemizer](https://github.com/TigreGotico/mwl_phonemizer): Mirandese phonemizer.
|
|
502
|
+
- [g2p_barranquenho](https://github.com/TigreGotico/g2p_barranquenho): Barranquenho phonemizer.
|
|
503
|
+
- [pycotovia](https://github.com/TigreGotico/pycotovia): a pure-Python
|
|
504
|
+
phonemizer port of the Cotovía Galician TTS engine, whose notation
|
|
505
|
+
scriptconv transcodes.
|
|
506
|
+
- [espyak](https://github.com/TigreGotico/espyak): a pure-Python port of
|
|
507
|
+
espeak-ng's G2P, and the espeak fallback when the binary is absent.
|
|
508
|
+
|
|
494
509
|
## Development
|
|
495
510
|
|
|
496
511
|
```bash
|
|
@@ -499,3 +514,4 @@ pytest tests/
|
|
|
499
514
|
```
|
|
500
515
|
|
|
501
516
|
The documentation lives in [`docs/`](docs/index.md), one page per module.
|
|
517
|
+
</content>
|