scriptconv 0.0.4a9__tar.gz → 0.0.4a10__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scriptconv-0.0.4a9/scriptconv.egg-info → scriptconv-0.0.4a10}/PKG-INFO +3 -2
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/README.md +2 -1
- scriptconv-0.0.4a10/scriptconv/phonemizers/_thirdparty/vosk_g2p.py +135 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/enums.py +2 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/registry.py +5 -0
- scriptconv-0.0.4a10/scriptconv/phonemizers/ru.py +114 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/version.py +1 -1
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10/scriptconv.egg-info}/PKG-INFO +3 -2
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/SOURCES.txt +3 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_phonemizers_base.py +4 -2
- scriptconv-0.0.4a10/tests/test_phonemizers_ru.py +218 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/LICENSE +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/pyproject.toml +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/requirements.txt +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/__main__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/cangjie.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/conventions.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/data/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/data/cangjie5_tc.tsv.gz +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/diacritics.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/graph.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/notation.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/bw2ipa.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/hangul2ipa.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/aspiration.csv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/assimilation.csv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/double_coda.csv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/hanja.tsv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/ipa.csv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/neutralization.csv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/tensification.csv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/yale.csv +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/codeswitch.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/diacritize.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/dialectal.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/english_g2p.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/espeak.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/frontend.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/levantine_g2p.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/normalize.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/phoneme_inventory.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/zh_num.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/LICENSE.md +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/rulebook.txt +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/LICENSE.md +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/__init__.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/phonetise_buckwalter.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/symbols.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/tokenization.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/num2words.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/unicode_symbol2label.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/ar.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/base.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/en.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/eu.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/fa.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/gl.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/he.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/ja.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/ko.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/mul.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/mwl.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/o2ipa.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/pt.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/shami.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/vi.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/zh.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/py.typed +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/readings.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/scripts.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/translit.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/dependency_links.txt +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/requires.txt +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/top_level.txt +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/setup.cfg +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_arpa_stress.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_cangjie.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_cli.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_conventions.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_diacritics.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_diacritics_graph.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_errors_policy.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_examples.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_graph.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_notation.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_phonemizers_cjk_ar.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_phonemizers_friendly_import_errors.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_readings.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_readings_zh.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_scripts.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_scripts_stressonnx_compat.py +0 -0
- {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_translit.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scriptconv
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.4a10
|
|
4
4
|
Summary: Zero-dependency script & phoneme-notation core — ISO-15924 detection & metadata, IPA↔ARPABET/X-SAMPA/Lexique/Kirshenbaum/Cotovía/RFE, Buckwalter↔Arabic, Hangul→jamo, kana
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Project-URL: Homepage, https://github.com/TigreGotico/scriptconv
|
|
@@ -350,7 +350,8 @@ phonemize("hello", "en", override=Phonemizer.GRUUT)
|
|
|
350
350
|
|
|
351
351
|
Defaults resolve in-house engines first: an explicit per-language chain
|
|
352
352
|
(Arabic → arbtok, Basque → euskaphone, Mirandese, Portuguese → tugaphone,
|
|
353
|
-
Hebrew → phonikud, Galician → Cotovía for
|
|
353
|
+
Hebrew → phonikud, Galician → Cotovía and Russian → vosk for their own
|
|
354
|
+
notations), then
|
|
354
355
|
orthography2ipa wherever it has a language spec, then espeak as the last
|
|
355
356
|
resort. Arabic never falls back past arbtok — a missing engine raises rather
|
|
356
357
|
than silently degrading. Every backend resolves lazily; a missing package
|
|
@@ -223,7 +223,8 @@ phonemize("hello", "en", override=Phonemizer.GRUUT)
|
|
|
223
223
|
|
|
224
224
|
Defaults resolve in-house engines first: an explicit per-language chain
|
|
225
225
|
(Arabic → arbtok, Basque → euskaphone, Mirandese, Portuguese → tugaphone,
|
|
226
|
-
Hebrew → phonikud, Galician → Cotovía for
|
|
226
|
+
Hebrew → phonikud, Galician → Cotovía and Russian → vosk for their own
|
|
227
|
+
notations), then
|
|
227
228
|
orthography2ipa wherever it has a language spec, then espeak as the last
|
|
228
229
|
resort. Arabic never falls back past arbtok — a missing engine raises rather
|
|
229
230
|
than silently degrading. Every backend resolves lazily; a missing package
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Russian grapheme-to-phoneme rules and pronunciation-dictionary loader for the
|
|
3
|
+
Vosk-TTS voices.
|
|
4
|
+
|
|
5
|
+
Vendored from ``vosk_tts/g2p.py`` — https://github.com/alphacep/vosk-tts
|
|
6
|
+
(Apache-2.0) — so the Vosk Russian front-end is available without the
|
|
7
|
+
``vosk-tts`` package at runtime.
|
|
8
|
+
|
|
9
|
+
``convert`` is a faithful port of ``vosk_tts/g2p.py``: it turns an (optionally
|
|
10
|
+
stress-marked) Russian word into the Vosk phoneme inventory — palatalised
|
|
11
|
+
consonants get a trailing ``j`` (``bj``, ``tj`` …), vowels carry a stress digit
|
|
12
|
+
(``a0`` unstressed, ``a1`` stressed). A ``+`` immediately before a vowel marks
|
|
13
|
+
it as stressed; without any ``+`` every vowel is emitted unstressed.
|
|
14
|
+
|
|
15
|
+
The shipped ``dictionary`` file (word → phonemes) overrides the rules for known
|
|
16
|
+
words; ``load_dictionary`` reads both the historical ``word phon…`` layout and
|
|
17
|
+
the newer ``word prob phon…`` layout (keeping the highest-probability variant).
|
|
18
|
+
The rules alone already produce usable Russian, so the dictionary is optional.
|
|
19
|
+
"""
|
|
20
|
+
import os
|
|
21
|
+
from typing import Dict, List, Optional
|
|
22
|
+
|
|
23
|
+
# Cyrillic letters that soften the preceding consonant.
|
|
24
|
+
softletters = set(u"яёюиье")
|
|
25
|
+
# Contexts after which я/ю/е/ё gains a leading glide /j/.
|
|
26
|
+
startsyl = set(u"#ъьаяоёуюэеиы-")
|
|
27
|
+
# Markers dropped from the final phoneme stream.
|
|
28
|
+
others = set(["#", "+", "-", u"ь", u"ъ"])
|
|
29
|
+
|
|
30
|
+
softhard_cons = {
|
|
31
|
+
u"б": u"b", u"в": u"v", u"г": u"g", u"Г": u"g", u"д": u"d",
|
|
32
|
+
u"з": u"z", u"к": u"k", u"л": u"l", u"м": u"m", u"н": u"n",
|
|
33
|
+
u"п": u"p", u"р": u"r", u"с": u"s", u"т": u"t", u"ф": u"f",
|
|
34
|
+
u"х": u"h",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
other_cons = {
|
|
38
|
+
u"ж": u"zh", u"ц": u"c", u"ч": u"ch", u"ш": u"sh",
|
|
39
|
+
u"щ": u"sch", u"й": u"j",
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
vowels = {
|
|
43
|
+
u"а": u"a", u"я": u"a", u"у": u"u", u"ю": u"u", u"о": u"o",
|
|
44
|
+
u"ё": u"o", u"э": u"e", u"е": u"e", u"и": u"i", u"ы": u"y",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def pallatize(phones: List[tuple]) -> None:
|
|
49
|
+
"""In-place: map consonants to their (palatalised) phoneme, looking one
|
|
50
|
+
character ahead to decide whether a soft vowel follows."""
|
|
51
|
+
for i, phone in enumerate(phones[:-1]):
|
|
52
|
+
if phone[0] in softhard_cons:
|
|
53
|
+
if phones[i + 1][0] in softletters:
|
|
54
|
+
phones[i] = (softhard_cons[phone[0]] + "j", 0)
|
|
55
|
+
else:
|
|
56
|
+
phones[i] = (softhard_cons[phone[0]], 0)
|
|
57
|
+
if phone[0] in other_cons:
|
|
58
|
+
phones[i] = (other_cons[phone[0]], 0)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def convert_vowels(phones: List[tuple]) -> List[str]:
|
|
62
|
+
"""Emit vowels with their stress digit, inserting a glide /j/ before
|
|
63
|
+
iotated vowels at syllable starts."""
|
|
64
|
+
new_phones: List[str] = []
|
|
65
|
+
prev = ""
|
|
66
|
+
for phone in phones:
|
|
67
|
+
if prev in startsyl:
|
|
68
|
+
if phone[0] in set(u"яюеё"):
|
|
69
|
+
new_phones.append("j")
|
|
70
|
+
if phone[0] in vowels:
|
|
71
|
+
new_phones.append(vowels[phone[0]] + str(phone[1]))
|
|
72
|
+
else:
|
|
73
|
+
new_phones.append(phone[0])
|
|
74
|
+
prev = phone[0]
|
|
75
|
+
return new_phones
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def convert(stressword: str) -> str:
|
|
79
|
+
"""Convert a (possibly ``+``-stress-marked) Russian word to a
|
|
80
|
+
space-separated Vosk phoneme string."""
|
|
81
|
+
phones = ("#" + stressword + "#")
|
|
82
|
+
|
|
83
|
+
# Assign stress marks: a '+' sets the stress flag for the next character.
|
|
84
|
+
stress_phones = []
|
|
85
|
+
stress = 0
|
|
86
|
+
for phone in phones:
|
|
87
|
+
if phone == "+":
|
|
88
|
+
stress = 1
|
|
89
|
+
else:
|
|
90
|
+
stress_phones.append((phone, stress))
|
|
91
|
+
stress = 0
|
|
92
|
+
|
|
93
|
+
pallatize(stress_phones)
|
|
94
|
+
phones = convert_vowels(stress_phones)
|
|
95
|
+
phones = [x for x in phones if x not in others]
|
|
96
|
+
return " ".join(phones)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def load_dictionary(path: Optional[str]) -> Dict[str, List[str]]:
|
|
100
|
+
"""
|
|
101
|
+
Load a Vosk pronunciation dictionary into a ``word -> [phoneme, …]`` map.
|
|
102
|
+
|
|
103
|
+
Handles both file layouts:
|
|
104
|
+
- ``word phon1 phon2 …`` (older voices, e.g. 0.1)
|
|
105
|
+
- ``word prob phon1 phon2 …`` (newer voices; highest prob wins)
|
|
106
|
+
|
|
107
|
+
Returns an empty dict when ``path`` is falsy or missing — the rule-based
|
|
108
|
+
:func:`convert` fallback covers any out-of-dictionary word.
|
|
109
|
+
"""
|
|
110
|
+
dic: Dict[str, List[str]] = {}
|
|
111
|
+
if not path or not os.path.isfile(path):
|
|
112
|
+
return dic
|
|
113
|
+
probs: Dict[str, float] = {}
|
|
114
|
+
with open(path, encoding="utf-8") as f:
|
|
115
|
+
for line in f:
|
|
116
|
+
parts = line.split()
|
|
117
|
+
if len(parts) < 2:
|
|
118
|
+
continue
|
|
119
|
+
word, rest = parts[0], parts[1:]
|
|
120
|
+
# The second column is a probability only when it parses as a float;
|
|
121
|
+
# a real phoneme (a0, sch, …) never does.
|
|
122
|
+
try:
|
|
123
|
+
prob = float(rest[0])
|
|
124
|
+
phones = rest[1:]
|
|
125
|
+
except ValueError:
|
|
126
|
+
prob = None
|
|
127
|
+
phones = rest
|
|
128
|
+
if not phones:
|
|
129
|
+
continue
|
|
130
|
+
if prob is None:
|
|
131
|
+
dic.setdefault(word, phones)
|
|
132
|
+
elif probs.get(word, -1.0) < prob:
|
|
133
|
+
dic[word] = phones
|
|
134
|
+
probs[word] = prob
|
|
135
|
+
return dic
|
|
@@ -30,6 +30,7 @@ class Alphabet(str, Enum):
|
|
|
30
30
|
BUCKWALTER = "buckwalter"
|
|
31
31
|
MANTOQ = "mantoq" # ar — Halabi Arabic-Phonetiser inventory # ar
|
|
32
32
|
CANGJIE = "cangjie" # zh (Cangjie input method)
|
|
33
|
+
VOSK = "vosk" # ru — vosk-tts phoneme inventory (a0, bj, sch ...)
|
|
33
34
|
GRAPHEMES = "graphemes" # plain text / grapheme input (user-side)
|
|
34
35
|
|
|
35
36
|
|
|
@@ -73,6 +74,7 @@ class Phonemizer(str, Enum):
|
|
|
73
74
|
PYPINYIN = "pypinyin" # chinese
|
|
74
75
|
XPINYIN = "xpinyin" # chinese
|
|
75
76
|
JIEBA = "jieba" # chinese (not a real phonemizer!)
|
|
77
|
+
VOSK = "vosk" # russian (no ipa!)
|
|
76
78
|
SHAMI = "shami" # Levantine Arabic / English code-switching (ShamiVITS)
|
|
77
79
|
ARBTOK = "arbtok" # arabic (dialect-aware, undiacritized text; o2i lattice)
|
|
78
80
|
EUSKAPHONE = "euskaphone" # basque (dialect-aware; o2i lattice)
|
|
@@ -71,6 +71,7 @@ PHONEMIZER_REGISTRY: Dict[Phonemizer, Tuple[str, str, Optional[str]]] = {
|
|
|
71
71
|
_P.MANTOQ: (f"{_BASE}.ar", "MantoqPhonemizer", "ar-phonemizers"),
|
|
72
72
|
_P.ARBTOK: (f"{_BASE}.ar", "ArbtokPhonemizer", "ar-phonemizers"),
|
|
73
73
|
_P.SHAMI: (f"{_BASE}.shami", "ShamiPhonemizer", "shami"),
|
|
74
|
+
_P.VOSK: (f"{_BASE}.ru", "VoskPhonemizer", "phonemizers"),
|
|
74
75
|
}
|
|
75
76
|
|
|
76
77
|
|
|
@@ -135,6 +136,7 @@ _EMITS: Dict[Phonemizer, Tuple[Alphabet, ...]] = {
|
|
|
135
136
|
_P.TUGAPHONE: (Alphabet.IPA,),
|
|
136
137
|
_P.PHONIKUD: (Alphabet.IPA,),
|
|
137
138
|
_P.COTOVIA: (Alphabet.COTOVIA,),
|
|
139
|
+
_P.VOSK: (Alphabet.VOSK,),
|
|
138
140
|
_P.ESPEAK: (Alphabet.IPA,),
|
|
139
141
|
}
|
|
140
142
|
|
|
@@ -147,6 +149,9 @@ LANG_DEFAULTS: Dict[str, Tuple[Phonemizer, ...]] = {
|
|
|
147
149
|
"pt": (_P.TUGAPHONE,),
|
|
148
150
|
"he": (_P.PHONIKUD,),
|
|
149
151
|
"gl": (_P.COTOVIA, _P.ORTHOGRAPHY2IPA, _P.ESPEAK),
|
|
152
|
+
# vosk emits its own inventory, so it is the Russian default only
|
|
153
|
+
# when that notation is requested (same shape as Cotovía above)
|
|
154
|
+
"ru": (_P.VOSK, _P.ORTHOGRAPHY2IPA, _P.ESPEAK),
|
|
150
155
|
}
|
|
151
156
|
|
|
152
157
|
# Languages whose explicit entry is exhaustive: no generic fallback beyond it.
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Russian phonemizers.
|
|
2
|
+
|
|
3
|
+
Currently one backend: the Vosk-TTS front-end, wrapping the vendored
|
|
4
|
+
:mod:`scriptconv.phonemizers._thirdparty.vosk_g2p` rules so the alphacep
|
|
5
|
+
Russian voices can be driven from text without the ``vosk-tts`` package.
|
|
6
|
+
"""
|
|
7
|
+
import re
|
|
8
|
+
from typing import List, Optional
|
|
9
|
+
|
|
10
|
+
from quebra_frases import sentence_tokenize
|
|
11
|
+
|
|
12
|
+
from scriptconv.phonemizers.base import BasePhonemizer, PhonemizedChunks
|
|
13
|
+
from scriptconv.phonemizers.enums import Alphabet
|
|
14
|
+
from scriptconv.phonemizers._thirdparty.vosk_g2p import convert, load_dictionary
|
|
15
|
+
|
|
16
|
+
__all__ = ["VoskPhonemizer"]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class VoskPhonemizer(BasePhonemizer):
|
|
20
|
+
"""
|
|
21
|
+
Russian phonemizer for the Vosk-TTS voices (alphacep).
|
|
22
|
+
|
|
23
|
+
It reproduces ``vosk_tts``'s grapheme-to-phoneme exactly: each word is
|
|
24
|
+
looked up in the voice's pronunciation ``dictionary`` (word -> phonemes),
|
|
25
|
+
falling back to the rule-based
|
|
26
|
+
:func:`~scriptconv.phonemizers._thirdparty.vosk_g2p.convert` for
|
|
27
|
+
out-of-dictionary words. Spaces and punctuation are kept as their own
|
|
28
|
+
tokens (Vosk feeds them to the model as short/long pauses); the BOS ``^`` /
|
|
29
|
+
EOS ``$`` markers and the inter-phoneme blanks belong to the consumer's
|
|
30
|
+
tokenizer, so they are *not* emitted here.
|
|
31
|
+
|
|
32
|
+
The output is the Vosk phoneme inventory (``a0``, ``bj``, ``sch`` …), not
|
|
33
|
+
IPA, so the only supported alphabet is :attr:`Alphabet.VOSK` — like
|
|
34
|
+
Cotovía, this backend is eligible only when its own notation is requested.
|
|
35
|
+
|
|
36
|
+
The dictionary is optional: without it the rules alone still produce usable
|
|
37
|
+
Russian (only the curated stress and exception entries are lost).
|
|
38
|
+
scriptconv never downloads anything — the caller resolves the file and
|
|
39
|
+
passes its path as ``model`` (the registry's ``phonemizer_model`` knob).
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
alphabet (Alphabet): must be :attr:`Alphabet.VOSK`.
|
|
43
|
+
model (Optional[str]): path to the voice's ``dictionary`` file. When
|
|
44
|
+
absent or missing, only the rule-based fallback is used.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
# Matches the per-character split used by vosk_tts: spaces and punctuation
|
|
48
|
+
# are captured so they survive as standalone pause tokens.
|
|
49
|
+
_SPLIT = re.compile(r'([,.?!;:"() ])')
|
|
50
|
+
|
|
51
|
+
def __init__(self, alphabet: Alphabet = Alphabet.VOSK,
|
|
52
|
+
model: Optional[str] = None):
|
|
53
|
+
if alphabet != Alphabet.VOSK:
|
|
54
|
+
raise ValueError(
|
|
55
|
+
"VoskPhonemizer emits the vosk-tts phoneme inventory, not "
|
|
56
|
+
f"{Alphabet(alphabet).value!r} — use Alphabet.VOSK")
|
|
57
|
+
self._dict_path = model
|
|
58
|
+
self._dictionary: Optional[dict] = None # lazy: dictionaries are large
|
|
59
|
+
super().__init__(alphabet)
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def dictionary(self) -> dict:
|
|
63
|
+
"""The loaded ``word -> [phoneme, …]`` map (empty without a path)."""
|
|
64
|
+
if self._dictionary is None:
|
|
65
|
+
self._dictionary = load_dictionary(self._dict_path)
|
|
66
|
+
return self._dictionary
|
|
67
|
+
|
|
68
|
+
@classmethod
|
|
69
|
+
def get_lang(cls, target_lang: str) -> str:
|
|
70
|
+
return cls.match_lang(target_lang, ["ru-RU"])
|
|
71
|
+
|
|
72
|
+
def _g2p_tokens(self, text: str) -> List[str]:
|
|
73
|
+
"""Word/punctuation stream -> Vosk phoneme tokens (no BOS/EOS/blanks)."""
|
|
74
|
+
tokens: List[str] = []
|
|
75
|
+
# the em dash is a pause, and vosk only knows the ASCII hyphen
|
|
76
|
+
text = text.replace("—", "-")
|
|
77
|
+
for word in self._SPLIT.split(text.lower()):
|
|
78
|
+
if word == "":
|
|
79
|
+
continue
|
|
80
|
+
if self._SPLIT.match(word) or word == "-":
|
|
81
|
+
# space or punctuation: kept verbatim as a pause token
|
|
82
|
+
tokens.append(word)
|
|
83
|
+
elif word in self.dictionary:
|
|
84
|
+
tokens.extend(self.dictionary[word])
|
|
85
|
+
else:
|
|
86
|
+
tokens.extend(convert(word).split())
|
|
87
|
+
return tokens
|
|
88
|
+
|
|
89
|
+
def phonemize(self, text: str, lang: str) -> PhonemizedChunks:
|
|
90
|
+
"""Sentence-level lists of Vosk phoneme tokens.
|
|
91
|
+
|
|
92
|
+
Punctuation is preserved (it drives pausing); each sentence becomes one
|
|
93
|
+
synthesis chunk. Multi-character tokens (``sch``, ``bj``, ``a1``) stay
|
|
94
|
+
whole, so :meth:`BasePhonemizer.phonemize`'s per-character split is
|
|
95
|
+
deliberately bypassed.
|
|
96
|
+
"""
|
|
97
|
+
self.get_lang(lang)
|
|
98
|
+
if not text:
|
|
99
|
+
return []
|
|
100
|
+
if self.normalizer is not None:
|
|
101
|
+
text = self.normalizer(text, lang)
|
|
102
|
+
results: PhonemizedChunks = []
|
|
103
|
+
for sentence in sentence_tokenize(text):
|
|
104
|
+
tokens = self._g2p_tokens(sentence)
|
|
105
|
+
if tokens:
|
|
106
|
+
results.append(tokens)
|
|
107
|
+
return results
|
|
108
|
+
|
|
109
|
+
def phonemize_to_list(self, text: str, lang: str) -> List[str]:
|
|
110
|
+
self.get_lang(lang)
|
|
111
|
+
return self._g2p_tokens(text.lower())
|
|
112
|
+
|
|
113
|
+
def phonemize_string(self, text: str, lang: str) -> str:
|
|
114
|
+
return " ".join(self.phonemize_to_list(text, lang))
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scriptconv
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.4a10
|
|
4
4
|
Summary: Zero-dependency script & phoneme-notation core — ISO-15924 detection & metadata, IPA↔ARPABET/X-SAMPA/Lexique/Kirshenbaum/Cotovía/RFE, Buckwalter↔Arabic, Hangul→jamo, kana
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Project-URL: Homepage, https://github.com/TigreGotico/scriptconv
|
|
@@ -350,7 +350,8 @@ phonemize("hello", "en", override=Phonemizer.GRUUT)
|
|
|
350
350
|
|
|
351
351
|
Defaults resolve in-house engines first: an explicit per-language chain
|
|
352
352
|
(Arabic → arbtok, Basque → euskaphone, Mirandese, Portuguese → tugaphone,
|
|
353
|
-
Hebrew → phonikud, Galician → Cotovía for
|
|
353
|
+
Hebrew → phonikud, Galician → Cotovía and Russian → vosk for their own
|
|
354
|
+
notations), then
|
|
354
355
|
orthography2ipa wherever it has a language spec, then espeak as the last
|
|
355
356
|
resort. Arabic never falls back past arbtok — a missing engine raises rather
|
|
356
357
|
than silently degrading. Every backend resolves lazily; a missing package
|
|
@@ -37,12 +37,14 @@ scriptconv/phonemizers/mwl.py
|
|
|
37
37
|
scriptconv/phonemizers/o2ipa.py
|
|
38
38
|
scriptconv/phonemizers/pt.py
|
|
39
39
|
scriptconv/phonemizers/registry.py
|
|
40
|
+
scriptconv/phonemizers/ru.py
|
|
40
41
|
scriptconv/phonemizers/shami.py
|
|
41
42
|
scriptconv/phonemizers/vi.py
|
|
42
43
|
scriptconv/phonemizers/zh.py
|
|
43
44
|
scriptconv/phonemizers/_thirdparty/__init__.py
|
|
44
45
|
scriptconv/phonemizers/_thirdparty/bw2ipa.py
|
|
45
46
|
scriptconv/phonemizers/_thirdparty/hangul2ipa.py
|
|
47
|
+
scriptconv/phonemizers/_thirdparty/vosk_g2p.py
|
|
46
48
|
scriptconv/phonemizers/_thirdparty/zh_num.py
|
|
47
49
|
scriptconv/phonemizers/_thirdparty/ko_tables/aspiration.csv
|
|
48
50
|
scriptconv/phonemizers/_thirdparty/ko_tables/assimilation.csv
|
|
@@ -87,6 +89,7 @@ tests/test_notation.py
|
|
|
87
89
|
tests/test_phonemizers_base.py
|
|
88
90
|
tests/test_phonemizers_cjk_ar.py
|
|
89
91
|
tests/test_phonemizers_friendly_import_errors.py
|
|
92
|
+
tests/test_phonemizers_ru.py
|
|
90
93
|
tests/test_readings.py
|
|
91
94
|
tests/test_readings_zh.py
|
|
92
95
|
tests/test_scripts.py
|
|
@@ -21,9 +21,11 @@ class TestEnums(unittest.TestCase):
|
|
|
21
21
|
self.assertEqual(Phonemizer.ARBTOK.value, "arbtok")
|
|
22
22
|
self.assertEqual(Phonemizer.MIRANDESE.value, "mwl_phonemizer")
|
|
23
23
|
self.assertEqual(Phonemizer.KOG2PK.value, "kog2p")
|
|
24
|
+
self.assertEqual(Phonemizer.VOSK.value, "vosk")
|
|
24
25
|
self.assertEqual(Alphabet.XSAMPA.value, "x-sampa")
|
|
25
|
-
self.assertEqual(
|
|
26
|
-
self.assertEqual(len(list(
|
|
26
|
+
self.assertEqual(Alphabet.VOSK.value, "vosk")
|
|
27
|
+
self.assertEqual(len(list(Phonemizer)), 42)
|
|
28
|
+
self.assertEqual(len(list(Alphabet)), 22) # incl. MANTOQ, VOSK
|
|
27
29
|
|
|
28
30
|
|
|
29
31
|
class TestRegistryCompleteness(unittest.TestCase):
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""Tests for the Russian (vosk-tts) phonemizer and its vendored G2P rules."""
|
|
2
|
+
import os
|
|
3
|
+
import tempfile
|
|
4
|
+
import unittest
|
|
5
|
+
|
|
6
|
+
from scriptconv.phonemizers import (
|
|
7
|
+
Alphabet,
|
|
8
|
+
Phonemizer,
|
|
9
|
+
PHONEMIZER_REGISTRY,
|
|
10
|
+
get_phonemizer,
|
|
11
|
+
phonemize,
|
|
12
|
+
phonemizer_for_lang,
|
|
13
|
+
)
|
|
14
|
+
from scriptconv.phonemizers._thirdparty.vosk_g2p import convert, load_dictionary
|
|
15
|
+
from scriptconv.phonemizers.ru import VoskPhonemizer
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TestVoskRules(unittest.TestCase):
|
|
19
|
+
"""The vendored rules, exercised directly (no dictionary involved)."""
|
|
20
|
+
|
|
21
|
+
def test_palatalization_before_soft_vowel(self):
|
|
22
|
+
# рь/ви soften: r -> rj, v -> vj; unstressed vowels get the 0 digit
|
|
23
|
+
self.assertEqual(convert("привет"), "p rj i0 vj e0 t")
|
|
24
|
+
|
|
25
|
+
def test_hard_consonant_before_hard_vowel(self):
|
|
26
|
+
self.assertEqual(convert("как"), "k a0 k")
|
|
27
|
+
|
|
28
|
+
def test_plus_marks_the_next_vowel_as_stressed(self):
|
|
29
|
+
self.assertEqual(convert("прив+ет"), "p rj i0 vj e1 t")
|
|
30
|
+
# without the marker every vowel is unstressed
|
|
31
|
+
self.assertEqual(convert("привет"), "p rj i0 vj e0 t")
|
|
32
|
+
|
|
33
|
+
def test_stress_marker_only_affects_its_own_vowel(self):
|
|
34
|
+
self.assertEqual(convert("м+олоко"), "m o1 l o0 k o0")
|
|
35
|
+
self.assertEqual(convert("молок+о"), "m o0 l o0 k o1")
|
|
36
|
+
|
|
37
|
+
def test_multi_character_phonemes(self):
|
|
38
|
+
# щ -> sch, ш -> sh, ж -> zh, ч -> ch, ц -> c
|
|
39
|
+
self.assertEqual(convert("щука"), "sch u0 k a0")
|
|
40
|
+
self.assertEqual(convert("шар"), "sh a0 r")
|
|
41
|
+
self.assertEqual(convert("жук"), "zh u0 k")
|
|
42
|
+
self.assertEqual(convert("час"), "ch a0 s")
|
|
43
|
+
self.assertEqual(convert("цирк"), "c i0 r k")
|
|
44
|
+
|
|
45
|
+
def test_iotated_vowel_gains_a_glide_at_syllable_start(self):
|
|
46
|
+
self.assertEqual(convert("яма"), "j a0 m a0")
|
|
47
|
+
self.assertEqual(convert("ёж"), "j o0 zh")
|
|
48
|
+
# ... but not after a consonant, where it palatalizes instead
|
|
49
|
+
self.assertEqual(convert("тётя"), "tj o0 tj a0")
|
|
50
|
+
|
|
51
|
+
def test_soft_and_hard_signs_are_dropped_from_the_stream(self):
|
|
52
|
+
out = convert("съешь")
|
|
53
|
+
self.assertEqual(out, "s j e0 sh")
|
|
54
|
+
self.assertNotIn("ь", out)
|
|
55
|
+
self.assertNotIn("ъ", out)
|
|
56
|
+
|
|
57
|
+
def test_empty_word_yields_empty_string(self):
|
|
58
|
+
self.assertEqual(convert(""), "")
|
|
59
|
+
|
|
60
|
+
def test_non_cyrillic_passes_through_verbatim(self):
|
|
61
|
+
self.assertEqual(convert("abc"), "a b c")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class TestLoadDictionary(unittest.TestCase):
|
|
65
|
+
def _write(self, content: str) -> str:
|
|
66
|
+
fd, path = tempfile.mkstemp(suffix=".dict")
|
|
67
|
+
with os.fdopen(fd, "w", encoding="utf-8") as f:
|
|
68
|
+
f.write(content)
|
|
69
|
+
self.addCleanup(os.remove, path)
|
|
70
|
+
return path
|
|
71
|
+
|
|
72
|
+
def test_missing_path_yields_empty_map(self):
|
|
73
|
+
self.assertEqual(load_dictionary(None), {})
|
|
74
|
+
self.assertEqual(load_dictionary("/nonexistent/vosk/dictionary"), {})
|
|
75
|
+
|
|
76
|
+
def test_legacy_layout_word_then_phonemes(self):
|
|
77
|
+
path = self._write("привет p rj i0 vj e1 t\nкак k a1 k\n")
|
|
78
|
+
dic = load_dictionary(path)
|
|
79
|
+
self.assertEqual(dic["привет"], ["p", "rj", "i0", "vj", "e1", "t"])
|
|
80
|
+
self.assertEqual(dic["как"], ["k", "a1", "k"])
|
|
81
|
+
|
|
82
|
+
def test_probability_layout_keeps_highest_variant(self):
|
|
83
|
+
path = self._write("что 0.2 ch t o1\nчто 0.8 sh t o1\n")
|
|
84
|
+
self.assertEqual(load_dictionary(path)["что"], ["sh", "t", "o1"])
|
|
85
|
+
|
|
86
|
+
def test_legacy_layout_keeps_first_variant(self):
|
|
87
|
+
path = self._write("что ch t o1\nчто sh t o1\n")
|
|
88
|
+
self.assertEqual(load_dictionary(path)["что"], ["ch", "t", "o1"])
|
|
89
|
+
|
|
90
|
+
def test_blank_and_short_lines_ignored(self):
|
|
91
|
+
path = self._write("\n\nдом d o1 m\nбезфонем\n \n")
|
|
92
|
+
dic = load_dictionary(path)
|
|
93
|
+
self.assertEqual(list(dic), ["дом"])
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class TestVoskPhonemizer(unittest.TestCase):
|
|
97
|
+
def setUp(self):
|
|
98
|
+
self.p = VoskPhonemizer()
|
|
99
|
+
|
|
100
|
+
def test_sentence_string(self):
|
|
101
|
+
self.assertEqual(self.p.phonemize_string("Привет, как дела?", "ru"),
|
|
102
|
+
"p rj i0 vj e0 t , k a0 k dj e0 l a0 ?")
|
|
103
|
+
|
|
104
|
+
def test_case_is_folded(self):
|
|
105
|
+
self.assertEqual(self.p.phonemize_string("ПРИВЕТ", "ru"),
|
|
106
|
+
self.p.phonemize_string("привет", "ru"))
|
|
107
|
+
|
|
108
|
+
def test_punctuation_and_space_survive_as_tokens(self):
|
|
109
|
+
toks = self.p.phonemize_to_list("да, нет!", "ru")
|
|
110
|
+
self.assertIn(",", toks)
|
|
111
|
+
self.assertIn(" ", toks)
|
|
112
|
+
self.assertEqual(toks[-1], "!")
|
|
113
|
+
|
|
114
|
+
def test_multi_char_tokens_stay_whole(self):
|
|
115
|
+
# the base class would split "sch" into s/c/h — this backend must not
|
|
116
|
+
self.assertIn("sch", self.p.phonemize_to_list("щука", "ru"))
|
|
117
|
+
|
|
118
|
+
def test_em_dash_becomes_a_hyphen_pause(self):
|
|
119
|
+
toks = self.p.phonemize_to_list("Москва — столица", "ru")
|
|
120
|
+
self.assertIn("-", toks)
|
|
121
|
+
self.assertNotIn("—", toks)
|
|
122
|
+
|
|
123
|
+
def test_phonemize_returns_one_list_per_sentence(self):
|
|
124
|
+
chunks = self.p.phonemize("Привет. Как дела?", "ru")
|
|
125
|
+
self.assertEqual(len(chunks), 2)
|
|
126
|
+
self.assertTrue(all(isinstance(c, list) for c in chunks))
|
|
127
|
+
self.assertEqual(chunks[0][:3], ["p", "rj", "i0"])
|
|
128
|
+
|
|
129
|
+
def test_empty_text_yields_no_chunks(self):
|
|
130
|
+
self.assertEqual(self.p.phonemize("", "ru"), [])
|
|
131
|
+
self.assertEqual(self.p.phonemize_to_list("", "ru"), [])
|
|
132
|
+
self.assertEqual(self.p.phonemize_string("", "ru"), "")
|
|
133
|
+
|
|
134
|
+
def test_punctuation_only_input(self):
|
|
135
|
+
self.assertEqual(self.p.phonemize_to_list("...", "ru"), [".", ".", "."])
|
|
136
|
+
|
|
137
|
+
def test_non_cyrillic_input_is_not_dropped(self):
|
|
138
|
+
# OOV latin text has no vosk rule; it must survive rather than vanish
|
|
139
|
+
self.assertEqual(self.p.phonemize_string("hello world", "ru"),
|
|
140
|
+
"h e l l o w o r l d")
|
|
141
|
+
|
|
142
|
+
def test_digits_survive_unnormalized(self):
|
|
143
|
+
# scriptconv performs no normalization of its own
|
|
144
|
+
self.assertIn("3", self.p.phonemize_to_list("3 кота", "ru"))
|
|
145
|
+
|
|
146
|
+
def test_normalizer_hook_runs_before_g2p(self):
|
|
147
|
+
p = VoskPhonemizer()
|
|
148
|
+
p.normalizer = lambda t, l: t.replace("3", "три")
|
|
149
|
+
self.assertEqual(p.phonemize("3", "ru"), [["t", "rj", "i0"]])
|
|
150
|
+
|
|
151
|
+
def test_rejects_unsupported_language(self):
|
|
152
|
+
with self.assertRaises(ValueError):
|
|
153
|
+
self.p.phonemize_string("hello", "en-US")
|
|
154
|
+
with self.assertRaises(ValueError):
|
|
155
|
+
self.p.phonemize("привет", "zh-CN")
|
|
156
|
+
|
|
157
|
+
def test_rejects_non_vosk_alphabet(self):
|
|
158
|
+
with self.assertRaises(ValueError):
|
|
159
|
+
VoskPhonemizer(alphabet=Alphabet.IPA)
|
|
160
|
+
|
|
161
|
+
def test_dictionary_overrides_the_rules(self):
|
|
162
|
+
fd, path = tempfile.mkstemp(suffix=".dict")
|
|
163
|
+
with os.fdopen(fd, "w", encoding="utf-8") as f:
|
|
164
|
+
f.write("привет 1.0 p rj i0 vj e1 t\n")
|
|
165
|
+
self.addCleanup(os.remove, path)
|
|
166
|
+
p = VoskPhonemizer(model=path)
|
|
167
|
+
# e1 (stressed) comes from the dictionary; the rules alone give e0
|
|
168
|
+
self.assertEqual(p.phonemize_string("привет", "ru"),
|
|
169
|
+
"p rj i0 vj e1 t")
|
|
170
|
+
# an out-of-dictionary word still falls back to the rules
|
|
171
|
+
self.assertEqual(p.phonemize_string("щука", "ru"), "sch u0 k a0")
|
|
172
|
+
|
|
173
|
+
def test_missing_dictionary_falls_back_to_rules(self):
|
|
174
|
+
p = VoskPhonemizer(model="/nonexistent/vosk/dictionary")
|
|
175
|
+
self.assertEqual(p.dictionary, {})
|
|
176
|
+
self.assertEqual(p.phonemize_string("привет", "ru"), "p rj i0 vj e0 t")
|
|
177
|
+
|
|
178
|
+
def test_dictionary_is_lazy(self):
|
|
179
|
+
p = VoskPhonemizer(model="/nonexistent/vosk/dictionary")
|
|
180
|
+
self.assertIsNone(p._dictionary)
|
|
181
|
+
p.dictionary
|
|
182
|
+
self.assertIsNotNone(p._dictionary)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
class TestVoskRegistration(unittest.TestCase):
|
|
186
|
+
def test_registered(self):
|
|
187
|
+
self.assertIn(Phonemizer.VOSK, PHONEMIZER_REGISTRY)
|
|
188
|
+
|
|
189
|
+
def test_get_phonemizer_builds_it(self):
|
|
190
|
+
p = get_phonemizer(Phonemizer.VOSK, Alphabet.VOSK)
|
|
191
|
+
self.assertIsInstance(p, VoskPhonemizer)
|
|
192
|
+
|
|
193
|
+
def test_model_threads_through_the_phonemizer_model_knob(self):
|
|
194
|
+
fd, path = tempfile.mkstemp(suffix=".dict")
|
|
195
|
+
with os.fdopen(fd, "w", encoding="utf-8") as f:
|
|
196
|
+
f.write("дом 1.0 d o1 m\n")
|
|
197
|
+
self.addCleanup(os.remove, path)
|
|
198
|
+
p = get_phonemizer(Phonemizer.VOSK, Alphabet.VOSK, model=path)
|
|
199
|
+
self.assertEqual(p.phonemize_string("дом", "ru"), "d o1 m")
|
|
200
|
+
|
|
201
|
+
def test_russian_default_for_vosk_alphabet(self):
|
|
202
|
+
p = phonemizer_for_lang("ru", alphabet=Alphabet.VOSK)
|
|
203
|
+
self.assertIsInstance(p, VoskPhonemizer)
|
|
204
|
+
self.assertIsInstance(phonemizer_for_lang("ru-RU", alphabet=Alphabet.VOSK),
|
|
205
|
+
VoskPhonemizer)
|
|
206
|
+
|
|
207
|
+
def test_vosk_never_selected_for_ipa(self):
|
|
208
|
+
# it emits its own inventory — requesting IPA must fall through
|
|
209
|
+
p = phonemizer_for_lang("ru", alphabet=Alphabet.IPA)
|
|
210
|
+
self.assertNotIsInstance(p, VoskPhonemizer)
|
|
211
|
+
|
|
212
|
+
def test_module_level_phonemize_facade(self):
|
|
213
|
+
self.assertEqual(phonemize("Привет", "ru", alphabet=Alphabet.VOSK),
|
|
214
|
+
"p rj i0 vj e0 t")
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
if __name__ == "__main__":
|
|
218
|
+
unittest.main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/hanja.tsv
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/ipa.csv
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/yale.csv
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/__init__.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/codeswitch.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/diacritize.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/dialectal.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/english_g2p.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/espeak.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/frontend.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/levantine_g2p.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/normalize.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/LICENSE.md
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/__init__.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/rulebook.txt
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/LICENSE.md
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/__init__.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/symbols.py
RENAMED
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/num2words.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|