scriptconv 0.0.4a9__tar.gz → 0.0.4a10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. {scriptconv-0.0.4a9/scriptconv.egg-info → scriptconv-0.0.4a10}/PKG-INFO +3 -2
  2. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/README.md +2 -1
  3. scriptconv-0.0.4a10/scriptconv/phonemizers/_thirdparty/vosk_g2p.py +135 -0
  4. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/enums.py +2 -0
  5. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/registry.py +5 -0
  6. scriptconv-0.0.4a10/scriptconv/phonemizers/ru.py +114 -0
  7. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/version.py +1 -1
  8. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10/scriptconv.egg-info}/PKG-INFO +3 -2
  9. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/SOURCES.txt +3 -0
  10. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_phonemizers_base.py +4 -2
  11. scriptconv-0.0.4a10/tests/test_phonemizers_ru.py +218 -0
  12. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/LICENSE +0 -0
  13. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/pyproject.toml +0 -0
  14. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/requirements.txt +0 -0
  15. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/__init__.py +0 -0
  16. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/__main__.py +0 -0
  17. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/cangjie.py +0 -0
  18. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/conventions.py +0 -0
  19. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/data/__init__.py +0 -0
  20. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/data/cangjie5_tc.tsv.gz +0 -0
  21. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/diacritics.py +0 -0
  22. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/graph.py +0 -0
  23. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/notation.py +0 -0
  24. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/__init__.py +0 -0
  25. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/__init__.py +0 -0
  26. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/bw2ipa.py +0 -0
  27. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/hangul2ipa.py +0 -0
  28. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/aspiration.csv +0 -0
  29. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/assimilation.csv +0 -0
  30. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/double_coda.csv +0 -0
  31. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/hanja.tsv +0 -0
  32. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/ipa.csv +0 -0
  33. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/neutralization.csv +0 -0
  34. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/tensification.csv +0 -0
  35. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/ko_tables/yale.csv +0 -0
  36. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/__init__.py +0 -0
  37. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/codeswitch.py +0 -0
  38. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/diacritize.py +0 -0
  39. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/dialectal.py +0 -0
  40. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/english_g2p.py +0 -0
  41. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/espeak.py +0 -0
  42. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/frontend.py +0 -0
  43. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/levantine_g2p.py +0 -0
  44. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/normalize.py +0 -0
  45. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/shami/phoneme_inventory.py +0 -0
  46. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_thirdparty/zh_num.py +0 -0
  47. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/__init__.py +0 -0
  48. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/LICENSE.md +0 -0
  49. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/__init__.py +0 -0
  50. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/kog2p/rulebook.txt +0 -0
  51. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/LICENSE.md +0 -0
  52. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/__init__.py +0 -0
  53. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/__init__.py +0 -0
  54. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/phonetise_buckwalter.py +0 -0
  55. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/symbols.py +0 -0
  56. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/buck/tokenization.py +0 -0
  57. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/num2words.py +0 -0
  58. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/_vendored/mantoq/unicode_symbol2label.py +0 -0
  59. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/ar.py +0 -0
  60. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/base.py +0 -0
  61. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/en.py +0 -0
  62. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/eu.py +0 -0
  63. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/fa.py +0 -0
  64. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/gl.py +0 -0
  65. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/he.py +0 -0
  66. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/ja.py +0 -0
  67. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/ko.py +0 -0
  68. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/mul.py +0 -0
  69. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/mwl.py +0 -0
  70. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/o2ipa.py +0 -0
  71. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/pt.py +0 -0
  72. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/shami.py +0 -0
  73. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/vi.py +0 -0
  74. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/phonemizers/zh.py +0 -0
  75. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/py.typed +0 -0
  76. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/readings.py +0 -0
  77. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/scripts.py +0 -0
  78. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv/translit.py +0 -0
  79. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/dependency_links.txt +0 -0
  80. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/requires.txt +0 -0
  81. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/scriptconv.egg-info/top_level.txt +0 -0
  82. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/setup.cfg +0 -0
  83. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_arpa_stress.py +0 -0
  84. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_cangjie.py +0 -0
  85. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_cli.py +0 -0
  86. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_conventions.py +0 -0
  87. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_diacritics.py +0 -0
  88. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_diacritics_graph.py +0 -0
  89. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_errors_policy.py +0 -0
  90. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_examples.py +0 -0
  91. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_graph.py +0 -0
  92. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_notation.py +0 -0
  93. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_phonemizers_cjk_ar.py +0 -0
  94. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_phonemizers_friendly_import_errors.py +0 -0
  95. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_readings.py +0 -0
  96. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_readings_zh.py +0 -0
  97. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_scripts.py +0 -0
  98. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_scripts_stressonnx_compat.py +0 -0
  99. {scriptconv-0.0.4a9 → scriptconv-0.0.4a10}/tests/test_translit.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scriptconv
3
- Version: 0.0.4a9
3
+ Version: 0.0.4a10
4
4
  Summary: Zero-dependency script & phoneme-notation core — ISO-15924 detection & metadata, IPA↔ARPABET/X-SAMPA/Lexique/Kirshenbaum/Cotovía/RFE, Buckwalter↔Arabic, Hangul→jamo, kana
5
5
  License: Apache-2.0
6
6
  Project-URL: Homepage, https://github.com/TigreGotico/scriptconv
@@ -350,7 +350,8 @@ phonemize("hello", "en", override=Phonemizer.GRUUT)
350
350
 
351
351
  Defaults resolve in-house engines first: an explicit per-language chain
352
352
  (Arabic → arbtok, Basque → euskaphone, Mirandese, Portuguese → tugaphone,
353
- Hebrew → phonikud, Galician → Cotovía for its own notation), then
353
+ Hebrew → phonikud, Galician → Cotovía and Russian → vosk for their own
354
+ notations), then
354
355
  orthography2ipa wherever it has a language spec, then espeak as the last
355
356
  resort. Arabic never falls back past arbtok — a missing engine raises rather
356
357
  than silently degrading. Every backend resolves lazily; a missing package
@@ -223,7 +223,8 @@ phonemize("hello", "en", override=Phonemizer.GRUUT)
223
223
 
224
224
  Defaults resolve in-house engines first: an explicit per-language chain
225
225
  (Arabic → arbtok, Basque → euskaphone, Mirandese, Portuguese → tugaphone,
226
- Hebrew → phonikud, Galician → Cotovía for its own notation), then
226
+ Hebrew → phonikud, Galician → Cotovía and Russian → vosk for their own
227
+ notations), then
227
228
  orthography2ipa wherever it has a language spec, then espeak as the last
228
229
  resort. Arabic never falls back past arbtok — a missing engine raises rather
229
230
  than silently degrading. Every backend resolves lazily; a missing package
@@ -0,0 +1,135 @@
1
+ """
2
+ Russian grapheme-to-phoneme rules and pronunciation-dictionary loader for the
3
+ Vosk-TTS voices.
4
+
5
+ Vendored from ``vosk_tts/g2p.py`` — https://github.com/alphacep/vosk-tts
6
+ (Apache-2.0) — so the Vosk Russian front-end is available without the
7
+ ``vosk-tts`` package at runtime.
8
+
9
+ ``convert`` is a faithful port of ``vosk_tts/g2p.py``: it turns an (optionally
10
+ stress-marked) Russian word into the Vosk phoneme inventory — palatalised
11
+ consonants get a trailing ``j`` (``bj``, ``tj`` …), vowels carry a stress digit
12
+ (``a0`` unstressed, ``a1`` stressed). A ``+`` immediately before a vowel marks
13
+ it as stressed; without any ``+`` every vowel is emitted unstressed.
14
+
15
+ The shipped ``dictionary`` file (word → phonemes) overrides the rules for known
16
+ words; ``load_dictionary`` reads both the historical ``word phon…`` layout and
17
+ the newer ``word prob phon…`` layout (keeping the highest-probability variant).
18
+ The rules alone already produce usable Russian, so the dictionary is optional.
19
+ """
20
+ import os
21
+ from typing import Dict, List, Optional
22
+
23
+ # Cyrillic letters that soften the preceding consonant.
24
+ softletters = set(u"яёюиье")
25
+ # Contexts after which я/ю/е/ё gains a leading glide /j/.
26
+ startsyl = set(u"#ъьаяоёуюэеиы-")
27
+ # Markers dropped from the final phoneme stream.
28
+ others = set(["#", "+", "-", u"ь", u"ъ"])
29
+
30
+ softhard_cons = {
31
+ u"б": u"b", u"в": u"v", u"г": u"g", u"Г": u"g", u"д": u"d",
32
+ u"з": u"z", u"к": u"k", u"л": u"l", u"м": u"m", u"н": u"n",
33
+ u"п": u"p", u"р": u"r", u"с": u"s", u"т": u"t", u"ф": u"f",
34
+ u"х": u"h",
35
+ }
36
+
37
+ other_cons = {
38
+ u"ж": u"zh", u"ц": u"c", u"ч": u"ch", u"ш": u"sh",
39
+ u"щ": u"sch", u"й": u"j",
40
+ }
41
+
42
+ vowels = {
43
+ u"а": u"a", u"я": u"a", u"у": u"u", u"ю": u"u", u"о": u"o",
44
+ u"ё": u"o", u"э": u"e", u"е": u"e", u"и": u"i", u"ы": u"y",
45
+ }
46
+
47
+
48
+ def pallatize(phones: List[tuple]) -> None:
49
+ """In-place: map consonants to their (palatalised) phoneme, looking one
50
+ character ahead to decide whether a soft vowel follows."""
51
+ for i, phone in enumerate(phones[:-1]):
52
+ if phone[0] in softhard_cons:
53
+ if phones[i + 1][0] in softletters:
54
+ phones[i] = (softhard_cons[phone[0]] + "j", 0)
55
+ else:
56
+ phones[i] = (softhard_cons[phone[0]], 0)
57
+ if phone[0] in other_cons:
58
+ phones[i] = (other_cons[phone[0]], 0)
59
+
60
+
61
+ def convert_vowels(phones: List[tuple]) -> List[str]:
62
+ """Emit vowels with their stress digit, inserting a glide /j/ before
63
+ iotated vowels at syllable starts."""
64
+ new_phones: List[str] = []
65
+ prev = ""
66
+ for phone in phones:
67
+ if prev in startsyl:
68
+ if phone[0] in set(u"яюеё"):
69
+ new_phones.append("j")
70
+ if phone[0] in vowels:
71
+ new_phones.append(vowels[phone[0]] + str(phone[1]))
72
+ else:
73
+ new_phones.append(phone[0])
74
+ prev = phone[0]
75
+ return new_phones
76
+
77
+
78
+ def convert(stressword: str) -> str:
79
+ """Convert a (possibly ``+``-stress-marked) Russian word to a
80
+ space-separated Vosk phoneme string."""
81
+ phones = ("#" + stressword + "#")
82
+
83
+ # Assign stress marks: a '+' sets the stress flag for the next character.
84
+ stress_phones = []
85
+ stress = 0
86
+ for phone in phones:
87
+ if phone == "+":
88
+ stress = 1
89
+ else:
90
+ stress_phones.append((phone, stress))
91
+ stress = 0
92
+
93
+ pallatize(stress_phones)
94
+ phones = convert_vowels(stress_phones)
95
+ phones = [x for x in phones if x not in others]
96
+ return " ".join(phones)
97
+
98
+
99
+ def load_dictionary(path: Optional[str]) -> Dict[str, List[str]]:
100
+ """
101
+ Load a Vosk pronunciation dictionary into a ``word -> [phoneme, …]`` map.
102
+
103
+ Handles both file layouts:
104
+ - ``word phon1 phon2 …`` (older voices, e.g. 0.1)
105
+ - ``word prob phon1 phon2 …`` (newer voices; highest prob wins)
106
+
107
+ Returns an empty dict when ``path`` is falsy or missing — the rule-based
108
+ :func:`convert` fallback covers any out-of-dictionary word.
109
+ """
110
+ dic: Dict[str, List[str]] = {}
111
+ if not path or not os.path.isfile(path):
112
+ return dic
113
+ probs: Dict[str, float] = {}
114
+ with open(path, encoding="utf-8") as f:
115
+ for line in f:
116
+ parts = line.split()
117
+ if len(parts) < 2:
118
+ continue
119
+ word, rest = parts[0], parts[1:]
120
+ # The second column is a probability only when it parses as a float;
121
+ # a real phoneme (a0, sch, …) never does.
122
+ try:
123
+ prob = float(rest[0])
124
+ phones = rest[1:]
125
+ except ValueError:
126
+ prob = None
127
+ phones = rest
128
+ if not phones:
129
+ continue
130
+ if prob is None:
131
+ dic.setdefault(word, phones)
132
+ elif probs.get(word, -1.0) < prob:
133
+ dic[word] = phones
134
+ probs[word] = prob
135
+ return dic
@@ -30,6 +30,7 @@ class Alphabet(str, Enum):
30
30
  BUCKWALTER = "buckwalter"
31
31
  MANTOQ = "mantoq" # ar — Halabi Arabic-Phonetiser inventory # ar
32
32
  CANGJIE = "cangjie" # zh (Cangjie input method)
33
+ VOSK = "vosk" # ru — vosk-tts phoneme inventory (a0, bj, sch ...)
33
34
  GRAPHEMES = "graphemes" # plain text / grapheme input (user-side)
34
35
 
35
36
 
@@ -73,6 +74,7 @@ class Phonemizer(str, Enum):
73
74
  PYPINYIN = "pypinyin" # chinese
74
75
  XPINYIN = "xpinyin" # chinese
75
76
  JIEBA = "jieba" # chinese (not a real phonemizer!)
77
+ VOSK = "vosk" # russian (no ipa!)
76
78
  SHAMI = "shami" # Levantine Arabic / English code-switching (ShamiVITS)
77
79
  ARBTOK = "arbtok" # arabic (dialect-aware, undiacritized text; o2i lattice)
78
80
  EUSKAPHONE = "euskaphone" # basque (dialect-aware; o2i lattice)
@@ -71,6 +71,7 @@ PHONEMIZER_REGISTRY: Dict[Phonemizer, Tuple[str, str, Optional[str]]] = {
71
71
  _P.MANTOQ: (f"{_BASE}.ar", "MantoqPhonemizer", "ar-phonemizers"),
72
72
  _P.ARBTOK: (f"{_BASE}.ar", "ArbtokPhonemizer", "ar-phonemizers"),
73
73
  _P.SHAMI: (f"{_BASE}.shami", "ShamiPhonemizer", "shami"),
74
+ _P.VOSK: (f"{_BASE}.ru", "VoskPhonemizer", "phonemizers"),
74
75
  }
75
76
 
76
77
 
@@ -135,6 +136,7 @@ _EMITS: Dict[Phonemizer, Tuple[Alphabet, ...]] = {
135
136
  _P.TUGAPHONE: (Alphabet.IPA,),
136
137
  _P.PHONIKUD: (Alphabet.IPA,),
137
138
  _P.COTOVIA: (Alphabet.COTOVIA,),
139
+ _P.VOSK: (Alphabet.VOSK,),
138
140
  _P.ESPEAK: (Alphabet.IPA,),
139
141
  }
140
142
 
@@ -147,6 +149,9 @@ LANG_DEFAULTS: Dict[str, Tuple[Phonemizer, ...]] = {
147
149
  "pt": (_P.TUGAPHONE,),
148
150
  "he": (_P.PHONIKUD,),
149
151
  "gl": (_P.COTOVIA, _P.ORTHOGRAPHY2IPA, _P.ESPEAK),
152
+ # vosk emits its own inventory, so it is the Russian default only
153
+ # when that notation is requested (same shape as Cotovía above)
154
+ "ru": (_P.VOSK, _P.ORTHOGRAPHY2IPA, _P.ESPEAK),
150
155
  }
151
156
 
152
157
  # Languages whose explicit entry is exhaustive: no generic fallback beyond it.
@@ -0,0 +1,114 @@
1
+ """Russian phonemizers.
2
+
3
+ Currently one backend: the Vosk-TTS front-end, wrapping the vendored
4
+ :mod:`scriptconv.phonemizers._thirdparty.vosk_g2p` rules so the alphacep
5
+ Russian voices can be driven from text without the ``vosk-tts`` package.
6
+ """
7
+ import re
8
+ from typing import List, Optional
9
+
10
+ from quebra_frases import sentence_tokenize
11
+
12
+ from scriptconv.phonemizers.base import BasePhonemizer, PhonemizedChunks
13
+ from scriptconv.phonemizers.enums import Alphabet
14
+ from scriptconv.phonemizers._thirdparty.vosk_g2p import convert, load_dictionary
15
+
16
+ __all__ = ["VoskPhonemizer"]
17
+
18
+
19
+ class VoskPhonemizer(BasePhonemizer):
20
+ """
21
+ Russian phonemizer for the Vosk-TTS voices (alphacep).
22
+
23
+ It reproduces ``vosk_tts``'s grapheme-to-phoneme exactly: each word is
24
+ looked up in the voice's pronunciation ``dictionary`` (word -> phonemes),
25
+ falling back to the rule-based
26
+ :func:`~scriptconv.phonemizers._thirdparty.vosk_g2p.convert` for
27
+ out-of-dictionary words. Spaces and punctuation are kept as their own
28
+ tokens (Vosk feeds them to the model as short/long pauses); the BOS ``^`` /
29
+ EOS ``$`` markers and the inter-phoneme blanks belong to the consumer's
30
+ tokenizer, so they are *not* emitted here.
31
+
32
+ The output is the Vosk phoneme inventory (``a0``, ``bj``, ``sch`` …), not
33
+ IPA, so the only supported alphabet is :attr:`Alphabet.VOSK` — like
34
+ Cotovía, this backend is eligible only when its own notation is requested.
35
+
36
+ The dictionary is optional: without it the rules alone still produce usable
37
+ Russian (only the curated stress and exception entries are lost).
38
+ scriptconv never downloads anything — the caller resolves the file and
39
+ passes its path as ``model`` (the registry's ``phonemizer_model`` knob).
40
+
41
+ Args:
42
+ alphabet (Alphabet): must be :attr:`Alphabet.VOSK`.
43
+ model (Optional[str]): path to the voice's ``dictionary`` file. When
44
+ absent or missing, only the rule-based fallback is used.
45
+ """
46
+
47
+ # Matches the per-character split used by vosk_tts: spaces and punctuation
48
+ # are captured so they survive as standalone pause tokens.
49
+ _SPLIT = re.compile(r'([,.?!;:"() ])')
50
+
51
+ def __init__(self, alphabet: Alphabet = Alphabet.VOSK,
52
+ model: Optional[str] = None):
53
+ if alphabet != Alphabet.VOSK:
54
+ raise ValueError(
55
+ "VoskPhonemizer emits the vosk-tts phoneme inventory, not "
56
+ f"{Alphabet(alphabet).value!r} — use Alphabet.VOSK")
57
+ self._dict_path = model
58
+ self._dictionary: Optional[dict] = None # lazy: dictionaries are large
59
+ super().__init__(alphabet)
60
+
61
+ @property
62
+ def dictionary(self) -> dict:
63
+ """The loaded ``word -> [phoneme, …]`` map (empty without a path)."""
64
+ if self._dictionary is None:
65
+ self._dictionary = load_dictionary(self._dict_path)
66
+ return self._dictionary
67
+
68
+ @classmethod
69
+ def get_lang(cls, target_lang: str) -> str:
70
+ return cls.match_lang(target_lang, ["ru-RU"])
71
+
72
+ def _g2p_tokens(self, text: str) -> List[str]:
73
+ """Word/punctuation stream -> Vosk phoneme tokens (no BOS/EOS/blanks)."""
74
+ tokens: List[str] = []
75
+ # the em dash is a pause, and vosk only knows the ASCII hyphen
76
+ text = text.replace("—", "-")
77
+ for word in self._SPLIT.split(text.lower()):
78
+ if word == "":
79
+ continue
80
+ if self._SPLIT.match(word) or word == "-":
81
+ # space or punctuation: kept verbatim as a pause token
82
+ tokens.append(word)
83
+ elif word in self.dictionary:
84
+ tokens.extend(self.dictionary[word])
85
+ else:
86
+ tokens.extend(convert(word).split())
87
+ return tokens
88
+
89
+ def phonemize(self, text: str, lang: str) -> PhonemizedChunks:
90
+ """Sentence-level lists of Vosk phoneme tokens.
91
+
92
+ Punctuation is preserved (it drives pausing); each sentence becomes one
93
+ synthesis chunk. Multi-character tokens (``sch``, ``bj``, ``a1``) stay
94
+ whole, so :meth:`BasePhonemizer.phonemize`'s per-character split is
95
+ deliberately bypassed.
96
+ """
97
+ self.get_lang(lang)
98
+ if not text:
99
+ return []
100
+ if self.normalizer is not None:
101
+ text = self.normalizer(text, lang)
102
+ results: PhonemizedChunks = []
103
+ for sentence in sentence_tokenize(text):
104
+ tokens = self._g2p_tokens(sentence)
105
+ if tokens:
106
+ results.append(tokens)
107
+ return results
108
+
109
+ def phonemize_to_list(self, text: str, lang: str) -> List[str]:
110
+ self.get_lang(lang)
111
+ return self._g2p_tokens(text.lower())
112
+
113
+ def phonemize_string(self, text: str, lang: str) -> str:
114
+ return " ".join(self.phonemize_to_list(text, lang))
@@ -2,7 +2,7 @@
2
2
  VERSION_MAJOR = 0
3
3
  VERSION_MINOR = 0
4
4
  VERSION_BUILD = 4
5
- VERSION_ALPHA = 9
5
+ VERSION_ALPHA = 10
6
6
  # END_VERSION_BLOCK
7
7
 
8
8
  VERSION_STR = f"{VERSION_MAJOR}.{VERSION_MINOR}.{VERSION_BUILD}"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scriptconv
3
- Version: 0.0.4a9
3
+ Version: 0.0.4a10
4
4
  Summary: Zero-dependency script & phoneme-notation core — ISO-15924 detection & metadata, IPA↔ARPABET/X-SAMPA/Lexique/Kirshenbaum/Cotovía/RFE, Buckwalter↔Arabic, Hangul→jamo, kana
5
5
  License: Apache-2.0
6
6
  Project-URL: Homepage, https://github.com/TigreGotico/scriptconv
@@ -350,7 +350,8 @@ phonemize("hello", "en", override=Phonemizer.GRUUT)
350
350
 
351
351
  Defaults resolve in-house engines first: an explicit per-language chain
352
352
  (Arabic → arbtok, Basque → euskaphone, Mirandese, Portuguese → tugaphone,
353
- Hebrew → phonikud, Galician → Cotovía for its own notation), then
353
+ Hebrew → phonikud, Galician → Cotovía and Russian → vosk for their own
354
+ notations), then
354
355
  orthography2ipa wherever it has a language spec, then espeak as the last
355
356
  resort. Arabic never falls back past arbtok — a missing engine raises rather
356
357
  than silently degrading. Every backend resolves lazily; a missing package
@@ -37,12 +37,14 @@ scriptconv/phonemizers/mwl.py
37
37
  scriptconv/phonemizers/o2ipa.py
38
38
  scriptconv/phonemizers/pt.py
39
39
  scriptconv/phonemizers/registry.py
40
+ scriptconv/phonemizers/ru.py
40
41
  scriptconv/phonemizers/shami.py
41
42
  scriptconv/phonemizers/vi.py
42
43
  scriptconv/phonemizers/zh.py
43
44
  scriptconv/phonemizers/_thirdparty/__init__.py
44
45
  scriptconv/phonemizers/_thirdparty/bw2ipa.py
45
46
  scriptconv/phonemizers/_thirdparty/hangul2ipa.py
47
+ scriptconv/phonemizers/_thirdparty/vosk_g2p.py
46
48
  scriptconv/phonemizers/_thirdparty/zh_num.py
47
49
  scriptconv/phonemizers/_thirdparty/ko_tables/aspiration.csv
48
50
  scriptconv/phonemizers/_thirdparty/ko_tables/assimilation.csv
@@ -87,6 +89,7 @@ tests/test_notation.py
87
89
  tests/test_phonemizers_base.py
88
90
  tests/test_phonemizers_cjk_ar.py
89
91
  tests/test_phonemizers_friendly_import_errors.py
92
+ tests/test_phonemizers_ru.py
90
93
  tests/test_readings.py
91
94
  tests/test_readings_zh.py
92
95
  tests/test_scripts.py
@@ -21,9 +21,11 @@ class TestEnums(unittest.TestCase):
21
21
  self.assertEqual(Phonemizer.ARBTOK.value, "arbtok")
22
22
  self.assertEqual(Phonemizer.MIRANDESE.value, "mwl_phonemizer")
23
23
  self.assertEqual(Phonemizer.KOG2PK.value, "kog2p")
24
+ self.assertEqual(Phonemizer.VOSK.value, "vosk")
24
25
  self.assertEqual(Alphabet.XSAMPA.value, "x-sampa")
25
- self.assertEqual(len(list(Phonemizer)), 41)
26
- self.assertEqual(len(list(Alphabet)), 21) # incl. MANTOQ
26
+ self.assertEqual(Alphabet.VOSK.value, "vosk")
27
+ self.assertEqual(len(list(Phonemizer)), 42)
28
+ self.assertEqual(len(list(Alphabet)), 22) # incl. MANTOQ, VOSK
27
29
 
28
30
 
29
31
  class TestRegistryCompleteness(unittest.TestCase):
@@ -0,0 +1,218 @@
1
+ """Tests for the Russian (vosk-tts) phonemizer and its vendored G2P rules."""
2
+ import os
3
+ import tempfile
4
+ import unittest
5
+
6
+ from scriptconv.phonemizers import (
7
+ Alphabet,
8
+ Phonemizer,
9
+ PHONEMIZER_REGISTRY,
10
+ get_phonemizer,
11
+ phonemize,
12
+ phonemizer_for_lang,
13
+ )
14
+ from scriptconv.phonemizers._thirdparty.vosk_g2p import convert, load_dictionary
15
+ from scriptconv.phonemizers.ru import VoskPhonemizer
16
+
17
+
18
+ class TestVoskRules(unittest.TestCase):
19
+ """The vendored rules, exercised directly (no dictionary involved)."""
20
+
21
+ def test_palatalization_before_soft_vowel(self):
22
+ # рь/ви soften: r -> rj, v -> vj; unstressed vowels get the 0 digit
23
+ self.assertEqual(convert("привет"), "p rj i0 vj e0 t")
24
+
25
+ def test_hard_consonant_before_hard_vowel(self):
26
+ self.assertEqual(convert("как"), "k a0 k")
27
+
28
+ def test_plus_marks_the_next_vowel_as_stressed(self):
29
+ self.assertEqual(convert("прив+ет"), "p rj i0 vj e1 t")
30
+ # without the marker every vowel is unstressed
31
+ self.assertEqual(convert("привет"), "p rj i0 vj e0 t")
32
+
33
+ def test_stress_marker_only_affects_its_own_vowel(self):
34
+ self.assertEqual(convert("м+олоко"), "m o1 l o0 k o0")
35
+ self.assertEqual(convert("молок+о"), "m o0 l o0 k o1")
36
+
37
+ def test_multi_character_phonemes(self):
38
+ # щ -> sch, ш -> sh, ж -> zh, ч -> ch, ц -> c
39
+ self.assertEqual(convert("щука"), "sch u0 k a0")
40
+ self.assertEqual(convert("шар"), "sh a0 r")
41
+ self.assertEqual(convert("жук"), "zh u0 k")
42
+ self.assertEqual(convert("час"), "ch a0 s")
43
+ self.assertEqual(convert("цирк"), "c i0 r k")
44
+
45
+ def test_iotated_vowel_gains_a_glide_at_syllable_start(self):
46
+ self.assertEqual(convert("яма"), "j a0 m a0")
47
+ self.assertEqual(convert("ёж"), "j o0 zh")
48
+ # ... but not after a consonant, where it palatalizes instead
49
+ self.assertEqual(convert("тётя"), "tj o0 tj a0")
50
+
51
+ def test_soft_and_hard_signs_are_dropped_from_the_stream(self):
52
+ out = convert("съешь")
53
+ self.assertEqual(out, "s j e0 sh")
54
+ self.assertNotIn("ь", out)
55
+ self.assertNotIn("ъ", out)
56
+
57
+ def test_empty_word_yields_empty_string(self):
58
+ self.assertEqual(convert(""), "")
59
+
60
+ def test_non_cyrillic_passes_through_verbatim(self):
61
+ self.assertEqual(convert("abc"), "a b c")
62
+
63
+
64
+ class TestLoadDictionary(unittest.TestCase):
65
+ def _write(self, content: str) -> str:
66
+ fd, path = tempfile.mkstemp(suffix=".dict")
67
+ with os.fdopen(fd, "w", encoding="utf-8") as f:
68
+ f.write(content)
69
+ self.addCleanup(os.remove, path)
70
+ return path
71
+
72
+ def test_missing_path_yields_empty_map(self):
73
+ self.assertEqual(load_dictionary(None), {})
74
+ self.assertEqual(load_dictionary("/nonexistent/vosk/dictionary"), {})
75
+
76
+ def test_legacy_layout_word_then_phonemes(self):
77
+ path = self._write("привет p rj i0 vj e1 t\nкак k a1 k\n")
78
+ dic = load_dictionary(path)
79
+ self.assertEqual(dic["привет"], ["p", "rj", "i0", "vj", "e1", "t"])
80
+ self.assertEqual(dic["как"], ["k", "a1", "k"])
81
+
82
+ def test_probability_layout_keeps_highest_variant(self):
83
+ path = self._write("что 0.2 ch t o1\nчто 0.8 sh t o1\n")
84
+ self.assertEqual(load_dictionary(path)["что"], ["sh", "t", "o1"])
85
+
86
+ def test_legacy_layout_keeps_first_variant(self):
87
+ path = self._write("что ch t o1\nчто sh t o1\n")
88
+ self.assertEqual(load_dictionary(path)["что"], ["ch", "t", "o1"])
89
+
90
+ def test_blank_and_short_lines_ignored(self):
91
+ path = self._write("\n\nдом d o1 m\nбезфонем\n \n")
92
+ dic = load_dictionary(path)
93
+ self.assertEqual(list(dic), ["дом"])
94
+
95
+
96
+ class TestVoskPhonemizer(unittest.TestCase):
97
+ def setUp(self):
98
+ self.p = VoskPhonemizer()
99
+
100
+ def test_sentence_string(self):
101
+ self.assertEqual(self.p.phonemize_string("Привет, как дела?", "ru"),
102
+ "p rj i0 vj e0 t , k a0 k dj e0 l a0 ?")
103
+
104
+ def test_case_is_folded(self):
105
+ self.assertEqual(self.p.phonemize_string("ПРИВЕТ", "ru"),
106
+ self.p.phonemize_string("привет", "ru"))
107
+
108
+ def test_punctuation_and_space_survive_as_tokens(self):
109
+ toks = self.p.phonemize_to_list("да, нет!", "ru")
110
+ self.assertIn(",", toks)
111
+ self.assertIn(" ", toks)
112
+ self.assertEqual(toks[-1], "!")
113
+
114
+ def test_multi_char_tokens_stay_whole(self):
115
+ # the base class would split "sch" into s/c/h — this backend must not
116
+ self.assertIn("sch", self.p.phonemize_to_list("щука", "ru"))
117
+
118
+ def test_em_dash_becomes_a_hyphen_pause(self):
119
+ toks = self.p.phonemize_to_list("Москва — столица", "ru")
120
+ self.assertIn("-", toks)
121
+ self.assertNotIn("—", toks)
122
+
123
+ def test_phonemize_returns_one_list_per_sentence(self):
124
+ chunks = self.p.phonemize("Привет. Как дела?", "ru")
125
+ self.assertEqual(len(chunks), 2)
126
+ self.assertTrue(all(isinstance(c, list) for c in chunks))
127
+ self.assertEqual(chunks[0][:3], ["p", "rj", "i0"])
128
+
129
+ def test_empty_text_yields_no_chunks(self):
130
+ self.assertEqual(self.p.phonemize("", "ru"), [])
131
+ self.assertEqual(self.p.phonemize_to_list("", "ru"), [])
132
+ self.assertEqual(self.p.phonemize_string("", "ru"), "")
133
+
134
+ def test_punctuation_only_input(self):
135
+ self.assertEqual(self.p.phonemize_to_list("...", "ru"), [".", ".", "."])
136
+
137
+ def test_non_cyrillic_input_is_not_dropped(self):
138
+ # OOV latin text has no vosk rule; it must survive rather than vanish
139
+ self.assertEqual(self.p.phonemize_string("hello world", "ru"),
140
+ "h e l l o w o r l d")
141
+
142
+ def test_digits_survive_unnormalized(self):
143
+ # scriptconv performs no normalization of its own
144
+ self.assertIn("3", self.p.phonemize_to_list("3 кота", "ru"))
145
+
146
+ def test_normalizer_hook_runs_before_g2p(self):
147
+ p = VoskPhonemizer()
148
+ p.normalizer = lambda t, l: t.replace("3", "три")
149
+ self.assertEqual(p.phonemize("3", "ru"), [["t", "rj", "i0"]])
150
+
151
+ def test_rejects_unsupported_language(self):
152
+ with self.assertRaises(ValueError):
153
+ self.p.phonemize_string("hello", "en-US")
154
+ with self.assertRaises(ValueError):
155
+ self.p.phonemize("привет", "zh-CN")
156
+
157
+ def test_rejects_non_vosk_alphabet(self):
158
+ with self.assertRaises(ValueError):
159
+ VoskPhonemizer(alphabet=Alphabet.IPA)
160
+
161
+ def test_dictionary_overrides_the_rules(self):
162
+ fd, path = tempfile.mkstemp(suffix=".dict")
163
+ with os.fdopen(fd, "w", encoding="utf-8") as f:
164
+ f.write("привет 1.0 p rj i0 vj e1 t\n")
165
+ self.addCleanup(os.remove, path)
166
+ p = VoskPhonemizer(model=path)
167
+ # e1 (stressed) comes from the dictionary; the rules alone give e0
168
+ self.assertEqual(p.phonemize_string("привет", "ru"),
169
+ "p rj i0 vj e1 t")
170
+ # an out-of-dictionary word still falls back to the rules
171
+ self.assertEqual(p.phonemize_string("щука", "ru"), "sch u0 k a0")
172
+
173
+ def test_missing_dictionary_falls_back_to_rules(self):
174
+ p = VoskPhonemizer(model="/nonexistent/vosk/dictionary")
175
+ self.assertEqual(p.dictionary, {})
176
+ self.assertEqual(p.phonemize_string("привет", "ru"), "p rj i0 vj e0 t")
177
+
178
+ def test_dictionary_is_lazy(self):
179
+ p = VoskPhonemizer(model="/nonexistent/vosk/dictionary")
180
+ self.assertIsNone(p._dictionary)
181
+ p.dictionary
182
+ self.assertIsNotNone(p._dictionary)
183
+
184
+
185
+ class TestVoskRegistration(unittest.TestCase):
186
+ def test_registered(self):
187
+ self.assertIn(Phonemizer.VOSK, PHONEMIZER_REGISTRY)
188
+
189
+ def test_get_phonemizer_builds_it(self):
190
+ p = get_phonemizer(Phonemizer.VOSK, Alphabet.VOSK)
191
+ self.assertIsInstance(p, VoskPhonemizer)
192
+
193
+ def test_model_threads_through_the_phonemizer_model_knob(self):
194
+ fd, path = tempfile.mkstemp(suffix=".dict")
195
+ with os.fdopen(fd, "w", encoding="utf-8") as f:
196
+ f.write("дом 1.0 d o1 m\n")
197
+ self.addCleanup(os.remove, path)
198
+ p = get_phonemizer(Phonemizer.VOSK, Alphabet.VOSK, model=path)
199
+ self.assertEqual(p.phonemize_string("дом", "ru"), "d o1 m")
200
+
201
+ def test_russian_default_for_vosk_alphabet(self):
202
+ p = phonemizer_for_lang("ru", alphabet=Alphabet.VOSK)
203
+ self.assertIsInstance(p, VoskPhonemizer)
204
+ self.assertIsInstance(phonemizer_for_lang("ru-RU", alphabet=Alphabet.VOSK),
205
+ VoskPhonemizer)
206
+
207
+ def test_vosk_never_selected_for_ipa(self):
208
+ # it emits its own inventory — requesting IPA must fall through
209
+ p = phonemizer_for_lang("ru", alphabet=Alphabet.IPA)
210
+ self.assertNotIsInstance(p, VoskPhonemizer)
211
+
212
+ def test_module_level_phonemize_facade(self):
213
+ self.assertEqual(phonemize("Привет", "ru", alphabet=Alphabet.VOSK),
214
+ "p rj i0 vj e0 t")
215
+
216
+
217
+ if __name__ == "__main__":
218
+ unittest.main()
File without changes
File without changes