scriptconv 0.0.4a10__tar.gz → 0.0.4a12__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scriptconv-0.0.4a10/scriptconv.egg-info → scriptconv-0.0.4a12}/PKG-INFO +1 -1
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/base.py +19 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/ja.py +7 -5
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/ko.py +2 -2
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/ru.py +12 -9
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/zh.py +2 -2
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/version.py +1 -1
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12/scriptconv.egg-info}/PKG-INFO +1 -1
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_phonemizers_cjk_ar.py +33 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_phonemizers_ru.py +15 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/LICENSE +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/README.md +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/pyproject.toml +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/requirements.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/__main__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/cangjie.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/conventions.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/data/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/data/cangjie5_tc.tsv.gz +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/diacritics.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/graph.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/notation.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/bw2ipa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/hangul2ipa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/aspiration.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/assimilation.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/double_coda.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/hanja.tsv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/ipa.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/neutralization.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/tensification.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/yale.csv +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/codeswitch.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/diacritize.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/dialectal.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/english_g2p.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/espeak.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/frontend.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/levantine_g2p.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/normalize.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/phoneme_inventory.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/vosk_g2p.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/zh_num.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/kog2p/LICENSE.md +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/kog2p/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/kog2p/rulebook.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/LICENSE.md +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/buck/__init__.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/buck/phonetise_buckwalter.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/buck/symbols.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/buck/tokenization.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/num2words.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/unicode_symbol2label.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/ar.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/en.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/enums.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/eu.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/fa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/gl.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/he.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/mul.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/mwl.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/o2ipa.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/pt.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/registry.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/shami.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/vi.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/py.typed +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/readings.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/scripts.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/translit.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv.egg-info/SOURCES.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv.egg-info/dependency_links.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv.egg-info/requires.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv.egg-info/top_level.txt +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/setup.cfg +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_arpa_stress.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_cangjie.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_cli.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_conventions.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_diacritics.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_diacritics_graph.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_errors_policy.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_examples.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_graph.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_notation.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_phonemizers_base.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_phonemizers_friendly_import_errors.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_readings.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_readings_zh.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_scripts.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_scripts_stressonnx_compat.py +0 -0
- {scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_translit.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scriptconv
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.4a12
|
|
4
4
|
Summary: Zero-dependency script & phoneme-notation core — ISO-15924 detection & metadata, IPA↔ARPABET/X-SAMPA/Lexique/Kirshenbaum/Cotovía/RFE, Buckwalter↔Arabic, Hangul→jamo, kana
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Project-URL: Homepage, https://github.com/TigreGotico/scriptconv
|
|
@@ -45,6 +45,25 @@ def _primary_subtag(lang: str) -> str:
|
|
|
45
45
|
return lang.lower().replace("_", "-").split("-")[0]
|
|
46
46
|
|
|
47
47
|
|
|
48
|
+
def _check_alphabet(phonemizer, alphabet: Alphabet,
|
|
49
|
+
supported: List[Alphabet]) -> None:
|
|
50
|
+
"""Reject an output alphabet a wrapper cannot emit, with a usable message.
|
|
51
|
+
|
|
52
|
+
Wrappers used to guard this with a bare ``assert``, which surfaced as an
|
|
53
|
+
``AssertionError`` carrying an empty message — the caller learned nothing
|
|
54
|
+
about which alphabet it asked for or which ones exist. That matters because
|
|
55
|
+
:func:`~scriptconv.phonemizers.registry.get_phonemizer` injects its own
|
|
56
|
+
``alphabet=`` default into every constructor that declares the parameter,
|
|
57
|
+
so an unsupported default is hit by ordinary registry use, not just by
|
|
58
|
+
callers passing an odd value.
|
|
59
|
+
"""
|
|
60
|
+
if alphabet not in supported:
|
|
61
|
+
raise ValueError(
|
|
62
|
+
f"{type(phonemizer).__name__} cannot emit "
|
|
63
|
+
f"{Alphabet(alphabet).value!r}; supported alphabets are "
|
|
64
|
+
f"{[Alphabet(a).value for a in supported]}")
|
|
65
|
+
|
|
66
|
+
|
|
48
67
|
class BasePhonemizer(metaclass=abc.ABCMeta):
|
|
49
68
|
def __init__(self, alphabet: Alphabet = Alphabet.UNICODE,
|
|
50
69
|
normalizer: Optional[Callable[[str, str], str]] = None):
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
from scriptconv.phonemizers.base import BasePhonemizer
|
|
1
|
+
from scriptconv.phonemizers.base import BasePhonemizer, _check_alphabet
|
|
2
2
|
from scriptconv.phonemizers.enums import Alphabet
|
|
3
3
|
|
|
4
4
|
class OpenJTaklPhonemizer(BasePhonemizer):
|
|
5
5
|
|
|
6
|
-
def __init__(self, alphabet=Alphabet.
|
|
7
|
-
|
|
6
|
+
def __init__(self, alphabet=Alphabet.HEPBURN):
|
|
7
|
+
_check_alphabet(self, alphabet, [Alphabet.HEPBURN, Alphabet.KANA])
|
|
8
8
|
import pyopenjtalk
|
|
9
9
|
self.g2p = pyopenjtalk.g2p
|
|
10
10
|
super().__init__(alphabet)
|
|
@@ -36,7 +36,8 @@ class OpenJTaklPhonemizer(BasePhonemizer):
|
|
|
36
36
|
class CutletPhonemizer(BasePhonemizer):
|
|
37
37
|
|
|
38
38
|
def __init__(self, alphabet=Alphabet.HEPBURN, use_foreign_spelling = False):
|
|
39
|
-
|
|
39
|
+
_check_alphabet(self, alphabet,
|
|
40
|
+
[Alphabet.HEPBURN, Alphabet.KUNREI, Alphabet.NIHON])
|
|
40
41
|
# If `use_foreign_spelling` is true, output will use the foreign spelling
|
|
41
42
|
# provided in a UniDic lemma when available. For example, "カツ" will
|
|
42
43
|
# become "cutlet" instead of "katsu".
|
|
@@ -78,7 +79,8 @@ class CutletPhonemizer(BasePhonemizer):
|
|
|
78
79
|
class PyKakasiPhonemizer(BasePhonemizer):
|
|
79
80
|
|
|
80
81
|
def __init__(self, alphabet=Alphabet.HEPBURN):
|
|
81
|
-
|
|
82
|
+
_check_alphabet(self, alphabet,
|
|
83
|
+
[Alphabet.HEPBURN, Alphabet.KANA, Alphabet.HIRA])
|
|
82
84
|
# kana, hira, hepburn
|
|
83
85
|
try:
|
|
84
86
|
import pykakasi
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
|
|
2
2
|
|
|
3
|
-
from scriptconv.phonemizers.base import BasePhonemizer
|
|
3
|
+
from scriptconv.phonemizers.base import BasePhonemizer, _check_alphabet
|
|
4
4
|
from scriptconv.phonemizers._thirdparty.hangul2ipa import hangul2ipa
|
|
5
5
|
from scriptconv.phonemizers.enums import Alphabet
|
|
6
6
|
|
|
@@ -9,7 +9,7 @@ class G2PKPhonemizer(BasePhonemizer):
|
|
|
9
9
|
|
|
10
10
|
def __init__(self, descriptive=True, group_vowels=True, to_syl=True,
|
|
11
11
|
alphabet=Alphabet.IPA):
|
|
12
|
-
|
|
12
|
+
_check_alphabet(self, alphabet, [Alphabet.IPA, Alphabet.HANGUL])
|
|
13
13
|
try:
|
|
14
14
|
from g2pk import G2p
|
|
15
15
|
except ImportError as e:
|
|
@@ -5,7 +5,7 @@ Currently one backend: the Vosk-TTS front-end, wrapping the vendored
|
|
|
5
5
|
Russian voices can be driven from text without the ``vosk-tts`` package.
|
|
6
6
|
"""
|
|
7
7
|
import re
|
|
8
|
-
from typing import List, Optional
|
|
8
|
+
from typing import Iterator, List, Optional
|
|
9
9
|
|
|
10
10
|
from quebra_frases import sentence_tokenize
|
|
11
11
|
|
|
@@ -86,25 +86,28 @@ class VoskPhonemizer(BasePhonemizer):
|
|
|
86
86
|
tokens.extend(convert(word).split())
|
|
87
87
|
return tokens
|
|
88
88
|
|
|
89
|
-
def
|
|
90
|
-
"""Sentence-level lists of Vosk phoneme tokens.
|
|
89
|
+
def phonemize_lazy(self, text: str, lang: str) -> Iterator[List[str]]:
|
|
90
|
+
"""Sentence-level lists of Vosk phoneme tokens, one sentence at a time.
|
|
91
91
|
|
|
92
92
|
Punctuation is preserved (it drives pausing); each sentence becomes one
|
|
93
93
|
synthesis chunk. Multi-character tokens (``sch``, ``bj``, ``a1``) stay
|
|
94
|
-
whole, so :meth:`BasePhonemizer.
|
|
95
|
-
deliberately bypassed
|
|
94
|
+
whole, so :meth:`BasePhonemizer.phonemize_lazy`'s per-character split of
|
|
95
|
+
the phoneme *string* is deliberately bypassed — splitting ``s h`` into
|
|
96
|
+
characters would later fold back into ``sh`` (ш), a different phoneme.
|
|
96
97
|
"""
|
|
97
98
|
self.get_lang(lang)
|
|
98
99
|
if not text:
|
|
99
|
-
return
|
|
100
|
+
return
|
|
100
101
|
if self.normalizer is not None:
|
|
101
102
|
text = self.normalizer(text, lang)
|
|
102
|
-
results: PhonemizedChunks = []
|
|
103
103
|
for sentence in sentence_tokenize(text):
|
|
104
104
|
tokens = self._g2p_tokens(sentence)
|
|
105
105
|
if tokens:
|
|
106
|
-
|
|
107
|
-
|
|
106
|
+
yield tokens
|
|
107
|
+
|
|
108
|
+
def phonemize(self, text: str, lang: str) -> PhonemizedChunks:
|
|
109
|
+
"""Sentence-level lists of Vosk phoneme tokens."""
|
|
110
|
+
return list(self.phonemize_lazy(text, lang))
|
|
108
111
|
|
|
109
112
|
def phonemize_to_list(self, text: str, lang: str) -> List[str]:
|
|
110
113
|
self.get_lang(lang)
|
|
@@ -3,7 +3,7 @@ import unicodedata
|
|
|
3
3
|
from typing import List
|
|
4
4
|
|
|
5
5
|
|
|
6
|
-
from scriptconv.phonemizers.base import BasePhonemizer
|
|
6
|
+
from scriptconv.phonemizers.base import BasePhonemizer, _check_alphabet
|
|
7
7
|
from scriptconv.phonemizers._thirdparty.zh_num import num2str
|
|
8
8
|
from scriptconv.phonemizers.enums import Alphabet
|
|
9
9
|
|
|
@@ -74,7 +74,7 @@ class BaseChinesePinyinPhonemizer(BasePhonemizer):
|
|
|
74
74
|
ipa (bool): Whether to convert pinyin to IPA.
|
|
75
75
|
jieba (bool): Whether to segment text using Jieba before phonemization.
|
|
76
76
|
"""
|
|
77
|
-
|
|
77
|
+
_check_alphabet(self, alphabet, [Alphabet.PINYIN, Alphabet.IPA])
|
|
78
78
|
super().__init__(alphabet)
|
|
79
79
|
self.jieba = jieba
|
|
80
80
|
self.retone = retone
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scriptconv
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.4a12
|
|
4
4
|
Summary: Zero-dependency script & phoneme-notation core — ISO-15924 detection & metadata, IPA↔ARPABET/X-SAMPA/Lexique/Kirshenbaum/Cotovía/RFE, Buckwalter↔Arabic, Hangul→jamo, kana
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Project-URL: Homepage, https://github.com/TigreGotico/scriptconv
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import unittest
|
|
2
2
|
|
|
3
3
|
from scriptconv.phonemizers import Phonemizer, get_phonemizer_class
|
|
4
|
+
from scriptconv.phonemizers.registry import get_phonemizer
|
|
4
5
|
|
|
5
6
|
|
|
6
7
|
class TestVendoredKorean(unittest.TestCase):
|
|
@@ -174,3 +175,35 @@ class TestMantoqTokensToIpa(unittest.TestCase):
|
|
|
174
175
|
def test_pretokenized_dbl_geminate(self):
|
|
175
176
|
from scriptconv.notation import mantoq_to_ipa
|
|
176
177
|
self.assertEqual(mantoq_to_ipa(["b", "_dbl_", "a"]), "bːa")
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
class TestUnsupportedAlphabetIsAClearError(unittest.TestCase):
|
|
181
|
+
"""A wrapper asked for an alphabet it cannot emit must say so.
|
|
182
|
+
|
|
183
|
+
``get_phonemizer`` injects its own ``alphabet=`` default (IPA) into every
|
|
184
|
+
constructor declaring the parameter, so the Japanese wrappers — which emit
|
|
185
|
+
romanizations, never IPA — were hit by ordinary registry use and raised a
|
|
186
|
+
bare ``AssertionError`` with an empty message.
|
|
187
|
+
"""
|
|
188
|
+
|
|
189
|
+
def test_registry_default_raises_valueerror_naming_the_alphabets(self):
|
|
190
|
+
from scriptconv.phonemizers.enums import Alphabet
|
|
191
|
+
for member in (Phonemizer.OPENJTALK, Phonemizer.CUTLET,
|
|
192
|
+
Phonemizer.PYKAKASI):
|
|
193
|
+
with self.subTest(phonemizer=member.value):
|
|
194
|
+
try:
|
|
195
|
+
get_phonemizer(member)
|
|
196
|
+
except ImportError:
|
|
197
|
+
self.skipTest(f"{member.value} backend not installed")
|
|
198
|
+
except ValueError as e:
|
|
199
|
+
self.assertIn("ipa", str(e))
|
|
200
|
+
self.assertIn(Alphabet.HEPBURN.value, str(e))
|
|
201
|
+
else:
|
|
202
|
+
self.fail("expected ValueError for an unemittable alphabet")
|
|
203
|
+
|
|
204
|
+
def test_openjtalk_default_alphabet_is_one_it_can_emit(self):
|
|
205
|
+
cls = get_phonemizer_class(Phonemizer.OPENJTALK)
|
|
206
|
+
import inspect
|
|
207
|
+
from scriptconv.phonemizers.enums import Alphabet
|
|
208
|
+
default = inspect.signature(cls.__init__).parameters["alphabet"].default
|
|
209
|
+
self.assertIn(default, (Alphabet.HEPBURN, Alphabet.KANA))
|
|
@@ -209,6 +209,21 @@ class TestVoskRegistration(unittest.TestCase):
|
|
|
209
209
|
p = phonemizer_for_lang("ru", alphabet=Alphabet.IPA)
|
|
210
210
|
self.assertNotIsInstance(p, VoskPhonemizer)
|
|
211
211
|
|
|
212
|
+
def test_phonemize_lazy_keeps_tokens_whole(self):
|
|
213
|
+
# the lazy path is what consumers stream through: it must yield the same
|
|
214
|
+
# whole phoneme tokens as phonemize(), never a per-character split
|
|
215
|
+
# ("s", "h" would otherwise fold back into "sh" — a different phoneme)
|
|
216
|
+
p = VoskPhonemizer()
|
|
217
|
+
text = "Сходить в кино. Счастье рядом!"
|
|
218
|
+
self.assertEqual(list(p.phonemize_lazy(text, "ru")), p.phonemize(text, "ru"))
|
|
219
|
+
flat = [t for chunk in p.phonemize_lazy(text, "ru") for t in chunk]
|
|
220
|
+
self.assertIn("s", flat)
|
|
221
|
+
self.assertIn("h", flat)
|
|
222
|
+
self.assertTrue(all(len(t) <= 3 for t in flat))
|
|
223
|
+
# every emitted token is a real phoneme/pause, not a bare character of one
|
|
224
|
+
self.assertNotIn("0", flat)
|
|
225
|
+
self.assertNotIn("1", flat)
|
|
226
|
+
|
|
212
227
|
def test_module_level_phonemize_facade(self):
|
|
213
228
|
self.assertEqual(phonemize("Привет", "ru", alphabet=Alphabet.VOSK),
|
|
214
229
|
"p rj i0 vj e0 t")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/hangul2ipa.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/hanja.tsv
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/ipa.csv
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/ko_tables/yale.csv
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/__init__.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/codeswitch.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/diacritize.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/dialectal.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/english_g2p.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/espeak.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/frontend.py
RENAMED
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_thirdparty/shami/normalize.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/kog2p/LICENSE.md
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/kog2p/__init__.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/kog2p/rulebook.txt
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/LICENSE.md
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/__init__.py
RENAMED
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/buck/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/buck/symbols.py
RENAMED
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/scriptconv/phonemizers/_vendored/mantoq/num2words.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scriptconv-0.0.4a10 → scriptconv-0.0.4a12}/tests/test_phonemizers_friendly_import_errors.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|