py-pinyin-split 2.0.0__tar.gz → 3.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/.gitignore +1 -0
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/PKG-INFO +18 -14
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/README.md +13 -9
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/pyproject.toml +29 -25
- {py_pinyin_split-2.0.0/src/pinyin_split → py_pinyin_split-3.0.2/src/py_pinyin_split}/__about__.py +1 -1
- py_pinyin_split-2.0.0/src/pinyin_split/pinyinsplit.py → py_pinyin_split-3.0.2/src/py_pinyin_split/__init__.py +86 -63
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/tests/test_pinyinsplit.py +40 -12
- py_pinyin_split-3.0.2/uv.lock +437 -0
- py_pinyin_split-2.0.0/src/pinyin_split/__init__.py +0 -5
- py_pinyin_split-2.0.0/uv.lock +0 -202
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/.github/workflows/publish.yml +0 -0
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/.python-version +0 -0
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/LICENSE.txt +0 -0
- {py_pinyin_split-2.0.0/src/pinyin_split → py_pinyin_split-3.0.2/src/py_pinyin_split}/py.typed +0 -0
- {py_pinyin_split-2.0.0 → py_pinyin_split-3.0.2}/tests/__init__.py +0 -0
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: py-pinyin-split
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.0.2
|
|
4
4
|
Summary: Library for splitting Hanyu Pinyin phrases into all valid syllable combinations
|
|
5
|
-
Project-URL: Documentation, https://github.com/lstrobel/
|
|
6
|
-
Project-URL: Issues, https://github.com/lstrobel/
|
|
7
|
-
Project-URL: Source, https://github.com/lstrobel/
|
|
5
|
+
Project-URL: Documentation, https://github.com/lstrobel/py-pinyin-split#readme
|
|
6
|
+
Project-URL: Issues, https://github.com/lstrobel/py-pinyin-split/issues
|
|
7
|
+
Project-URL: Source, https://github.com/lstrobel/py-pinyin-split
|
|
8
8
|
Author-email: lstrobel <mail@lstrobel.com>, Thomas Lee <thomaslee@throput.com>
|
|
9
9
|
Maintainer-email: lstrobel <mail@lstrobel.com>
|
|
10
10
|
License: MIT
|
|
@@ -24,8 +24,8 @@ Classifier: Programming Language :: Python :: Implementation :: PyPy
|
|
|
24
24
|
Classifier: Topic :: Text Processing :: Linguistic
|
|
25
25
|
Classifier: Topic :: Utilities
|
|
26
26
|
Requires-Python: <3.13,>=3.8
|
|
27
|
+
Requires-Dist: marisa-trie>=1.2.1
|
|
27
28
|
Requires-Dist: nltk==3.9.1
|
|
28
|
-
Requires-Dist: pygtrie>=2.5.0
|
|
29
29
|
Description-Content-Type: text/markdown
|
|
30
30
|
|
|
31
31
|
# py-pinyin-split
|
|
@@ -47,16 +47,12 @@ pip install py-pinyin-split
|
|
|
47
47
|
|
|
48
48
|
Instantiate a tokenizer and split away.
|
|
49
49
|
|
|
50
|
-
|
|
51
|
-
- Remove punctuation and whitespace
|
|
52
|
-
- Convert numeric tones (pin1yin1) to tone marks (pīnyīn)
|
|
53
|
-
- Handle sentence boundaries
|
|
54
|
-
|
|
55
|
-
The tokenizer uses syllable frequency data to resolve ambiguous splits. It currently does not support apostrophes (sorry!) (e.g. "xi'an" will throw an error)
|
|
50
|
+
The tokenizer can handle standard Hanyu Pinyin with whitespaces and punctuation. However, invalid pinyin syllables will raise a `ValueError`
|
|
56
51
|
|
|
52
|
+
The tokenizer uses syllable frequency data to resolve ambiguous splits.
|
|
57
53
|
|
|
58
54
|
```python
|
|
59
|
-
from
|
|
55
|
+
from py_pinyin_split import PinyinTokenizer
|
|
60
56
|
|
|
61
57
|
tokenizer = PinyinTokenizer()
|
|
62
58
|
|
|
@@ -64,13 +60,21 @@ tokenizer = PinyinTokenizer()
|
|
|
64
60
|
tokenizer.tokenize("nǐhǎo") # ['nǐ', 'hǎo']
|
|
65
61
|
tokenizer.tokenize("Běijīng") # ['Běi', 'jīng']
|
|
66
62
|
|
|
63
|
+
# Handles whitespace and punctuation
|
|
64
|
+
tokenizer.tokenize("Nǐ hǎo ma?") # ['Nǐ', 'hǎo', 'ma', '?']
|
|
65
|
+
tokenizer.tokenize("Wǒ hěn hǎo!") # ['Wǒ', 'hěn', 'hǎo', '!']
|
|
66
|
+
|
|
67
67
|
# Handles ambiguous splits using frequency data
|
|
68
68
|
tokenizer.tokenize("xian") # ['xian'] not ['xi', 'an']
|
|
69
69
|
tokenizer.tokenize("wanan") # ['wan', 'an'] not ['wa', 'nan']
|
|
70
70
|
|
|
71
|
-
# Tone marks help resolve ambiguity
|
|
71
|
+
# Tone marks or punctuation help resolve ambiguity
|
|
72
72
|
tokenizer.tokenize("xīān") # ['xī', 'ān']
|
|
73
73
|
tokenizer.tokenize("xián") # ['xián']
|
|
74
|
+
tokenizer.tokenize("Xī'ān") # ["Xī", "'", "ān"]
|
|
75
|
+
|
|
76
|
+
# Raises ValueError for invalid pinyin
|
|
77
|
+
tokenizer.tokenize("hello") # ValueError
|
|
74
78
|
|
|
75
79
|
# Optional support for non-standard syllables
|
|
76
80
|
tokenizer = PinyinTokenizer(include_nonstandard=True)
|
|
@@ -80,4 +84,4 @@ tokenizer.tokenize("duang") # ['duang']
|
|
|
80
84
|
## Related Projects
|
|
81
85
|
- https://pypi.org/project/pinyintokenizer/
|
|
82
86
|
- https://pypi.org/project/pypinyin/
|
|
83
|
-
- https://github.com/throput/pinyinsplit
|
|
87
|
+
- https://github.com/throput/pinyinsplit
|
|
@@ -17,16 +17,12 @@ pip install py-pinyin-split
|
|
|
17
17
|
|
|
18
18
|
Instantiate a tokenizer and split away.
|
|
19
19
|
|
|
20
|
-
|
|
21
|
-
- Remove punctuation and whitespace
|
|
22
|
-
- Convert numeric tones (pin1yin1) to tone marks (pīnyīn)
|
|
23
|
-
- Handle sentence boundaries
|
|
24
|
-
|
|
25
|
-
The tokenizer uses syllable frequency data to resolve ambiguous splits. It currently does not support apostrophes (sorry!) (e.g. "xi'an" will throw an error)
|
|
20
|
+
The tokenizer can handle standard Hanyu Pinyin with whitespaces and punctuation. However, invalid pinyin syllables will raise a `ValueError`
|
|
26
21
|
|
|
22
|
+
The tokenizer uses syllable frequency data to resolve ambiguous splits.
|
|
27
23
|
|
|
28
24
|
```python
|
|
29
|
-
from
|
|
25
|
+
from py_pinyin_split import PinyinTokenizer
|
|
30
26
|
|
|
31
27
|
tokenizer = PinyinTokenizer()
|
|
32
28
|
|
|
@@ -34,13 +30,21 @@ tokenizer = PinyinTokenizer()
|
|
|
34
30
|
tokenizer.tokenize("nǐhǎo") # ['nǐ', 'hǎo']
|
|
35
31
|
tokenizer.tokenize("Běijīng") # ['Běi', 'jīng']
|
|
36
32
|
|
|
33
|
+
# Handles whitespace and punctuation
|
|
34
|
+
tokenizer.tokenize("Nǐ hǎo ma?") # ['Nǐ', 'hǎo', 'ma', '?']
|
|
35
|
+
tokenizer.tokenize("Wǒ hěn hǎo!") # ['Wǒ', 'hěn', 'hǎo', '!']
|
|
36
|
+
|
|
37
37
|
# Handles ambiguous splits using frequency data
|
|
38
38
|
tokenizer.tokenize("xian") # ['xian'] not ['xi', 'an']
|
|
39
39
|
tokenizer.tokenize("wanan") # ['wan', 'an'] not ['wa', 'nan']
|
|
40
40
|
|
|
41
|
-
# Tone marks help resolve ambiguity
|
|
41
|
+
# Tone marks or punctuation help resolve ambiguity
|
|
42
42
|
tokenizer.tokenize("xīān") # ['xī', 'ān']
|
|
43
43
|
tokenizer.tokenize("xián") # ['xián']
|
|
44
|
+
tokenizer.tokenize("Xī'ān") # ["Xī", "'", "ān"]
|
|
45
|
+
|
|
46
|
+
# Raises ValueError for invalid pinyin
|
|
47
|
+
tokenizer.tokenize("hello") # ValueError
|
|
44
48
|
|
|
45
49
|
# Optional support for non-standard syllables
|
|
46
50
|
tokenizer = PinyinTokenizer(include_nonstandard=True)
|
|
@@ -50,4 +54,4 @@ tokenizer.tokenize("duang") # ['duang']
|
|
|
50
54
|
## Related Projects
|
|
51
55
|
- https://pypi.org/project/pinyintokenizer/
|
|
52
56
|
- https://pypi.org/project/pypinyin/
|
|
53
|
-
- https://github.com/throput/pinyinsplit
|
|
57
|
+
- https://github.com/throput/pinyinsplit
|
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
requires = ["hatchling"]
|
|
3
3
|
build-backend = "hatchling.build"
|
|
4
4
|
|
|
5
|
-
[tool.hatch.
|
|
6
|
-
|
|
5
|
+
[tool.hatch.version]
|
|
6
|
+
path = "src/py_pinyin_split/__about__.py"
|
|
7
7
|
|
|
8
8
|
[project]
|
|
9
9
|
name = "py-pinyin-split"
|
|
@@ -34,33 +34,37 @@ classifiers = [
|
|
|
34
34
|
'Topic :: Text Processing :: Linguistic',
|
|
35
35
|
'Topic :: Utilities',
|
|
36
36
|
]
|
|
37
|
-
dependencies = ["
|
|
37
|
+
dependencies = ["marisa-trie>=1.2.1", "nltk==3.9.1"]
|
|
38
38
|
|
|
39
39
|
[dependency-groups]
|
|
40
|
-
dev = [
|
|
40
|
+
dev = [
|
|
41
|
+
"mypy>=1.13.0",
|
|
42
|
+
"pytest-benchmark>=4.0.0",
|
|
43
|
+
"pytest>=8.3.4",
|
|
44
|
+
"ruff>=0.7.4",
|
|
45
|
+
]
|
|
41
46
|
|
|
42
47
|
[project.urls]
|
|
43
|
-
Documentation = "https://github.com/lstrobel/
|
|
44
|
-
Issues = "https://github.com/lstrobel/
|
|
45
|
-
Source = "https://github.com/lstrobel/
|
|
46
|
-
|
|
47
|
-
[tool.hatch.version]
|
|
48
|
-
path = "src/pinyin_split/__about__.py"
|
|
49
|
-
|
|
50
|
-
[tool.hatch.envs.types]
|
|
51
|
-
extra-dependencies = ["mypy>=1.0.0"]
|
|
52
|
-
[tool.hatch.envs.types.scripts]
|
|
53
|
-
check = "mypy --install-types --non-interactive {args:src/pinyin_split tests}"
|
|
48
|
+
Documentation = "https://github.com/lstrobel/py-pinyin-split#readme"
|
|
49
|
+
Issues = "https://github.com/lstrobel/py-pinyin-split/issues"
|
|
50
|
+
Source = "https://github.com/lstrobel/py-pinyin-split"
|
|
54
51
|
|
|
55
|
-
[tool.coverage.run]
|
|
56
|
-
source_pkgs = ["pinyin_split", "tests"]
|
|
57
|
-
branch = true
|
|
58
|
-
parallel = true
|
|
59
|
-
omit = ["src/pinyin_split/__about__.py"]
|
|
60
52
|
|
|
61
|
-
[tool.
|
|
62
|
-
|
|
63
|
-
tests = ["tests", "*/pinyin-split/tests"]
|
|
53
|
+
[tool.ruff.format]
|
|
54
|
+
docstring-code-format = true
|
|
64
55
|
|
|
65
|
-
[tool.
|
|
66
|
-
|
|
56
|
+
[tool.ruff.lint]
|
|
57
|
+
select = [
|
|
58
|
+
# pycodestyle
|
|
59
|
+
"E",
|
|
60
|
+
# Pyflakes
|
|
61
|
+
"F",
|
|
62
|
+
# pyupgrade
|
|
63
|
+
"UP",
|
|
64
|
+
# flake8-bugbear
|
|
65
|
+
"B",
|
|
66
|
+
# flake8-simplify
|
|
67
|
+
"SIM",
|
|
68
|
+
# isort
|
|
69
|
+
"I",
|
|
70
|
+
]
|
|
@@ -1,17 +1,23 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: 2024-present Lukas Strobel <mail@lstrobel.com>
|
|
2
|
+
#
|
|
3
|
+
# SPDX-License-Identifier: MIT
|
|
4
|
+
|
|
1
5
|
import re
|
|
2
|
-
import string
|
|
3
6
|
from typing import Iterator, List, Tuple
|
|
4
|
-
|
|
5
|
-
|
|
7
|
+
|
|
8
|
+
import marisa_trie # type: ignore
|
|
9
|
+
from nltk.tokenize import WordPunctTokenizer # type: ignore
|
|
10
|
+
from nltk.tokenize.api import TokenizerI # type: ignore
|
|
6
11
|
|
|
7
12
|
|
|
8
13
|
class PinyinTokenizer(TokenizerI):
|
|
9
14
|
"""
|
|
10
|
-
Splits Hanyu Pinyin words on syllable boundaries.
|
|
15
|
+
Splits Hanyu Pinyin words on syllable boundaries.
|
|
16
|
+
Cannot handle punctuation or whitespace.
|
|
11
17
|
|
|
12
18
|
Args:
|
|
13
|
-
include_nonstandard: If True, includes rare/non-standard syllables in the valid
|
|
14
|
-
Defaults to False.
|
|
19
|
+
include_nonstandard: If True, includes rare/non-standard syllables in the valid
|
|
20
|
+
set. Defaults to False.
|
|
15
21
|
|
|
16
22
|
Example:
|
|
17
23
|
>>> tokenizer = PinyinTokenizer()
|
|
@@ -83,7 +89,7 @@ class PinyinTokenizer(TokenizerI):
|
|
|
83
89
|
'chi', 'cha', 'che', 'chai', 'chao', 'chou', 'chan', 'chen', 'chang', 'cheng',
|
|
84
90
|
'chu', 'chua', 'chuo', 'chuai', 'chui', 'chuan', 'chun', 'chuang', 'chong',
|
|
85
91
|
|
|
86
|
-
'shi', 'sha', 'she', 'shai', 'shei', 'shao', 'shou', 'shan', 'shen', 'shang', 'sheng',
|
|
92
|
+
'shi', 'sha', 'she', 'shai', 'shei', 'shao', 'shou', 'shan', 'shen', 'shang', 'sheng', # noqa: E501
|
|
87
93
|
'shu', 'shua', 'shuo', 'shuai', 'shui', 'shuan', 'shun', 'shuang',
|
|
88
94
|
|
|
89
95
|
'r', # Include to cover erhua
|
|
@@ -533,21 +539,12 @@ class PinyinTokenizer(TokenizerI):
|
|
|
533
539
|
"zuo": "893830",
|
|
534
540
|
}
|
|
535
541
|
|
|
536
|
-
|
|
537
|
-
# TODO: Handle this more gracefully
|
|
538
|
-
_ALLOWED_CHARS = set(string.ascii_letters)
|
|
539
|
-
for vowel, variants in VOWEL_TONE_VARIANTS.items():
|
|
540
|
-
_ALLOWED_CHARS.update(vowel) # Add lowercase
|
|
541
|
-
_ALLOWED_CHARS.update(vowel.upper()) # Add uppercase
|
|
542
|
-
for tone_char in variants:
|
|
543
|
-
_ALLOWED_CHARS.update(tone_char)
|
|
544
|
-
_ALLOWED_CHARS.update(tone_char.upper())
|
|
545
|
-
|
|
546
|
-
# Compile regex pattern for invalid character detection once at module level
|
|
547
|
-
VALID_CHARS_PATTERN = re.compile(f"[^{''.join(sorted(_ALLOWED_CHARS))}]")
|
|
542
|
+
NON_ALPHABETIC_REGEX = re.compile(r"[^\w\s]+")
|
|
548
543
|
|
|
549
544
|
def _get_tone_variants(self, syllable: str):
|
|
550
|
-
"""
|
|
545
|
+
"""
|
|
546
|
+
Generate all valid tone variants for a syllable. Assumes syllabe is lowercase
|
|
547
|
+
"""
|
|
551
548
|
variants = [syllable] # Include toneless variation
|
|
552
549
|
|
|
553
550
|
# Find the vowels in the syllable (both upper and lower case)
|
|
@@ -565,7 +562,8 @@ class PinyinTokenizer(TokenizerI):
|
|
|
565
562
|
elif any(v == "o" for v in vowels):
|
|
566
563
|
tone_vowel = next(v for v in vowels if v == "o")
|
|
567
564
|
else:
|
|
568
|
-
# If there is no a e or o, the vowels are either 'iu', 'ui', or 'ê',
|
|
565
|
+
# If there is no a e or o, the vowels are either 'iu', 'ui', or 'ê',
|
|
566
|
+
# in which case the mark goes on the last vowel
|
|
569
567
|
tone_vowel = vowels[-1]
|
|
570
568
|
|
|
571
569
|
# Generate variants with each tone mark
|
|
@@ -578,27 +576,31 @@ class PinyinTokenizer(TokenizerI):
|
|
|
578
576
|
return variants
|
|
579
577
|
|
|
580
578
|
def __init__(self, include_nonstandard=False):
|
|
581
|
-
self.
|
|
582
|
-
# Add standard syllables
|
|
579
|
+
self.preprocess_tokenizer = WordPunctTokenizer()
|
|
583
580
|
|
|
581
|
+
trie_contents = []
|
|
582
|
+
|
|
583
|
+
# Add standard syllables
|
|
584
584
|
for syllable in self.STANDARD_SYLLABLES:
|
|
585
585
|
for variant in self._get_tone_variants(syllable):
|
|
586
|
-
|
|
586
|
+
trie_contents.append(variant)
|
|
587
587
|
|
|
588
588
|
if include_nonstandard:
|
|
589
589
|
for syllable in self.NON_STANDARD_SYLLABLES:
|
|
590
590
|
for variant in self._get_tone_variants(syllable):
|
|
591
|
-
|
|
591
|
+
trie_contents.append(variant)
|
|
592
592
|
|
|
593
|
-
|
|
594
|
-
# Check for any invalid characters
|
|
595
|
-
if self.VALID_CHARS_PATTERN.search(s):
|
|
596
|
-
raise ValueError("Input string can only contain letters and tone marks")
|
|
593
|
+
self.trie = marisa_trie.Trie(trie_contents)
|
|
597
594
|
|
|
595
|
+
def _get_string_possibilites(self, s) -> List[List[Tuple[int, int]]]:
|
|
596
|
+
"""
|
|
597
|
+
For a given string, return all possible valid syllable spans of that string,
|
|
598
|
+
indexed local to the string
|
|
599
|
+
"""
|
|
598
600
|
candidates = []
|
|
599
601
|
|
|
600
602
|
# Generate all possible splits starting from the beginning
|
|
601
|
-
to_process = [(0, [])]
|
|
603
|
+
to_process: List[Tuple[int, List[int]]] = [(0, [])]
|
|
602
604
|
while to_process:
|
|
603
605
|
# Get next position and accumulated split indices to process
|
|
604
606
|
start_pos, split_indices = to_process.pop()
|
|
@@ -608,14 +610,15 @@ class PinyinTokenizer(TokenizerI):
|
|
|
608
610
|
prefix_matches = self.trie.prefixes(remaining_s)
|
|
609
611
|
|
|
610
612
|
# For each possible syllable length
|
|
611
|
-
for
|
|
613
|
+
for match in prefix_matches:
|
|
612
614
|
# Create a new split point list with this syllable's endpoint
|
|
613
615
|
new_splits = split_indices.copy()
|
|
614
|
-
new_splits.append(start_pos +
|
|
616
|
+
new_splits.append(start_pos + len(match))
|
|
615
617
|
|
|
616
|
-
if start_pos +
|
|
617
|
-
# If we haven't reached the end,
|
|
618
|
-
|
|
618
|
+
if start_pos + len(match) < len(s):
|
|
619
|
+
# If we haven't reached the end,
|
|
620
|
+
# continue processing from end of this syllable
|
|
621
|
+
to_process.append((start_pos + len(match), new_splits))
|
|
619
622
|
else:
|
|
620
623
|
# We've reached the end - construct the output span tuples
|
|
621
624
|
spans = []
|
|
@@ -625,35 +628,55 @@ class PinyinTokenizer(TokenizerI):
|
|
|
625
628
|
prev = pos
|
|
626
629
|
candidates.append(spans)
|
|
627
630
|
|
|
628
|
-
|
|
629
|
-
raise ValueError(f"No valid pinyin syllable splits found for '{s}'")
|
|
630
|
-
|
|
631
|
-
# Find the shortest candidate(s) by comparing total number of splits
|
|
632
|
-
min_length = min(len(splits) for splits in candidates)
|
|
633
|
-
shortest = [c for c in candidates if len(c) == min_length]
|
|
631
|
+
return candidates
|
|
634
632
|
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
)
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
633
|
+
def span_tokenize(self, s: str) -> Iterator[Tuple[int, int]]:
|
|
634
|
+
# Start by splitting into sub-spans to examine
|
|
635
|
+
starting_spans = self.preprocess_tokenizer.span_tokenize(s)
|
|
636
|
+
|
|
637
|
+
final_spans = []
|
|
638
|
+
for start, end in starting_spans:
|
|
639
|
+
subspan_possibilities = self._get_string_possibilites(s[start:end])
|
|
640
|
+
if not subspan_possibilities:
|
|
641
|
+
# We were passed a subspan that wasnt valid pinyin
|
|
642
|
+
# If it is a non-alphabetic span, then we return as-is
|
|
643
|
+
if self.NON_ALPHABETIC_REGEX.match(s[start:end]):
|
|
644
|
+
final_spans.append((start, end))
|
|
645
|
+
continue
|
|
646
|
+
else:
|
|
647
|
+
raise ValueError(f"Invalid pinyin at substring: {s[start:end]}")
|
|
648
|
+
|
|
649
|
+
# Find the shortest candidate(s) for this span
|
|
650
|
+
min_length = min(len(splits) for splits in subspan_possibilities)
|
|
651
|
+
shortest = [c for c in subspan_possibilities if len(c) == min_length]
|
|
652
|
+
|
|
653
|
+
if len(shortest) > 1:
|
|
654
|
+
# Use syllable frequencies as tiebreaker
|
|
655
|
+
max_freq = float("-inf")
|
|
656
|
+
best_split = None
|
|
657
|
+
|
|
658
|
+
for split in shortest:
|
|
659
|
+
# Get syllables without tones and sum their frequencies
|
|
660
|
+
syllables = [s[start:end].lower() for start, end in split]
|
|
661
|
+
total_freq = sum(
|
|
662
|
+
int(self.SYLLABLE_FREQUENCIES.get(syl, "0"))
|
|
663
|
+
for syl in syllables
|
|
664
|
+
)
|
|
665
|
+
|
|
666
|
+
if total_freq > max_freq:
|
|
667
|
+
max_freq = total_freq
|
|
668
|
+
best_split = split
|
|
669
|
+
|
|
670
|
+
assert best_split is not None
|
|
671
|
+
chosen = best_split
|
|
672
|
+
else:
|
|
673
|
+
chosen = shortest[0]
|
|
674
|
+
|
|
675
|
+
for subspan in chosen:
|
|
676
|
+
# Convert from local span indices to full string indices
|
|
677
|
+
final_spans.append((start + subspan[0], start + subspan[1]))
|
|
678
|
+
|
|
679
|
+
yield from final_spans
|
|
657
680
|
|
|
658
681
|
def tokenize(self, s: str) -> List[str]:
|
|
659
682
|
return [s[start:end] for start, end in self.span_tokenize(s)]
|
|
@@ -1,5 +1,12 @@
|
|
|
1
|
-
import pytest
|
|
2
|
-
|
|
1
|
+
import pytest
|
|
2
|
+
|
|
3
|
+
from py_pinyin_split import PinyinTokenizer
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_benchmark_simple(benchmark):
|
|
7
|
+
tokenizer = PinyinTokenizer()
|
|
8
|
+
result = benchmark(tokenizer.tokenize, "xīnniánkuàilè")
|
|
9
|
+
assert result == ["xīn", "nián", "kuài", "lè"]
|
|
3
10
|
|
|
4
11
|
|
|
5
12
|
def test_no_tone_splits():
|
|
@@ -22,18 +29,10 @@ def test_tone_splits():
|
|
|
22
29
|
assert tokenizer.tokenize("màn") == ["màn"]
|
|
23
30
|
|
|
24
31
|
|
|
25
|
-
def
|
|
26
|
-
"""
|
|
32
|
+
def test_invalid_pinyin():
|
|
33
|
+
"""Inputs with invalid pinyin throw ValueErrors"""
|
|
27
34
|
tokenizer = PinyinTokenizer()
|
|
28
35
|
|
|
29
|
-
# Empty string should raise ValueError
|
|
30
|
-
with pytest.raises(ValueError):
|
|
31
|
-
tokenizer.tokenize("")
|
|
32
|
-
|
|
33
|
-
# Invalid characters should raise ValueError
|
|
34
|
-
with pytest.raises(ValueError):
|
|
35
|
-
tokenizer.tokenize("hello!")
|
|
36
|
-
|
|
37
36
|
# Single consonant should raise ValueError
|
|
38
37
|
with pytest.raises(ValueError):
|
|
39
38
|
tokenizer.tokenize("x")
|
|
@@ -47,6 +46,30 @@ def test_edge_cases():
|
|
|
47
46
|
tokenizer.tokenize("ni3hao3")
|
|
48
47
|
|
|
49
48
|
|
|
49
|
+
def test_text_with_whitespace():
|
|
50
|
+
"""Test handling of longer text with whitespace"""
|
|
51
|
+
tokenizer = PinyinTokenizer()
|
|
52
|
+
|
|
53
|
+
# Simple whitespace
|
|
54
|
+
assert tokenizer.tokenize("nǐ hǎo") == ["nǐ", "hǎo"]
|
|
55
|
+
|
|
56
|
+
# Multiple words with mixed tones
|
|
57
|
+
assert tokenizer.tokenize("Wǒ hěn xǐhuān Zhōngguó") == [
|
|
58
|
+
"Wǒ",
|
|
59
|
+
"hěn",
|
|
60
|
+
"xǐ",
|
|
61
|
+
"huān",
|
|
62
|
+
"Zhōng",
|
|
63
|
+
"guó",
|
|
64
|
+
]
|
|
65
|
+
|
|
66
|
+
# Leading/trailing whitespace
|
|
67
|
+
assert tokenizer.tokenize(" nǐ hǎo ") == ["nǐ", "hǎo"]
|
|
68
|
+
|
|
69
|
+
# Multiple whitespace characters
|
|
70
|
+
assert tokenizer.tokenize("nǐ hǎo\t\nma") == ["nǐ", "hǎo", "ma"]
|
|
71
|
+
|
|
72
|
+
|
|
50
73
|
def test_nonstandard_syllables():
|
|
51
74
|
"""Test handling of non-standard syllables"""
|
|
52
75
|
standard_tokenizer = PinyinTokenizer(include_nonstandard=False)
|
|
@@ -108,3 +131,8 @@ def test_ambiguous_splits():
|
|
|
108
131
|
# Test that tones help resolve ambiguity
|
|
109
132
|
assert tokenizer.tokenize("xīan") == ["xī", "an"]
|
|
110
133
|
assert tokenizer.tokenize("xián") == ["xián"]
|
|
134
|
+
|
|
135
|
+
# Test apostrophe handling
|
|
136
|
+
assert tokenizer.tokenize("Xī'ān") == ["Xī", "'", "ān"]
|
|
137
|
+
assert tokenizer.tokenize("yī'er") == ["yī", "'", "er"]
|
|
138
|
+
assert tokenizer.tokenize("tián'é") == ["tián", "'", "é"]
|