py-pinyin-split 2.0.0__tar.gz → 3.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -20,6 +20,7 @@ coverage.xml
20
20
  .hypothesis/
21
21
  .pytest_cache/
22
22
  cover/
23
+ .benchmarks/
23
24
 
24
25
  # Virtual environments
25
26
  .venv
@@ -1,10 +1,10 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: py-pinyin-split
3
- Version: 2.0.0
3
+ Version: 3.0.2
4
4
  Summary: Library for splitting Hanyu Pinyin phrases into all valid syllable combinations
5
- Project-URL: Documentation, https://github.com/lstrobel/pinyinsplit#readme
6
- Project-URL: Issues, https://github.com/lstrobel/pinyinsplit/issues
7
- Project-URL: Source, https://github.com/lstrobel/pinyinsplit
5
+ Project-URL: Documentation, https://github.com/lstrobel/py-pinyin-split#readme
6
+ Project-URL: Issues, https://github.com/lstrobel/py-pinyin-split/issues
7
+ Project-URL: Source, https://github.com/lstrobel/py-pinyin-split
8
8
  Author-email: lstrobel <mail@lstrobel.com>, Thomas Lee <thomaslee@throput.com>
9
9
  Maintainer-email: lstrobel <mail@lstrobel.com>
10
10
  License: MIT
@@ -24,8 +24,8 @@ Classifier: Programming Language :: Python :: Implementation :: PyPy
24
24
  Classifier: Topic :: Text Processing :: Linguistic
25
25
  Classifier: Topic :: Utilities
26
26
  Requires-Python: <3.13,>=3.8
27
+ Requires-Dist: marisa-trie>=1.2.1
27
28
  Requires-Dist: nltk==3.9.1
28
- Requires-Dist: pygtrie>=2.5.0
29
29
  Description-Content-Type: text/markdown
30
30
 
31
31
  # py-pinyin-split
@@ -47,16 +47,12 @@ pip install py-pinyin-split
47
47
 
48
48
  Instantiate a tokenizer and split away.
49
49
 
50
- The tokenizer expects a clean Hanyu Pinyin word as input - you'll need to preprocess text to:
51
- - Remove punctuation and whitespace
52
- - Convert numeric tones (pin1yin1) to tone marks (pīnyīn)
53
- - Handle sentence boundaries
54
-
55
- The tokenizer uses syllable frequency data to resolve ambiguous splits. It currently does not support apostrophes (sorry!) (e.g. "xi'an" will throw an error)
50
+ The tokenizer can handle standard Hanyu Pinyin with whitespaces and punctuation. However, invalid pinyin syllables will raise a `ValueError`
56
51
 
52
+ The tokenizer uses syllable frequency data to resolve ambiguous splits.
57
53
 
58
54
  ```python
59
- from pinyin_split import PinyinTokenizer
55
+ from py_pinyin_split import PinyinTokenizer
60
56
 
61
57
  tokenizer = PinyinTokenizer()
62
58
 
@@ -64,13 +60,21 @@ tokenizer = PinyinTokenizer()
64
60
  tokenizer.tokenize("nǐhǎo") # ['nǐ', 'hǎo']
65
61
  tokenizer.tokenize("Běijīng") # ['Běi', 'jīng']
66
62
 
63
+ # Handles whitespace and punctuation
64
+ tokenizer.tokenize("Nǐ hǎo ma?") # ['Nǐ', 'hǎo', 'ma', '?']
65
+ tokenizer.tokenize("Wǒ hěn hǎo!") # ['Wǒ', 'hěn', 'hǎo', '!']
66
+
67
67
  # Handles ambiguous splits using frequency data
68
68
  tokenizer.tokenize("xian") # ['xian'] not ['xi', 'an']
69
69
  tokenizer.tokenize("wanan") # ['wan', 'an'] not ['wa', 'nan']
70
70
 
71
- # Tone marks help resolve ambiguity
71
+ # Tone marks or punctuation help resolve ambiguity
72
72
  tokenizer.tokenize("xīān") # ['xī', 'ān']
73
73
  tokenizer.tokenize("xián") # ['xián']
74
+ tokenizer.tokenize("Xī'ān") # ["Xī", "'", "ān"]
75
+
76
+ # Raises ValueError for invalid pinyin
77
+ tokenizer.tokenize("hello") # ValueError
74
78
 
75
79
  # Optional support for non-standard syllables
76
80
  tokenizer = PinyinTokenizer(include_nonstandard=True)
@@ -80,4 +84,4 @@ tokenizer.tokenize("duang") # ['duang']
80
84
  ## Related Projects
81
85
  - https://pypi.org/project/pinyintokenizer/
82
86
  - https://pypi.org/project/pypinyin/
83
- - https://github.com/throput/pinyinsplit
87
+ - https://github.com/throput/pinyinsplit
@@ -17,16 +17,12 @@ pip install py-pinyin-split
17
17
 
18
18
  Instantiate a tokenizer and split away.
19
19
 
20
- The tokenizer expects a clean Hanyu Pinyin word as input - you'll need to preprocess text to:
21
- - Remove punctuation and whitespace
22
- - Convert numeric tones (pin1yin1) to tone marks (pīnyīn)
23
- - Handle sentence boundaries
24
-
25
- The tokenizer uses syllable frequency data to resolve ambiguous splits. It currently does not support apostrophes (sorry!) (e.g. "xi'an" will throw an error)
20
+ The tokenizer can handle standard Hanyu Pinyin with whitespaces and punctuation. However, invalid pinyin syllables will raise a `ValueError`
26
21
 
22
+ The tokenizer uses syllable frequency data to resolve ambiguous splits.
27
23
 
28
24
  ```python
29
- from pinyin_split import PinyinTokenizer
25
+ from py_pinyin_split import PinyinTokenizer
30
26
 
31
27
  tokenizer = PinyinTokenizer()
32
28
 
@@ -34,13 +30,21 @@ tokenizer = PinyinTokenizer()
34
30
  tokenizer.tokenize("nǐhǎo") # ['nǐ', 'hǎo']
35
31
  tokenizer.tokenize("Běijīng") # ['Běi', 'jīng']
36
32
 
33
+ # Handles whitespace and punctuation
34
+ tokenizer.tokenize("Nǐ hǎo ma?") # ['Nǐ', 'hǎo', 'ma', '?']
35
+ tokenizer.tokenize("Wǒ hěn hǎo!") # ['Wǒ', 'hěn', 'hǎo', '!']
36
+
37
37
  # Handles ambiguous splits using frequency data
38
38
  tokenizer.tokenize("xian") # ['xian'] not ['xi', 'an']
39
39
  tokenizer.tokenize("wanan") # ['wan', 'an'] not ['wa', 'nan']
40
40
 
41
- # Tone marks help resolve ambiguity
41
+ # Tone marks or punctuation help resolve ambiguity
42
42
  tokenizer.tokenize("xīān") # ['xī', 'ān']
43
43
  tokenizer.tokenize("xián") # ['xián']
44
+ tokenizer.tokenize("Xī'ān") # ["Xī", "'", "ān"]
45
+
46
+ # Raises ValueError for invalid pinyin
47
+ tokenizer.tokenize("hello") # ValueError
44
48
 
45
49
  # Optional support for non-standard syllables
46
50
  tokenizer = PinyinTokenizer(include_nonstandard=True)
@@ -50,4 +54,4 @@ tokenizer.tokenize("duang") # ['duang']
50
54
  ## Related Projects
51
55
  - https://pypi.org/project/pinyintokenizer/
52
56
  - https://pypi.org/project/pypinyin/
53
- - https://github.com/throput/pinyinsplit
57
+ - https://github.com/throput/pinyinsplit
@@ -2,8 +2,8 @@
2
2
  requires = ["hatchling"]
3
3
  build-backend = "hatchling.build"
4
4
 
5
- [tool.hatch.build.targets.wheel]
6
- packages = ["src/pinyin_split"]
5
+ [tool.hatch.version]
6
+ path = "src/py_pinyin_split/__about__.py"
7
7
 
8
8
  [project]
9
9
  name = "py-pinyin-split"
@@ -34,33 +34,37 @@ classifiers = [
34
34
  'Topic :: Text Processing :: Linguistic',
35
35
  'Topic :: Utilities',
36
36
  ]
37
- dependencies = ["nltk==3.9.1", "pygtrie>=2.5.0"]
37
+ dependencies = ["marisa-trie>=1.2.1", "nltk==3.9.1"]
38
38
 
39
39
  [dependency-groups]
40
- dev = ["ruff>=0.7.4"]
40
+ dev = [
41
+ "mypy>=1.13.0",
42
+ "pytest-benchmark>=4.0.0",
43
+ "pytest>=8.3.4",
44
+ "ruff>=0.7.4",
45
+ ]
41
46
 
42
47
  [project.urls]
43
- Documentation = "https://github.com/lstrobel/pinyinsplit#readme"
44
- Issues = "https://github.com/lstrobel/pinyinsplit/issues"
45
- Source = "https://github.com/lstrobel/pinyinsplit"
46
-
47
- [tool.hatch.version]
48
- path = "src/pinyin_split/__about__.py"
49
-
50
- [tool.hatch.envs.types]
51
- extra-dependencies = ["mypy>=1.0.0"]
52
- [tool.hatch.envs.types.scripts]
53
- check = "mypy --install-types --non-interactive {args:src/pinyin_split tests}"
48
+ Documentation = "https://github.com/lstrobel/py-pinyin-split#readme"
49
+ Issues = "https://github.com/lstrobel/py-pinyin-split/issues"
50
+ Source = "https://github.com/lstrobel/py-pinyin-split"
54
51
 
55
- [tool.coverage.run]
56
- source_pkgs = ["pinyin_split", "tests"]
57
- branch = true
58
- parallel = true
59
- omit = ["src/pinyin_split/__about__.py"]
60
52
 
61
- [tool.coverage.paths]
62
- pinyin_split = ["src/pinyin_split", "*/pinyin-split/src/pinyin_split"]
63
- tests = ["tests", "*/pinyin-split/tests"]
53
+ [tool.ruff.format]
54
+ docstring-code-format = true
64
55
 
65
- [tool.coverage.report]
66
- exclude_lines = ["no cov", "if __name__ == .__main__.:", "if TYPE_CHECKING:"]
56
+ [tool.ruff.lint]
57
+ select = [
58
+ # pycodestyle
59
+ "E",
60
+ # Pyflakes
61
+ "F",
62
+ # pyupgrade
63
+ "UP",
64
+ # flake8-bugbear
65
+ "B",
66
+ # flake8-simplify
67
+ "SIM",
68
+ # isort
69
+ "I",
70
+ ]
@@ -1,4 +1,4 @@
1
1
  # SPDX-FileCopyrightText: 2024-present Lukas Strobel <mail@lstrobel.com>
2
2
  #
3
3
  # SPDX-License-Identifier: MIT
4
- __version__ = "2.0.0"
4
+ __version__ = "3.0.2"
@@ -1,17 +1,23 @@
1
+ # SPDX-FileCopyrightText: 2024-present Lukas Strobel <mail@lstrobel.com>
2
+ #
3
+ # SPDX-License-Identifier: MIT
4
+
1
5
  import re
2
- import string
3
6
  from typing import Iterator, List, Tuple
4
- from nltk.tokenize.api import TokenizerI
5
- from pygtrie import CharTrie
7
+
8
+ import marisa_trie # type: ignore
9
+ from nltk.tokenize import WordPunctTokenizer # type: ignore
10
+ from nltk.tokenize.api import TokenizerI # type: ignore
6
11
 
7
12
 
8
13
  class PinyinTokenizer(TokenizerI):
9
14
  """
10
- Splits Hanyu Pinyin words on syllable boundaries. Cannot handle punctuation or whitespace.
15
+ Splits Hanyu Pinyin words on syllable boundaries.
16
+ Cannot handle punctuation or whitespace.
11
17
 
12
18
  Args:
13
- include_nonstandard: If True, includes rare/non-standard syllables in the valid set.
14
- Defaults to False.
19
+ include_nonstandard: If True, includes rare/non-standard syllables in the valid
20
+ set. Defaults to False.
15
21
 
16
22
  Example:
17
23
  >>> tokenizer = PinyinTokenizer()
@@ -83,7 +89,7 @@ class PinyinTokenizer(TokenizerI):
83
89
  'chi', 'cha', 'che', 'chai', 'chao', 'chou', 'chan', 'chen', 'chang', 'cheng',
84
90
  'chu', 'chua', 'chuo', 'chuai', 'chui', 'chuan', 'chun', 'chuang', 'chong',
85
91
 
86
- 'shi', 'sha', 'she', 'shai', 'shei', 'shao', 'shou', 'shan', 'shen', 'shang', 'sheng',
92
+ 'shi', 'sha', 'she', 'shai', 'shei', 'shao', 'shou', 'shan', 'shen', 'shang', 'sheng', # noqa: E501
87
93
  'shu', 'shua', 'shuo', 'shuai', 'shui', 'shuan', 'shun', 'shuang',
88
94
 
89
95
  'r', # Include to cover erhua
@@ -533,21 +539,12 @@ class PinyinTokenizer(TokenizerI):
533
539
  "zuo": "893830",
534
540
  }
535
541
 
536
- # Build allowed characters pattern
537
- # TODO: Handle this more gracefully
538
- _ALLOWED_CHARS = set(string.ascii_letters)
539
- for vowel, variants in VOWEL_TONE_VARIANTS.items():
540
- _ALLOWED_CHARS.update(vowel) # Add lowercase
541
- _ALLOWED_CHARS.update(vowel.upper()) # Add uppercase
542
- for tone_char in variants:
543
- _ALLOWED_CHARS.update(tone_char)
544
- _ALLOWED_CHARS.update(tone_char.upper())
545
-
546
- # Compile regex pattern for invalid character detection once at module level
547
- VALID_CHARS_PATTERN = re.compile(f"[^{''.join(sorted(_ALLOWED_CHARS))}]")
542
+ NON_ALPHABETIC_REGEX = re.compile(r"[^\w\s]+")
548
543
 
549
544
  def _get_tone_variants(self, syllable: str):
550
- """Generate all valid tone variants for a syllable. Assumes syllabe is lowercase"""
545
+ """
546
+ Generate all valid tone variants for a syllable. Assumes syllabe is lowercase
547
+ """
551
548
  variants = [syllable] # Include toneless variation
552
549
 
553
550
  # Find the vowels in the syllable (both upper and lower case)
@@ -565,7 +562,8 @@ class PinyinTokenizer(TokenizerI):
565
562
  elif any(v == "o" for v in vowels):
566
563
  tone_vowel = next(v for v in vowels if v == "o")
567
564
  else:
568
- # If there is no a e or o, the vowels are either 'iu', 'ui', or 'ê', in which case the mark goes on the last vowel
565
+ # If there is no a e or o, the vowels are either 'iu', 'ui', or 'ê',
566
+ # in which case the mark goes on the last vowel
569
567
  tone_vowel = vowels[-1]
570
568
 
571
569
  # Generate variants with each tone mark
@@ -578,27 +576,31 @@ class PinyinTokenizer(TokenizerI):
578
576
  return variants
579
577
 
580
578
  def __init__(self, include_nonstandard=False):
581
- self.trie = CharTrie()
582
- # Add standard syllables
579
+ self.preprocess_tokenizer = WordPunctTokenizer()
583
580
 
581
+ trie_contents = []
582
+
583
+ # Add standard syllables
584
584
  for syllable in self.STANDARD_SYLLABLES:
585
585
  for variant in self._get_tone_variants(syllable):
586
- self.trie[variant] = len(variant)
586
+ trie_contents.append(variant)
587
587
 
588
588
  if include_nonstandard:
589
589
  for syllable in self.NON_STANDARD_SYLLABLES:
590
590
  for variant in self._get_tone_variants(syllable):
591
- self.trie[variant] = len(variant)
591
+ trie_contents.append(variant)
592
592
 
593
- def span_tokenize(self, s: str) -> Iterator[Tuple[int, int]]:
594
- # Check for any invalid characters
595
- if self.VALID_CHARS_PATTERN.search(s):
596
- raise ValueError("Input string can only contain letters and tone marks")
593
+ self.trie = marisa_trie.Trie(trie_contents)
597
594
 
595
+ def _get_string_possibilites(self, s) -> List[List[Tuple[int, int]]]:
596
+ """
597
+ For a given string, return all possible valid syllable spans of that string,
598
+ indexed local to the string
599
+ """
598
600
  candidates = []
599
601
 
600
602
  # Generate all possible splits starting from the beginning
601
- to_process = [(0, [])]
603
+ to_process: List[Tuple[int, List[int]]] = [(0, [])]
602
604
  while to_process:
603
605
  # Get next position and accumulated split indices to process
604
606
  start_pos, split_indices = to_process.pop()
@@ -608,14 +610,15 @@ class PinyinTokenizer(TokenizerI):
608
610
  prefix_matches = self.trie.prefixes(remaining_s)
609
611
 
610
612
  # For each possible syllable length
611
- for _, length in prefix_matches:
613
+ for match in prefix_matches:
612
614
  # Create a new split point list with this syllable's endpoint
613
615
  new_splits = split_indices.copy()
614
- new_splits.append(start_pos + length)
616
+ new_splits.append(start_pos + len(match))
615
617
 
616
- if start_pos + length < len(s):
617
- # If we haven't reached the end, continue processing from end of this syllable
618
- to_process.append((start_pos + length, new_splits))
618
+ if start_pos + len(match) < len(s):
619
+ # If we haven't reached the end,
620
+ # continue processing from end of this syllable
621
+ to_process.append((start_pos + len(match), new_splits))
619
622
  else:
620
623
  # We've reached the end - construct the output span tuples
621
624
  spans = []
@@ -625,35 +628,55 @@ class PinyinTokenizer(TokenizerI):
625
628
  prev = pos
626
629
  candidates.append(spans)
627
630
 
628
- if not candidates:
629
- raise ValueError(f"No valid pinyin syllable splits found for '{s}'")
630
-
631
- # Find the shortest candidate(s) by comparing total number of splits
632
- min_length = min(len(splits) for splits in candidates)
633
- shortest = [c for c in candidates if len(c) == min_length]
631
+ return candidates
634
632
 
635
- if len(shortest) > 1:
636
- # Use syllable frequencies as tiebreaker
637
- max_freq = float("-inf")
638
- best_split = None
639
-
640
- for split in shortest:
641
- # Get syllables without tones and sum their frequencies
642
- syllables = [s[start:end].lower() for start, end in split]
643
- total_freq = sum(
644
- int(self.SYLLABLE_FREQUENCIES.get(syl, "0")) for syl in syllables
645
- )
646
-
647
- if total_freq > max_freq:
648
- max_freq = total_freq
649
- best_split = split
650
-
651
- spans = best_split
652
- else:
653
- spans = shortest[0]
654
-
655
- for span in spans:
656
- yield span
633
+ def span_tokenize(self, s: str) -> Iterator[Tuple[int, int]]:
634
+ # Start by splitting into sub-spans to examine
635
+ starting_spans = self.preprocess_tokenizer.span_tokenize(s)
636
+
637
+ final_spans = []
638
+ for start, end in starting_spans:
639
+ subspan_possibilities = self._get_string_possibilites(s[start:end])
640
+ if not subspan_possibilities:
641
+ # We were passed a subspan that wasnt valid pinyin
642
+ # If it is a non-alphabetic span, then we return as-is
643
+ if self.NON_ALPHABETIC_REGEX.match(s[start:end]):
644
+ final_spans.append((start, end))
645
+ continue
646
+ else:
647
+ raise ValueError(f"Invalid pinyin at substring: {s[start:end]}")
648
+
649
+ # Find the shortest candidate(s) for this span
650
+ min_length = min(len(splits) for splits in subspan_possibilities)
651
+ shortest = [c for c in subspan_possibilities if len(c) == min_length]
652
+
653
+ if len(shortest) > 1:
654
+ # Use syllable frequencies as tiebreaker
655
+ max_freq = float("-inf")
656
+ best_split = None
657
+
658
+ for split in shortest:
659
+ # Get syllables without tones and sum their frequencies
660
+ syllables = [s[start:end].lower() for start, end in split]
661
+ total_freq = sum(
662
+ int(self.SYLLABLE_FREQUENCIES.get(syl, "0"))
663
+ for syl in syllables
664
+ )
665
+
666
+ if total_freq > max_freq:
667
+ max_freq = total_freq
668
+ best_split = split
669
+
670
+ assert best_split is not None
671
+ chosen = best_split
672
+ else:
673
+ chosen = shortest[0]
674
+
675
+ for subspan in chosen:
676
+ # Convert from local span indices to full string indices
677
+ final_spans.append((start + subspan[0], start + subspan[1]))
678
+
679
+ yield from final_spans
657
680
 
658
681
  def tokenize(self, s: str) -> List[str]:
659
682
  return [s[start:end] for start, end in self.span_tokenize(s)]
@@ -1,5 +1,12 @@
1
- import pytest # type: ignore
2
- from pinyin_split.pinyinsplit import PinyinTokenizer
1
+ import pytest
2
+
3
+ from py_pinyin_split import PinyinTokenizer
4
+
5
+
6
+ def test_benchmark_simple(benchmark):
7
+ tokenizer = PinyinTokenizer()
8
+ result = benchmark(tokenizer.tokenize, "xīnniánkuàilè")
9
+ assert result == ["xīn", "nián", "kuài", "lè"]
3
10
 
4
11
 
5
12
  def test_no_tone_splits():
@@ -22,18 +29,10 @@ def test_tone_splits():
22
29
  assert tokenizer.tokenize("màn") == ["màn"]
23
30
 
24
31
 
25
- def test_edge_cases():
26
- """Test edge cases and invalid inputs"""
32
+ def test_invalid_pinyin():
33
+ """Inputs with invalid pinyin throw ValueErrors"""
27
34
  tokenizer = PinyinTokenizer()
28
35
 
29
- # Empty string should raise ValueError
30
- with pytest.raises(ValueError):
31
- tokenizer.tokenize("")
32
-
33
- # Invalid characters should raise ValueError
34
- with pytest.raises(ValueError):
35
- tokenizer.tokenize("hello!")
36
-
37
36
  # Single consonant should raise ValueError
38
37
  with pytest.raises(ValueError):
39
38
  tokenizer.tokenize("x")
@@ -47,6 +46,30 @@ def test_edge_cases():
47
46
  tokenizer.tokenize("ni3hao3")
48
47
 
49
48
 
49
+ def test_text_with_whitespace():
50
+ """Test handling of longer text with whitespace"""
51
+ tokenizer = PinyinTokenizer()
52
+
53
+ # Simple whitespace
54
+ assert tokenizer.tokenize("nǐ hǎo") == ["nǐ", "hǎo"]
55
+
56
+ # Multiple words with mixed tones
57
+ assert tokenizer.tokenize("Wǒ hěn xǐhuān Zhōngguó") == [
58
+ "Wǒ",
59
+ "hěn",
60
+ "xǐ",
61
+ "huān",
62
+ "Zhōng",
63
+ "guó",
64
+ ]
65
+
66
+ # Leading/trailing whitespace
67
+ assert tokenizer.tokenize(" nǐ hǎo ") == ["nǐ", "hǎo"]
68
+
69
+ # Multiple whitespace characters
70
+ assert tokenizer.tokenize("nǐ hǎo\t\nma") == ["nǐ", "hǎo", "ma"]
71
+
72
+
50
73
  def test_nonstandard_syllables():
51
74
  """Test handling of non-standard syllables"""
52
75
  standard_tokenizer = PinyinTokenizer(include_nonstandard=False)
@@ -108,3 +131,8 @@ def test_ambiguous_splits():
108
131
  # Test that tones help resolve ambiguity
109
132
  assert tokenizer.tokenize("xīan") == ["xī", "an"]
110
133
  assert tokenizer.tokenize("xián") == ["xián"]
134
+
135
+ # Test apostrophe handling
136
+ assert tokenizer.tokenize("Xī'ān") == ["Xī", "'", "ān"]
137
+ assert tokenizer.tokenize("yī'er") == ["yī", "'", "er"]
138
+ assert tokenizer.tokenize("tián'é") == ["tián", "'", "é"]