sinlib 0.1.13__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sinlib-0.2.0/.github/workflows/test.yml +32 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/.gitignore +6 -0
- sinlib-0.2.0/.readthedocs.yaml +11 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/PKG-INFO +1 -1
- {sinlib-0.1.13 → sinlib-0.2.0}/pyproject.toml +1 -1
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/__init__.py +10 -11
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/encoding.py +54 -15
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/romanize.py +20 -32
- sinlib-0.2.0/src/sinlib/spellcheck.py +774 -0
- sinlib-0.2.0/src/sinlib/subword.py +179 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/tokenizer.py +14 -4
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/transliterate.py +8 -2
- sinlib-0.2.0/src/sinlib/utils/__init__.py +13 -0
- sinlib-0.2.0/src/sinlib/utils/dataset_utils.py +10 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/model_utils.py +6 -7
- sinlib-0.2.0/src/sinlib/utils/models/akshara_ngram.json +51217 -0
- sinlib-0.2.0/src/sinlib/utils/models/akshara_vocab.json +542 -0
- sinlib-0.2.0/src/sinlib/utils/models/bigru_detector.pt +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/preprocessing.py +84 -13
- sinlib-0.2.0/tests/test_combined_detector.py +49 -0
- sinlib-0.2.0/tests/test_phase3_enhancements.py +127 -0
- sinlib-0.2.0/tests/test_romanizer.py +41 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_spellcheck.py +1 -1
- sinlib-0.2.0/tests/test_transliterator.py +30 -0
- sinlib-0.2.0/website/.gitignore +21 -0
- sinlib-0.2.0/website/.vscode/extensions.json +4 -0
- sinlib-0.2.0/website/.vscode/launch.json +11 -0
- sinlib-0.2.0/website/README.md +49 -0
- sinlib-0.2.0/website/astro.config.mjs +84 -0
- sinlib-0.2.0/website/package-lock.json +6855 -0
- sinlib-0.2.0/website/package.json +17 -0
- sinlib-0.2.0/website/public/favicon.svg +1 -0
- sinlib-0.2.0/website/src/assets/houston.webp +0 -0
- sinlib-0.2.0/website/src/assets/logo-dark.png +0 -0
- sinlib-0.2.0/website/src/assets/logo-light.png +0 -0
- sinlib-0.2.0/website/src/content/docs/api/encoding.mdx +121 -0
- sinlib-0.2.0/website/src/content/docs/api/preprocessing.mdx +120 -0
- sinlib-0.2.0/website/src/content/docs/api/romanizer.mdx +91 -0
- sinlib-0.2.0/website/src/content/docs/api/spellcheck.mdx +191 -0
- sinlib-0.2.0/website/src/content/docs/api/tokenizer.mdx +192 -0
- sinlib-0.2.0/website/src/content/docs/examples/tokenization.md +46 -0
- sinlib-0.2.0/website/src/content/docs/examples/typo-correction.md +97 -0
- sinlib-0.1.13/docs/guides/spellcheck.md → sinlib-0.2.0/website/src/content/docs/guides/spellcheck.mdx +15 -37
- sinlib-0.1.13/docs/guides/tokenization.md → sinlib-0.2.0/website/src/content/docs/guides/tokenization.mdx +26 -29
- sinlib-0.2.0/website/src/content/docs/index.mdx +120 -0
- sinlib-0.2.0/website/src/content.config.ts +7 -0
- sinlib-0.2.0/website/src/styles/custom.css +337 -0
- sinlib-0.2.0/website/tsconfig.json +5 -0
- sinlib-0.1.13/.readthedocs.yaml +0 -15
- sinlib-0.1.13/docs/api/encoding.md +0 -53
- sinlib-0.1.13/docs/api/index.md +0 -26
- sinlib-0.1.13/docs/api/preprocessing.md +0 -59
- sinlib-0.1.13/docs/api/spellcheck.md +0 -90
- sinlib-0.1.13/docs/api/tokenizer.md +0 -109
- sinlib-0.1.13/docs/assets/logo.svg +0 -15
- sinlib-0.1.13/docs/css/custom.css +0 -327
- sinlib-0.1.13/docs/css/extra.css +0 -280
- sinlib-0.1.13/docs/custom_theme/main.html +0 -44
- sinlib-0.1.13/docs/examples/Tokenization Example.md +0 -1
- sinlib-0.1.13/docs/examples/Typo Correction.md +0 -1
- sinlib-0.1.13/docs/guides/index.md +0 -16
- sinlib-0.1.13/docs/index.md +0 -129
- sinlib-0.1.13/docs/requirements.txt +0 -7
- sinlib-0.1.13/mkdocs.yml +0 -139
- sinlib-0.1.13/src/sinlib/spellcheck.py +0 -367
- sinlib-0.1.13/src/sinlib/utils/__init__.py +0 -7
- sinlib-0.1.13/src/sinlib/utils/dataset_utils.py +0 -12
- sinlib-0.1.13/tests/test_romanizer.py +0 -37
- sinlib-0.1.13/tests/test_transliterator.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/.github/workflows/python-publish.yml +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/LICENSE +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/README.md +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/requirements-dev.txt +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/requirements.txt +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/chars.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/models/__init__.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/models/transliterator-checkpoint.pth +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/models/transliterator_model.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer_batch.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer_bos.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer_special_tokens.py +0 -0
- {sinlib-0.1.13 → sinlib-0.2.0}/welcome.png +0 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
name: Run Tests
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [ main ]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [ main ]
|
|
8
|
+
|
|
9
|
+
jobs:
|
|
10
|
+
test:
|
|
11
|
+
name: Test on Python ${{ matrix.python-version }}
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
strategy:
|
|
14
|
+
matrix:
|
|
15
|
+
python-version: ["3.9", "3.10", "3.11", "3.12"]
|
|
16
|
+
|
|
17
|
+
steps:
|
|
18
|
+
- uses: actions/checkout@v4
|
|
19
|
+
|
|
20
|
+
- name: Set up Python ${{ matrix.python-version }}
|
|
21
|
+
uses: actions/setup-python@v5
|
|
22
|
+
with:
|
|
23
|
+
python-version: ${{ matrix.python-version }}
|
|
24
|
+
|
|
25
|
+
- name: Install dependencies
|
|
26
|
+
run: |
|
|
27
|
+
python -m pip install --upgrade pip
|
|
28
|
+
pip install -e ".[dev]" -r requirements.txt
|
|
29
|
+
|
|
30
|
+
- name: Run test suite with pytest
|
|
31
|
+
run: |
|
|
32
|
+
pytest --tb=short -v
|
|
@@ -7,16 +7,16 @@ transliteration of Sinhala text, along with preprocessing utilities.
|
|
|
7
7
|
Available Classes
|
|
8
8
|
-----------------
|
|
9
9
|
Tokenizer
|
|
10
|
-
Character-level tokenizer for Sinhala text.
|
|
10
|
+
Character-level tokenizer for Sinhala text. Mirrors the HuggingFace
|
|
11
11
|
``PreTrainedTokenizer`` interface — use ``Tokenizer.from_pretrained()``
|
|
12
12
|
to load the default pretrained vocabulary.
|
|
13
13
|
|
|
14
14
|
BatchEncoding
|
|
15
|
-
Dict-like container returned by the tokenizer.
|
|
15
|
+
Dict-like container returned by the tokenizer. Supports attribute access
|
|
16
16
|
(``enc.input_ids``, ``enc.attention_mask``) and dict-style access.
|
|
17
17
|
|
|
18
18
|
TypoDetector
|
|
19
|
-
N-gram–based spell checker for Sinhala.
|
|
19
|
+
N-gram–based spell checker for Sinhala. Use
|
|
20
20
|
``TypoDetector.from_pretrained()`` or instantiate directly.
|
|
21
21
|
|
|
22
22
|
preprocessing
|
|
@@ -25,30 +25,29 @@ preprocessing
|
|
|
25
25
|
|
|
26
26
|
Romanizer
|
|
27
27
|
Converts Sinhala text to Roman (Latin) script.
|
|
28
|
-
*Currently disabled from the default namespace due to an ongoing fix.*
|
|
29
28
|
|
|
30
29
|
Transliterator
|
|
31
30
|
ML-based Sinhala → Roman transliteration using a pre-trained BiLSTM.
|
|
32
|
-
*Currently disabled from
|
|
31
|
+
*Currently disabled from default __all__ pending model vocabulary realignment.*
|
|
33
32
|
"""
|
|
34
33
|
|
|
35
34
|
from typing import List
|
|
36
35
|
|
|
37
36
|
from sinlib.encoding import BatchEncoding
|
|
38
37
|
from sinlib.tokenizer import Tokenizer
|
|
38
|
+
from sinlib.subword import SubwordTokenizer
|
|
39
39
|
from sinlib.spellcheck import TypoDetector
|
|
40
|
-
from sinlib.
|
|
41
|
-
|
|
42
|
-
# These are still importable directly but excluded from __all__ while
|
|
43
|
-
# their respective bugs are being tracked down.
|
|
44
|
-
from sinlib.romanize import Romanizer # noqa: F401
|
|
40
|
+
from sinlib.romanize import Romanizer
|
|
45
41
|
from sinlib.transliterate import Transliterator # noqa: F401
|
|
42
|
+
from sinlib.utils import preprocessing
|
|
46
43
|
|
|
47
44
|
__all__: List[str] = [
|
|
48
45
|
"BatchEncoding",
|
|
49
46
|
"Tokenizer",
|
|
47
|
+
"SubwordTokenizer",
|
|
50
48
|
"TypoDetector",
|
|
49
|
+
"Romanizer",
|
|
51
50
|
"preprocessing",
|
|
52
51
|
]
|
|
53
52
|
|
|
54
|
-
__version__ = "0.1.
|
|
53
|
+
__version__ = "0.1.13"
|
|
@@ -24,6 +24,9 @@ class BatchEncoding:
|
|
|
24
24
|
data : dict
|
|
25
25
|
A dictionary whose values are lists of integers. Expected keys are
|
|
26
26
|
``"input_ids"`` and optionally ``"attention_mask"``.
|
|
27
|
+
tensor_type : str, optional
|
|
28
|
+
If set, converts lists to tensors. Supported values: ``"pt"`` (PyTorch),
|
|
29
|
+
``"tf"`` (TensorFlow), ``"np"`` (NumPy).
|
|
27
30
|
|
|
28
31
|
Examples
|
|
29
32
|
--------
|
|
@@ -36,35 +39,26 @@ class BatchEncoding:
|
|
|
36
39
|
True
|
|
37
40
|
"""
|
|
38
41
|
|
|
39
|
-
def __init__(self, data: Dict[str, Any]) -> None:
|
|
42
|
+
def __init__(self, data: Dict[str, Any], tensor_type: Optional[str] = None) -> None:
|
|
40
43
|
self._data: Dict[str, Any] = data
|
|
44
|
+
if tensor_type is not None:
|
|
45
|
+
self.convert_to_tensors(tensor_type)
|
|
41
46
|
|
|
42
47
|
# ------------------------------------------------------------------
|
|
43
48
|
# Attribute access
|
|
44
49
|
# ------------------------------------------------------------------
|
|
45
50
|
|
|
46
51
|
@property
|
|
47
|
-
def input_ids(self) ->
|
|
52
|
+
def input_ids(self) -> Any:
|
|
48
53
|
"""
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
Returns
|
|
52
|
-
-------
|
|
53
|
-
list of int
|
|
54
|
-
The encoded token IDs for the input text.
|
|
54
|
+
Token IDs (list or tensor depending on return_tensors).
|
|
55
55
|
"""
|
|
56
56
|
return self._data["input_ids"]
|
|
57
57
|
|
|
58
58
|
@property
|
|
59
|
-
def attention_mask(self) ->
|
|
59
|
+
def attention_mask(self) -> Any:
|
|
60
60
|
"""
|
|
61
61
|
Attention mask (1 for real tokens, 0 for padding).
|
|
62
|
-
|
|
63
|
-
Returns
|
|
64
|
-
-------
|
|
65
|
-
list of int or None
|
|
66
|
-
``1`` for each real token position, ``0`` for padding.
|
|
67
|
-
``None`` when the tokenizer was called without padding.
|
|
68
62
|
"""
|
|
69
63
|
return self._data.get("attention_mask")
|
|
70
64
|
|
|
@@ -103,6 +97,51 @@ class BatchEncoding:
|
|
|
103
97
|
# Conversion
|
|
104
98
|
# ------------------------------------------------------------------
|
|
105
99
|
|
|
100
|
+
def convert_to_tensors(self, tensor_type: str) -> None:
|
|
101
|
+
"""
|
|
102
|
+
Convert the internal input lists to tensors.
|
|
103
|
+
"""
|
|
104
|
+
if tensor_type == "pt":
|
|
105
|
+
import torch
|
|
106
|
+
for key, val in self._data.items():
|
|
107
|
+
if val is not None:
|
|
108
|
+
try:
|
|
109
|
+
self._data[key] = torch.tensor(val)
|
|
110
|
+
except ValueError as e:
|
|
111
|
+
raise ValueError(
|
|
112
|
+
f"Failed to convert {key} to PyTorch tensor. "
|
|
113
|
+
f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
|
|
114
|
+
f"Error: {e}"
|
|
115
|
+
) from e
|
|
116
|
+
elif tensor_type == "tf":
|
|
117
|
+
import tensorflow as tf
|
|
118
|
+
for key, val in self._data.items():
|
|
119
|
+
if val is not None:
|
|
120
|
+
try:
|
|
121
|
+
self._data[key] = tf.convert_to_tensor(val)
|
|
122
|
+
except (ValueError, TypeError) as e:
|
|
123
|
+
raise ValueError(
|
|
124
|
+
f"Failed to convert {key} to TensorFlow tensor. "
|
|
125
|
+
f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
|
|
126
|
+
f"Error: {e}"
|
|
127
|
+
) from e
|
|
128
|
+
elif tensor_type == "np":
|
|
129
|
+
import numpy as np
|
|
130
|
+
for key, val in self._data.items():
|
|
131
|
+
if val is not None:
|
|
132
|
+
try:
|
|
133
|
+
self._data[key] = np.array(val)
|
|
134
|
+
except ValueError as e:
|
|
135
|
+
raise ValueError(
|
|
136
|
+
f"Failed to convert {key} to NumPy array. "
|
|
137
|
+
f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
|
|
138
|
+
f"Error: {e}"
|
|
139
|
+
) from e
|
|
140
|
+
else:
|
|
141
|
+
raise ValueError(
|
|
142
|
+
f"Unsupported tensor_type '{tensor_type}'. Use 'pt', 'tf', or 'np'."
|
|
143
|
+
)
|
|
144
|
+
|
|
106
145
|
def to_dict(self) -> Dict[str, Any]:
|
|
107
146
|
"""
|
|
108
147
|
Convert to a plain Python dictionary.
|
|
@@ -11,7 +11,7 @@ from numpy.typing import NDArray
|
|
|
11
11
|
|
|
12
12
|
from .tokenizer import Tokenizer
|
|
13
13
|
from .utils.chars import ALL_SINHALA_CHARACTERS, NUMBERS_AND_PUNCTUATION
|
|
14
|
-
from .utils.preprocessing import load_char_mapper, remove_non_printable
|
|
14
|
+
from .utils.preprocessing import load_char_mapper, remove_non_printable, process_text
|
|
15
15
|
|
|
16
16
|
|
|
17
17
|
class Romanizer:
|
|
@@ -39,8 +39,7 @@ class Romanizer:
|
|
|
39
39
|
tokenizer_path: Path to tokenizer vocabulary file
|
|
40
40
|
"""
|
|
41
41
|
self.char_mapper = load_char_mapper()
|
|
42
|
-
self.tokenizer = Tokenizer(
|
|
43
|
-
self.tokenizer.load_from_pretrained(file_path=None, load_default_tokenizer=True)
|
|
42
|
+
self.tokenizer = Tokenizer.from_pretrained("Ransaka/sinlib")
|
|
44
43
|
|
|
45
44
|
def __call__(self, text: Union[str, List[str]]) -> Union[str, List[str]]:
|
|
46
45
|
"""
|
|
@@ -67,32 +66,21 @@ class Romanizer:
|
|
|
67
66
|
Romanized version of the input text
|
|
68
67
|
"""
|
|
69
68
|
text = remove_non_printable(text)
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
romanized_chars = [
|
|
89
|
-
self.char_mapper.get(ch, ch if ch in NUMBERS_AND_PUNCTUATION.union({" "}) else '')
|
|
90
|
-
for ch in decoded_chars
|
|
91
|
-
]
|
|
92
|
-
romanized_sinhala = "".join(romanized_chars)
|
|
93
|
-
|
|
94
|
-
# Create word mapping and apply to full text
|
|
95
|
-
word_mapping = dict(zip(sinhala_text.split(), romanized_sinhala.split()))
|
|
96
|
-
romanized_words = [word_mapping.get(word, word) for word in text.split()]
|
|
97
|
-
|
|
98
|
-
return " ".join(romanized_words)
|
|
69
|
+
if not text:
|
|
70
|
+
return ""
|
|
71
|
+
|
|
72
|
+
# Fallback mappings for diacritics/modifiers
|
|
73
|
+
char_map = dict(self.char_mapper)
|
|
74
|
+
char_map.setdefault('ං', 'n')
|
|
75
|
+
char_map.setdefault('ඃ', 'h')
|
|
76
|
+
|
|
77
|
+
tokens = process_text(text)
|
|
78
|
+
res = []
|
|
79
|
+
for tok in tokens:
|
|
80
|
+
clean = tok.strip()
|
|
81
|
+
roman = char_map.get(clean, clean)
|
|
82
|
+
if tok.endswith(' ') and not roman.endswith(' '):
|
|
83
|
+
roman += ' '
|
|
84
|
+
res.append(roman)
|
|
85
|
+
|
|
86
|
+
return "".join(res)
|