sinlib 0.1.12__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. sinlib-0.2.0/.github/workflows/test.yml +32 -0
  2. {sinlib-0.1.12 → sinlib-0.2.0}/.gitignore +6 -0
  3. sinlib-0.2.0/.readthedocs.yaml +11 -0
  4. sinlib-0.2.0/PKG-INFO +154 -0
  5. sinlib-0.2.0/README.md +110 -0
  6. {sinlib-0.1.12 → sinlib-0.2.0}/pyproject.toml +2 -2
  7. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/__init__.py +10 -11
  8. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/encoding.py +54 -15
  9. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/romanize.py +20 -32
  10. sinlib-0.2.0/src/sinlib/spellcheck.py +774 -0
  11. sinlib-0.2.0/src/sinlib/subword.py +179 -0
  12. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/tokenizer.py +14 -4
  13. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/transliterate.py +8 -2
  14. sinlib-0.2.0/src/sinlib/utils/__init__.py +13 -0
  15. sinlib-0.2.0/src/sinlib/utils/dataset_utils.py +10 -0
  16. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/utils/model_utils.py +6 -7
  17. sinlib-0.2.0/src/sinlib/utils/models/akshara_ngram.json +51217 -0
  18. sinlib-0.2.0/src/sinlib/utils/models/akshara_vocab.json +542 -0
  19. sinlib-0.2.0/src/sinlib/utils/models/bigru_detector.pt +0 -0
  20. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/utils/preprocessing.py +84 -13
  21. sinlib-0.2.0/tests/test_combined_detector.py +49 -0
  22. sinlib-0.2.0/tests/test_phase3_enhancements.py +127 -0
  23. sinlib-0.2.0/tests/test_romanizer.py +41 -0
  24. {sinlib-0.1.12 → sinlib-0.2.0}/tests/test_spellcheck.py +1 -1
  25. sinlib-0.2.0/tests/test_transliterator.py +30 -0
  26. sinlib-0.2.0/website/.gitignore +21 -0
  27. sinlib-0.2.0/website/.vscode/extensions.json +4 -0
  28. sinlib-0.2.0/website/.vscode/launch.json +11 -0
  29. sinlib-0.2.0/website/README.md +49 -0
  30. sinlib-0.2.0/website/astro.config.mjs +84 -0
  31. sinlib-0.2.0/website/package-lock.json +6855 -0
  32. sinlib-0.2.0/website/package.json +17 -0
  33. sinlib-0.2.0/website/public/favicon.svg +1 -0
  34. sinlib-0.2.0/website/src/assets/houston.webp +0 -0
  35. sinlib-0.2.0/website/src/assets/logo-dark.png +0 -0
  36. sinlib-0.2.0/website/src/assets/logo-light.png +0 -0
  37. sinlib-0.2.0/website/src/content/docs/api/encoding.mdx +121 -0
  38. sinlib-0.2.0/website/src/content/docs/api/preprocessing.mdx +120 -0
  39. sinlib-0.2.0/website/src/content/docs/api/romanizer.mdx +91 -0
  40. sinlib-0.2.0/website/src/content/docs/api/spellcheck.mdx +191 -0
  41. sinlib-0.2.0/website/src/content/docs/api/tokenizer.mdx +192 -0
  42. sinlib-0.2.0/website/src/content/docs/examples/tokenization.md +46 -0
  43. sinlib-0.2.0/website/src/content/docs/examples/typo-correction.md +97 -0
  44. sinlib-0.1.12/docs/guides/spellcheck.md → sinlib-0.2.0/website/src/content/docs/guides/spellcheck.mdx +15 -37
  45. sinlib-0.1.12/docs/guides/tokenization.md → sinlib-0.2.0/website/src/content/docs/guides/tokenization.mdx +26 -29
  46. sinlib-0.2.0/website/src/content/docs/index.mdx +120 -0
  47. sinlib-0.2.0/website/src/content.config.ts +7 -0
  48. sinlib-0.2.0/website/src/styles/custom.css +337 -0
  49. sinlib-0.2.0/website/tsconfig.json +5 -0
  50. sinlib-0.1.12/.readthedocs.yaml +0 -15
  51. sinlib-0.1.12/PKG-INFO +0 -183
  52. sinlib-0.1.12/README.md +0 -139
  53. sinlib-0.1.12/docs/api/encoding.md +0 -53
  54. sinlib-0.1.12/docs/api/preprocessing.md +0 -59
  55. sinlib-0.1.12/docs/api/spellcheck.md +0 -90
  56. sinlib-0.1.12/docs/api/tokenizer.md +0 -109
  57. sinlib-0.1.12/docs/css/custom.css +0 -184
  58. sinlib-0.1.12/docs/css/extra.css +0 -57
  59. sinlib-0.1.12/docs/custom_theme/main.html +0 -38
  60. sinlib-0.1.12/docs/examples/Tokenization Example.md +0 -1
  61. sinlib-0.1.12/docs/examples/Typo Correction.md +0 -1
  62. sinlib-0.1.12/docs/index.md +0 -76
  63. sinlib-0.1.12/docs/requirements.txt +0 -7
  64. sinlib-0.1.12/mkdocs.yml +0 -88
  65. sinlib-0.1.12/src/sinlib/spellcheck.py +0 -367
  66. sinlib-0.1.12/src/sinlib/utils/__init__.py +0 -7
  67. sinlib-0.1.12/src/sinlib/utils/dataset_utils.py +0 -12
  68. sinlib-0.1.12/tests/test_romanizer.py +0 -37
  69. sinlib-0.1.12/tests/test_transliterator.py +0 -0
  70. {sinlib-0.1.12 → sinlib-0.2.0}/.github/workflows/python-publish.yml +0 -0
  71. {sinlib-0.1.12 → sinlib-0.2.0}/LICENSE +0 -0
  72. {sinlib-0.1.12 → sinlib-0.2.0}/requirements-dev.txt +0 -0
  73. {sinlib-0.1.12 → sinlib-0.2.0}/requirements.txt +0 -0
  74. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/utils/chars.py +0 -0
  75. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/utils/models/__init__.py +0 -0
  76. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/utils/models/transliterator-checkpoint.pth +0 -0
  77. {sinlib-0.1.12 → sinlib-0.2.0}/src/sinlib/utils/models/transliterator_model.py +0 -0
  78. {sinlib-0.1.12 → sinlib-0.2.0}/tests/test_tokenizer.py +0 -0
  79. {sinlib-0.1.12 → sinlib-0.2.0}/tests/test_tokenizer_batch.py +0 -0
  80. {sinlib-0.1.12 → sinlib-0.2.0}/tests/test_tokenizer_bos.py +0 -0
  81. {sinlib-0.1.12 → sinlib-0.2.0}/tests/test_tokenizer_special_tokens.py +0 -0
  82. {sinlib-0.1.12 → sinlib-0.2.0}/welcome.png +0 -0
@@ -0,0 +1,32 @@
1
+ name: Run Tests
2
+
3
+ on:
4
+ push:
5
+ branches: [ main ]
6
+ pull_request:
7
+ branches: [ main ]
8
+
9
+ jobs:
10
+ test:
11
+ name: Test on Python ${{ matrix.python-version }}
12
+ runs-on: ubuntu-latest
13
+ strategy:
14
+ matrix:
15
+ python-version: ["3.9", "3.10", "3.11", "3.12"]
16
+
17
+ steps:
18
+ - uses: actions/checkout@v4
19
+
20
+ - name: Set up Python ${{ matrix.python-version }}
21
+ uses: actions/setup-python@v5
22
+ with:
23
+ python-version: ${{ matrix.python-version }}
24
+
25
+ - name: Install dependencies
26
+ run: |
27
+ python -m pip install --upgrade pip
28
+ pip install -e ".[dev]" -r requirements.txt
29
+
30
+ - name: Run test suite with pytest
31
+ run: |
32
+ pytest --tb=short -v
@@ -11,3 +11,9 @@ experiments/*
11
11
  *.ipynb
12
12
  .qodo
13
13
  examples/examples.ipynb
14
+
15
+ # Astro / Starlight
16
+ website/dist/
17
+ website/node_modules/
18
+ website/.astro/
19
+
@@ -0,0 +1,11 @@
1
+ version: 2
2
+
3
+ build:
4
+ os: ubuntu-24.04
5
+ tools:
6
+ nodejs: "20"
7
+ commands:
8
+ - cd website && npm ci
9
+ - cd website && npm run build
10
+ - mkdir -p $READTHEDOCS_OUTPUT/html
11
+ - cp -r website/dist/* $READTHEDOCS_OUTPUT/html/
sinlib-0.2.0/PKG-INFO ADDED
@@ -0,0 +1,154 @@
1
+ Metadata-Version: 2.4
2
+ Name: sinlib
3
+ Version: 0.2.0
4
+ Summary: Sinhala NLP Toolkit
5
+ Project-URL: Code, https://github.com/Ransaka/sinlib
6
+ Project-URL: Docs, https://sinlib.readthedocs.io
7
+ Author-email: Ransaka <ransaka.ravihara@gmail.com>
8
+ License: MIT License
9
+
10
+ Copyright (c) [2024] [Ransaka Ravihara]
11
+
12
+ Permission is hereby granted, free of charge, to any person obtaining a copy
13
+ of this software and associated documentation files (the "Software"), to deal
14
+ in the Software without restriction, including without limitation the rights
15
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
16
+ copies of the Software, and to permit persons to whom the Software is
17
+ furnished to do so, subject to the following conditions:
18
+
19
+ The above copyright notice and this permission notice shall be included in all
20
+ copies or substantial portions of the Software.
21
+
22
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
23
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
24
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
25
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
26
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
27
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
28
+ SOFTWARE.
29
+ License-File: LICENSE
30
+ Keywords: NLP,Sinhala,python
31
+ Classifier: License :: OSI Approved :: MIT License
32
+ Classifier: Programming Language :: Python :: 3.9
33
+ Classifier: Programming Language :: Python :: 3.10
34
+ Classifier: Programming Language :: Python :: 3.11
35
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
36
+ Requires-Python: >=3.9.7
37
+ Requires-Dist: huggingface-hub
38
+ Requires-Dist: numpy>=1.24.0
39
+ Requires-Dist: tqdm>=4.64.1
40
+ Requires-Dist: transformers>=4.31.0
41
+ Provides-Extra: dev
42
+ Requires-Dist: pytest; extra == 'dev'
43
+ Description-Content-Type: text/markdown
44
+
45
+ # Sinlib
46
+
47
+ <div align="center">
48
+
49
+ ![Sinlib Logo](welcome.png)
50
+
51
+ [![PyPI version](https://badge.fury.io/py/sinlib.svg)](https://badge.fury.io/py/sinlib)
52
+ [![Python Versions](https://img.shields.io/pypi/pyversions/sinlib.svg)](https://pypi.org/project/sinlib/)
53
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
54
+ [![Docs](https://img.shields.io/badge/docs-readthedocs-blue.svg)](https://sinlib.readthedocs.io)
55
+
56
+ A Python toolkit for Sinhala natural language processing — phonological tokenization, spell checking, and text preprocessing.
57
+
58
+ </div>
59
+
60
+ > **Note:** The `Romanizer` and `Transliterator` modules are temporarily unavailable due to a known bug and will be restored in a future release.
61
+
62
+ ## Installation
63
+
64
+ ```bash
65
+ pip install sinlib
66
+ ```
67
+
68
+ ## Quick Start
69
+
70
+ ### Tokenization
71
+
72
+ ```python
73
+ from sinlib import Tokenizer
74
+
75
+ tokenizer = Tokenizer.from_pretrained("Ransaka/sinlib")
76
+
77
+ # Split into phonological units (base consonant + diacritics)
78
+ tokens = tokenizer.tokenize("ආයුබෝවන්")
79
+ # ['ආ', 'යු', 'බෝ', 'ව', 'න්']
80
+
81
+ # Encode to integer IDs
82
+ encoding = tokenizer("ආයුබෝවන්")
83
+ encoding.input_ids # [4, 23, 18, 7, 12]
84
+ encoding.attention_mask # [1, 1, 1, 1, 1]
85
+
86
+ # Batch encode with padding
87
+ batch = tokenizer(["ආයුබෝවන්", "සිංහල"], padding=True)
88
+ batch.input_ids # [[4, 23, 18, 7, 12], [9, 31, 6, 0, 0]]
89
+ ```
90
+
91
+ ### Spell Checking
92
+
93
+ ```python
94
+ from sinlib import TypoDetector
95
+
96
+ detector = TypoDetector.from_pretrained("Ransaka/sinlib")
97
+
98
+ # Auto-correct a sentence
99
+ detector("අපකරියට ගිය")
100
+ # 'අපකීර්තියට ගිය'
101
+
102
+ # Get correction suggestions
103
+ detector.suggest_correction("අඩිරාජ")
104
+ # ['අධිරාජ']
105
+ ```
106
+
107
+ ### Preprocessing
108
+
109
+ ```python
110
+ from sinlib import preprocessing
111
+
112
+ # Remove noise and normalise text
113
+ clean = preprocessing.process_text("Hello, මේ සිංහල වාක්‍යකි.")
114
+
115
+ # Compute Sinhala character ratio
116
+ ratio = preprocessing.get_sinhala_character_ratio(["මෙය සිංහල වාක්‍යක්"])
117
+ # [0.9]
118
+ ```
119
+
120
+ ## Why phonological tokenization?
121
+
122
+ Sinhala script combines a base consonant with one or more vowel diacritics into a single phonetic unit. Standard Unicode tokenization breaks these apart, producing incorrect representations for downstream tasks like ASR and TTS.
123
+
124
+ ```
125
+ "ආයුබෝවන්"
126
+
127
+ Sinlib → ['ආ', 'යු', 'බෝ', 'ව', 'න්'] ✓ phonological units
128
+ Unicode → ['ආ', 'ය', 'ු', 'බ', 'ෝ', 'ව', 'න', '්'] ✗ raw code points
129
+ ```
130
+
131
+ Vocab and model weights are fetched automatically from [`Ransaka/sinlib`](https://huggingface.co/Ransaka/sinlib) on HuggingFace Hub at first use — no manual setup required.
132
+
133
+ ## Documentation
134
+
135
+ Full documentation is available at **[sinlib.readthedocs.io](https://sinlib.readthedocs.io)**, including:
136
+
137
+ - [API Reference — Tokenizer](https://sinlib.readthedocs.io/en/latest/api/tokenizer/)
138
+ - [API Reference — TypoDetector](https://sinlib.readthedocs.io/en/latest/api/spellcheck/)
139
+ - [Guide: Tokenization](https://sinlib.readthedocs.io/en/latest/guides/tokenization/)
140
+ - [Guide: Spell Checking](https://sinlib.readthedocs.io/en/latest/guides/spellcheck/)
141
+
142
+ ## Contributing
143
+
144
+ Contributions are welcome. Please open an issue or submit a pull request on [GitHub](https://github.com/Ransaka/sinlib).
145
+
146
+ 1. Fork the repository
147
+ 2. Create a feature branch (`git checkout -b feature/my-feature`)
148
+ 3. Commit your changes (`git commit -m 'Add my feature'`)
149
+ 4. Push to the branch (`git push origin feature/my-feature`)
150
+ 5. Open a Pull Request
151
+
152
+ ## License
153
+
154
+ MIT License — see the [LICENSE](LICENSE) file for details.
sinlib-0.2.0/README.md ADDED
@@ -0,0 +1,110 @@
1
+ # Sinlib
2
+
3
+ <div align="center">
4
+
5
+ ![Sinlib Logo](welcome.png)
6
+
7
+ [![PyPI version](https://badge.fury.io/py/sinlib.svg)](https://badge.fury.io/py/sinlib)
8
+ [![Python Versions](https://img.shields.io/pypi/pyversions/sinlib.svg)](https://pypi.org/project/sinlib/)
9
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
10
+ [![Docs](https://img.shields.io/badge/docs-readthedocs-blue.svg)](https://sinlib.readthedocs.io)
11
+
12
+ A Python toolkit for Sinhala natural language processing — phonological tokenization, spell checking, and text preprocessing.
13
+
14
+ </div>
15
+
16
+ > **Note:** The `Romanizer` and `Transliterator` modules are temporarily unavailable due to a known bug and will be restored in a future release.
17
+
18
+ ## Installation
19
+
20
+ ```bash
21
+ pip install sinlib
22
+ ```
23
+
24
+ ## Quick Start
25
+
26
+ ### Tokenization
27
+
28
+ ```python
29
+ from sinlib import Tokenizer
30
+
31
+ tokenizer = Tokenizer.from_pretrained("Ransaka/sinlib")
32
+
33
+ # Split into phonological units (base consonant + diacritics)
34
+ tokens = tokenizer.tokenize("ආයුබෝවන්")
35
+ # ['ආ', 'යු', 'බෝ', 'ව', 'න්']
36
+
37
+ # Encode to integer IDs
38
+ encoding = tokenizer("ආයුබෝවන්")
39
+ encoding.input_ids # [4, 23, 18, 7, 12]
40
+ encoding.attention_mask # [1, 1, 1, 1, 1]
41
+
42
+ # Batch encode with padding
43
+ batch = tokenizer(["ආයුබෝවන්", "සිංහල"], padding=True)
44
+ batch.input_ids # [[4, 23, 18, 7, 12], [9, 31, 6, 0, 0]]
45
+ ```
46
+
47
+ ### Spell Checking
48
+
49
+ ```python
50
+ from sinlib import TypoDetector
51
+
52
+ detector = TypoDetector.from_pretrained("Ransaka/sinlib")
53
+
54
+ # Auto-correct a sentence
55
+ detector("අපකරියට ගිය")
56
+ # 'අපකීර්තියට ගිය'
57
+
58
+ # Get correction suggestions
59
+ detector.suggest_correction("අඩිරාජ")
60
+ # ['අධිරාජ']
61
+ ```
62
+
63
+ ### Preprocessing
64
+
65
+ ```python
66
+ from sinlib import preprocessing
67
+
68
+ # Remove noise and normalise text
69
+ clean = preprocessing.process_text("Hello, මේ සිංහල වාක්‍යකි.")
70
+
71
+ # Compute Sinhala character ratio
72
+ ratio = preprocessing.get_sinhala_character_ratio(["මෙය සිංහල වාක්‍යක්"])
73
+ # [0.9]
74
+ ```
75
+
76
+ ## Why phonological tokenization?
77
+
78
+ Sinhala script combines a base consonant with one or more vowel diacritics into a single phonetic unit. Standard Unicode tokenization breaks these apart, producing incorrect representations for downstream tasks like ASR and TTS.
79
+
80
+ ```
81
+ "ආයුබෝවන්"
82
+
83
+ Sinlib → ['ආ', 'යු', 'බෝ', 'ව', 'න්'] ✓ phonological units
84
+ Unicode → ['ආ', 'ය', 'ු', 'බ', 'ෝ', 'ව', 'න', '්'] ✗ raw code points
85
+ ```
86
+
87
+ Vocab and model weights are fetched automatically from [`Ransaka/sinlib`](https://huggingface.co/Ransaka/sinlib) on HuggingFace Hub at first use — no manual setup required.
88
+
89
+ ## Documentation
90
+
91
+ Full documentation is available at **[sinlib.readthedocs.io](https://sinlib.readthedocs.io)**, including:
92
+
93
+ - [API Reference — Tokenizer](https://sinlib.readthedocs.io/en/latest/api/tokenizer/)
94
+ - [API Reference — TypoDetector](https://sinlib.readthedocs.io/en/latest/api/spellcheck/)
95
+ - [Guide: Tokenization](https://sinlib.readthedocs.io/en/latest/guides/tokenization/)
96
+ - [Guide: Spell Checking](https://sinlib.readthedocs.io/en/latest/guides/spellcheck/)
97
+
98
+ ## Contributing
99
+
100
+ Contributions are welcome. Please open an issue or submit a pull request on [GitHub](https://github.com/Ransaka/sinlib).
101
+
102
+ 1. Fork the repository
103
+ 2. Create a feature branch (`git checkout -b feature/my-feature`)
104
+ 3. Commit your changes (`git commit -m 'Add my feature'`)
105
+ 4. Push to the branch (`git push origin feature/my-feature`)
106
+ 5. Open a Pull Request
107
+
108
+ ## License
109
+
110
+ MIT License — see the [LICENSE](LICENSE) file for details.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sinlib"
3
- version = "0.1.12"
3
+ version = "0.2.0"
4
4
  description = "Sinhala NLP Toolkit"
5
5
  authors = [
6
6
  { name = "Ransaka", email = "ransaka.ravihara@gmail.com" }
@@ -30,7 +30,7 @@ dev = [
30
30
 
31
31
  [project.urls]
32
32
  Code = "https://github.com/Ransaka/sinlib"
33
- Docs = "https://github.com/Ransaka/sinlib"
33
+ Docs = "https://sinlib.readthedocs.io"
34
34
 
35
35
  [build-system]
36
36
  requires = ["hatchling"]
@@ -7,16 +7,16 @@ transliteration of Sinhala text, along with preprocessing utilities.
7
7
  Available Classes
8
8
  -----------------
9
9
  Tokenizer
10
- Character-level tokenizer for Sinhala text. Mirrors the HuggingFace
10
+ Character-level tokenizer for Sinhala text. Mirrors the HuggingFace
11
11
  ``PreTrainedTokenizer`` interface — use ``Tokenizer.from_pretrained()``
12
12
  to load the default pretrained vocabulary.
13
13
 
14
14
  BatchEncoding
15
- Dict-like container returned by the tokenizer. Supports attribute access
15
+ Dict-like container returned by the tokenizer. Supports attribute access
16
16
  (``enc.input_ids``, ``enc.attention_mask``) and dict-style access.
17
17
 
18
18
  TypoDetector
19
- N-gram–based spell checker for Sinhala. Use
19
+ N-gram–based spell checker for Sinhala. Use
20
20
  ``TypoDetector.from_pretrained()`` or instantiate directly.
21
21
 
22
22
  preprocessing
@@ -25,30 +25,29 @@ preprocessing
25
25
 
26
26
  Romanizer
27
27
  Converts Sinhala text to Roman (Latin) script.
28
- *Currently disabled from the default namespace due to an ongoing fix.*
29
28
 
30
29
  Transliterator
31
30
  ML-based Sinhala → Roman transliteration using a pre-trained BiLSTM.
32
- *Currently disabled from the default namespace due to an ongoing fix.*
31
+ *Currently disabled from default __all__ pending model vocabulary realignment.*
33
32
  """
34
33
 
35
34
  from typing import List
36
35
 
37
36
  from sinlib.encoding import BatchEncoding
38
37
  from sinlib.tokenizer import Tokenizer
38
+ from sinlib.subword import SubwordTokenizer
39
39
  from sinlib.spellcheck import TypoDetector
40
- from sinlib.utils import preprocessing
41
-
42
- # These are still importable directly but excluded from __all__ while
43
- # their respective bugs are being tracked down.
44
- from sinlib.romanize import Romanizer # noqa: F401
40
+ from sinlib.romanize import Romanizer
45
41
  from sinlib.transliterate import Transliterator # noqa: F401
42
+ from sinlib.utils import preprocessing
46
43
 
47
44
  __all__: List[str] = [
48
45
  "BatchEncoding",
49
46
  "Tokenizer",
47
+ "SubwordTokenizer",
50
48
  "TypoDetector",
49
+ "Romanizer",
51
50
  "preprocessing",
52
51
  ]
53
52
 
54
- __version__ = "0.1.12"
53
+ __version__ = "0.1.13"
@@ -24,6 +24,9 @@ class BatchEncoding:
24
24
  data : dict
25
25
  A dictionary whose values are lists of integers. Expected keys are
26
26
  ``"input_ids"`` and optionally ``"attention_mask"``.
27
+ tensor_type : str, optional
28
+ If set, converts lists to tensors. Supported values: ``"pt"`` (PyTorch),
29
+ ``"tf"`` (TensorFlow), ``"np"`` (NumPy).
27
30
 
28
31
  Examples
29
32
  --------
@@ -36,35 +39,26 @@ class BatchEncoding:
36
39
  True
37
40
  """
38
41
 
39
- def __init__(self, data: Dict[str, Any]) -> None:
42
+ def __init__(self, data: Dict[str, Any], tensor_type: Optional[str] = None) -> None:
40
43
  self._data: Dict[str, Any] = data
44
+ if tensor_type is not None:
45
+ self.convert_to_tensors(tensor_type)
41
46
 
42
47
  # ------------------------------------------------------------------
43
48
  # Attribute access
44
49
  # ------------------------------------------------------------------
45
50
 
46
51
  @property
47
- def input_ids(self) -> List[int]:
52
+ def input_ids(self) -> Any:
48
53
  """
49
- List of token IDs.
50
-
51
- Returns
52
- -------
53
- list of int
54
- The encoded token IDs for the input text.
54
+ Token IDs (list or tensor depending on return_tensors).
55
55
  """
56
56
  return self._data["input_ids"]
57
57
 
58
58
  @property
59
- def attention_mask(self) -> Optional[List[int]]:
59
+ def attention_mask(self) -> Any:
60
60
  """
61
61
  Attention mask (1 for real tokens, 0 for padding).
62
-
63
- Returns
64
- -------
65
- list of int or None
66
- ``1`` for each real token position, ``0`` for padding.
67
- ``None`` when the tokenizer was called without padding.
68
62
  """
69
63
  return self._data.get("attention_mask")
70
64
 
@@ -103,6 +97,51 @@ class BatchEncoding:
103
97
  # Conversion
104
98
  # ------------------------------------------------------------------
105
99
 
100
+ def convert_to_tensors(self, tensor_type: str) -> None:
101
+ """
102
+ Convert the internal input lists to tensors.
103
+ """
104
+ if tensor_type == "pt":
105
+ import torch
106
+ for key, val in self._data.items():
107
+ if val is not None:
108
+ try:
109
+ self._data[key] = torch.tensor(val)
110
+ except ValueError as e:
111
+ raise ValueError(
112
+ f"Failed to convert {key} to PyTorch tensor. "
113
+ f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
114
+ f"Error: {e}"
115
+ ) from e
116
+ elif tensor_type == "tf":
117
+ import tensorflow as tf
118
+ for key, val in self._data.items():
119
+ if val is not None:
120
+ try:
121
+ self._data[key] = tf.convert_to_tensor(val)
122
+ except (ValueError, TypeError) as e:
123
+ raise ValueError(
124
+ f"Failed to convert {key} to TensorFlow tensor. "
125
+ f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
126
+ f"Error: {e}"
127
+ ) from e
128
+ elif tensor_type == "np":
129
+ import numpy as np
130
+ for key, val in self._data.items():
131
+ if val is not None:
132
+ try:
133
+ self._data[key] = np.array(val)
134
+ except ValueError as e:
135
+ raise ValueError(
136
+ f"Failed to convert {key} to NumPy array. "
137
+ f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
138
+ f"Error: {e}"
139
+ ) from e
140
+ else:
141
+ raise ValueError(
142
+ f"Unsupported tensor_type '{tensor_type}'. Use 'pt', 'tf', or 'np'."
143
+ )
144
+
106
145
  def to_dict(self) -> Dict[str, Any]:
107
146
  """
108
147
  Convert to a plain Python dictionary.
@@ -11,7 +11,7 @@ from numpy.typing import NDArray
11
11
 
12
12
  from .tokenizer import Tokenizer
13
13
  from .utils.chars import ALL_SINHALA_CHARACTERS, NUMBERS_AND_PUNCTUATION
14
- from .utils.preprocessing import load_char_mapper, remove_non_printable
14
+ from .utils.preprocessing import load_char_mapper, remove_non_printable, process_text
15
15
 
16
16
 
17
17
  class Romanizer:
@@ -39,8 +39,7 @@ class Romanizer:
39
39
  tokenizer_path: Path to tokenizer vocabulary file
40
40
  """
41
41
  self.char_mapper = load_char_mapper()
42
- self.tokenizer = Tokenizer(max_length=None)
43
- self.tokenizer.load_from_pretrained(file_path=None, load_default_tokenizer=True)
42
+ self.tokenizer = Tokenizer.from_pretrained("Ransaka/sinlib")
44
43
 
45
44
  def __call__(self, text: Union[str, List[str]]) -> Union[str, List[str]]:
46
45
  """
@@ -67,32 +66,21 @@ class Romanizer:
67
66
  Romanized version of the input text
68
67
  """
69
68
  text = remove_non_printable(text)
70
- chars: NDArray = np.array(list(text))
71
-
72
- # Create mask for Sinhala characters and allowed punctuation
73
- sinhala_mask = [
74
- char in ALL_SINHALA_CHARACTERS + list(NUMBERS_AND_PUNCTUATION) + [" "]
75
- for char in chars
76
- ]
77
-
78
- # Extract Sinhala text
79
- sinhala_text = "".join(chars[sinhala_mask]).strip()
80
-
81
- # Tokenize and decode
82
- encodings = self.tokenizer(sinhala_text, truncate_and_pad=False)
83
- decoded_chars = [
84
- self.tokenizer.token_id_to_token_map[c] for c in encodings
85
- ]
86
-
87
- # Convert to Roman characters
88
- romanized_chars = [
89
- self.char_mapper.get(ch, ch if ch in NUMBERS_AND_PUNCTUATION.union({" "}) else '')
90
- for ch in decoded_chars
91
- ]
92
- romanized_sinhala = "".join(romanized_chars)
93
-
94
- # Create word mapping and apply to full text
95
- word_mapping = dict(zip(sinhala_text.split(), romanized_sinhala.split()))
96
- romanized_words = [word_mapping.get(word, word) for word in text.split()]
97
-
98
- return " ".join(romanized_words)
69
+ if not text:
70
+ return ""
71
+
72
+ # Fallback mappings for diacritics/modifiers
73
+ char_map = dict(self.char_mapper)
74
+ char_map.setdefault('ං', 'n')
75
+ char_map.setdefault('ඃ', 'h')
76
+
77
+ tokens = process_text(text)
78
+ res = []
79
+ for tok in tokens:
80
+ clean = tok.strip()
81
+ roman = char_map.get(clean, clean)
82
+ if tok.endswith(' ') and not roman.endswith(' '):
83
+ roman += ' '
84
+ res.append(roman)
85
+
86
+ return "".join(res)