sinlib 0.1.13__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. sinlib-0.2.0/.github/workflows/test.yml +32 -0
  2. {sinlib-0.1.13 → sinlib-0.2.0}/.gitignore +6 -0
  3. sinlib-0.2.0/.readthedocs.yaml +11 -0
  4. {sinlib-0.1.13 → sinlib-0.2.0}/PKG-INFO +1 -1
  5. {sinlib-0.1.13 → sinlib-0.2.0}/pyproject.toml +1 -1
  6. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/__init__.py +10 -11
  7. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/encoding.py +54 -15
  8. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/romanize.py +20 -32
  9. sinlib-0.2.0/src/sinlib/spellcheck.py +774 -0
  10. sinlib-0.2.0/src/sinlib/subword.py +179 -0
  11. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/tokenizer.py +14 -4
  12. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/transliterate.py +8 -2
  13. sinlib-0.2.0/src/sinlib/utils/__init__.py +13 -0
  14. sinlib-0.2.0/src/sinlib/utils/dataset_utils.py +10 -0
  15. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/model_utils.py +6 -7
  16. sinlib-0.2.0/src/sinlib/utils/models/akshara_ngram.json +51217 -0
  17. sinlib-0.2.0/src/sinlib/utils/models/akshara_vocab.json +542 -0
  18. sinlib-0.2.0/src/sinlib/utils/models/bigru_detector.pt +0 -0
  19. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/preprocessing.py +84 -13
  20. sinlib-0.2.0/tests/test_combined_detector.py +49 -0
  21. sinlib-0.2.0/tests/test_phase3_enhancements.py +127 -0
  22. sinlib-0.2.0/tests/test_romanizer.py +41 -0
  23. {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_spellcheck.py +1 -1
  24. sinlib-0.2.0/tests/test_transliterator.py +30 -0
  25. sinlib-0.2.0/website/.gitignore +21 -0
  26. sinlib-0.2.0/website/.vscode/extensions.json +4 -0
  27. sinlib-0.2.0/website/.vscode/launch.json +11 -0
  28. sinlib-0.2.0/website/README.md +49 -0
  29. sinlib-0.2.0/website/astro.config.mjs +84 -0
  30. sinlib-0.2.0/website/package-lock.json +6855 -0
  31. sinlib-0.2.0/website/package.json +17 -0
  32. sinlib-0.2.0/website/public/favicon.svg +1 -0
  33. sinlib-0.2.0/website/src/assets/houston.webp +0 -0
  34. sinlib-0.2.0/website/src/assets/logo-dark.png +0 -0
  35. sinlib-0.2.0/website/src/assets/logo-light.png +0 -0
  36. sinlib-0.2.0/website/src/content/docs/api/encoding.mdx +121 -0
  37. sinlib-0.2.0/website/src/content/docs/api/preprocessing.mdx +120 -0
  38. sinlib-0.2.0/website/src/content/docs/api/romanizer.mdx +91 -0
  39. sinlib-0.2.0/website/src/content/docs/api/spellcheck.mdx +191 -0
  40. sinlib-0.2.0/website/src/content/docs/api/tokenizer.mdx +192 -0
  41. sinlib-0.2.0/website/src/content/docs/examples/tokenization.md +46 -0
  42. sinlib-0.2.0/website/src/content/docs/examples/typo-correction.md +97 -0
  43. sinlib-0.1.13/docs/guides/spellcheck.md → sinlib-0.2.0/website/src/content/docs/guides/spellcheck.mdx +15 -37
  44. sinlib-0.1.13/docs/guides/tokenization.md → sinlib-0.2.0/website/src/content/docs/guides/tokenization.mdx +26 -29
  45. sinlib-0.2.0/website/src/content/docs/index.mdx +120 -0
  46. sinlib-0.2.0/website/src/content.config.ts +7 -0
  47. sinlib-0.2.0/website/src/styles/custom.css +337 -0
  48. sinlib-0.2.0/website/tsconfig.json +5 -0
  49. sinlib-0.1.13/.readthedocs.yaml +0 -15
  50. sinlib-0.1.13/docs/api/encoding.md +0 -53
  51. sinlib-0.1.13/docs/api/index.md +0 -26
  52. sinlib-0.1.13/docs/api/preprocessing.md +0 -59
  53. sinlib-0.1.13/docs/api/spellcheck.md +0 -90
  54. sinlib-0.1.13/docs/api/tokenizer.md +0 -109
  55. sinlib-0.1.13/docs/assets/logo.svg +0 -15
  56. sinlib-0.1.13/docs/css/custom.css +0 -327
  57. sinlib-0.1.13/docs/css/extra.css +0 -280
  58. sinlib-0.1.13/docs/custom_theme/main.html +0 -44
  59. sinlib-0.1.13/docs/examples/Tokenization Example.md +0 -1
  60. sinlib-0.1.13/docs/examples/Typo Correction.md +0 -1
  61. sinlib-0.1.13/docs/guides/index.md +0 -16
  62. sinlib-0.1.13/docs/index.md +0 -129
  63. sinlib-0.1.13/docs/requirements.txt +0 -7
  64. sinlib-0.1.13/mkdocs.yml +0 -139
  65. sinlib-0.1.13/src/sinlib/spellcheck.py +0 -367
  66. sinlib-0.1.13/src/sinlib/utils/__init__.py +0 -7
  67. sinlib-0.1.13/src/sinlib/utils/dataset_utils.py +0 -12
  68. sinlib-0.1.13/tests/test_romanizer.py +0 -37
  69. sinlib-0.1.13/tests/test_transliterator.py +0 -0
  70. {sinlib-0.1.13 → sinlib-0.2.0}/.github/workflows/python-publish.yml +0 -0
  71. {sinlib-0.1.13 → sinlib-0.2.0}/LICENSE +0 -0
  72. {sinlib-0.1.13 → sinlib-0.2.0}/README.md +0 -0
  73. {sinlib-0.1.13 → sinlib-0.2.0}/requirements-dev.txt +0 -0
  74. {sinlib-0.1.13 → sinlib-0.2.0}/requirements.txt +0 -0
  75. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/chars.py +0 -0
  76. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/models/__init__.py +0 -0
  77. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/models/transliterator-checkpoint.pth +0 -0
  78. {sinlib-0.1.13 → sinlib-0.2.0}/src/sinlib/utils/models/transliterator_model.py +0 -0
  79. {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer.py +0 -0
  80. {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer_batch.py +0 -0
  81. {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer_bos.py +0 -0
  82. {sinlib-0.1.13 → sinlib-0.2.0}/tests/test_tokenizer_special_tokens.py +0 -0
  83. {sinlib-0.1.13 → sinlib-0.2.0}/welcome.png +0 -0
@@ -0,0 +1,32 @@
1
+ name: Run Tests
2
+
3
+ on:
4
+ push:
5
+ branches: [ main ]
6
+ pull_request:
7
+ branches: [ main ]
8
+
9
+ jobs:
10
+ test:
11
+ name: Test on Python ${{ matrix.python-version }}
12
+ runs-on: ubuntu-latest
13
+ strategy:
14
+ matrix:
15
+ python-version: ["3.9", "3.10", "3.11", "3.12"]
16
+
17
+ steps:
18
+ - uses: actions/checkout@v4
19
+
20
+ - name: Set up Python ${{ matrix.python-version }}
21
+ uses: actions/setup-python@v5
22
+ with:
23
+ python-version: ${{ matrix.python-version }}
24
+
25
+ - name: Install dependencies
26
+ run: |
27
+ python -m pip install --upgrade pip
28
+ pip install -e ".[dev]" -r requirements.txt
29
+
30
+ - name: Run test suite with pytest
31
+ run: |
32
+ pytest --tb=short -v
@@ -11,3 +11,9 @@ experiments/*
11
11
  *.ipynb
12
12
  .qodo
13
13
  examples/examples.ipynb
14
+
15
+ # Astro / Starlight
16
+ website/dist/
17
+ website/node_modules/
18
+ website/.astro/
19
+
@@ -0,0 +1,11 @@
1
+ version: 2
2
+
3
+ build:
4
+ os: ubuntu-24.04
5
+ tools:
6
+ nodejs: "20"
7
+ commands:
8
+ - cd website && npm ci
9
+ - cd website && npm run build
10
+ - mkdir -p $READTHEDOCS_OUTPUT/html
11
+ - cp -r website/dist/* $READTHEDOCS_OUTPUT/html/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sinlib
3
- Version: 0.1.13
3
+ Version: 0.2.0
4
4
  Summary: Sinhala NLP Toolkit
5
5
  Project-URL: Code, https://github.com/Ransaka/sinlib
6
6
  Project-URL: Docs, https://sinlib.readthedocs.io
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sinlib"
3
- version = "0.1.13"
3
+ version = "0.2.0"
4
4
  description = "Sinhala NLP Toolkit"
5
5
  authors = [
6
6
  { name = "Ransaka", email = "ransaka.ravihara@gmail.com" }
@@ -7,16 +7,16 @@ transliteration of Sinhala text, along with preprocessing utilities.
7
7
  Available Classes
8
8
  -----------------
9
9
  Tokenizer
10
- Character-level tokenizer for Sinhala text. Mirrors the HuggingFace
10
+ Character-level tokenizer for Sinhala text. Mirrors the HuggingFace
11
11
  ``PreTrainedTokenizer`` interface — use ``Tokenizer.from_pretrained()``
12
12
  to load the default pretrained vocabulary.
13
13
 
14
14
  BatchEncoding
15
- Dict-like container returned by the tokenizer. Supports attribute access
15
+ Dict-like container returned by the tokenizer. Supports attribute access
16
16
  (``enc.input_ids``, ``enc.attention_mask``) and dict-style access.
17
17
 
18
18
  TypoDetector
19
- N-gram–based spell checker for Sinhala. Use
19
+ N-gram–based spell checker for Sinhala. Use
20
20
  ``TypoDetector.from_pretrained()`` or instantiate directly.
21
21
 
22
22
  preprocessing
@@ -25,30 +25,29 @@ preprocessing
25
25
 
26
26
  Romanizer
27
27
  Converts Sinhala text to Roman (Latin) script.
28
- *Currently disabled from the default namespace due to an ongoing fix.*
29
28
 
30
29
  Transliterator
31
30
  ML-based Sinhala → Roman transliteration using a pre-trained BiLSTM.
32
- *Currently disabled from the default namespace due to an ongoing fix.*
31
+ *Currently disabled from default __all__ pending model vocabulary realignment.*
33
32
  """
34
33
 
35
34
  from typing import List
36
35
 
37
36
  from sinlib.encoding import BatchEncoding
38
37
  from sinlib.tokenizer import Tokenizer
38
+ from sinlib.subword import SubwordTokenizer
39
39
  from sinlib.spellcheck import TypoDetector
40
- from sinlib.utils import preprocessing
41
-
42
- # These are still importable directly but excluded from __all__ while
43
- # their respective bugs are being tracked down.
44
- from sinlib.romanize import Romanizer # noqa: F401
40
+ from sinlib.romanize import Romanizer
45
41
  from sinlib.transliterate import Transliterator # noqa: F401
42
+ from sinlib.utils import preprocessing
46
43
 
47
44
  __all__: List[str] = [
48
45
  "BatchEncoding",
49
46
  "Tokenizer",
47
+ "SubwordTokenizer",
50
48
  "TypoDetector",
49
+ "Romanizer",
51
50
  "preprocessing",
52
51
  ]
53
52
 
54
- __version__ = "0.1.12"
53
+ __version__ = "0.1.13"
@@ -24,6 +24,9 @@ class BatchEncoding:
24
24
  data : dict
25
25
  A dictionary whose values are lists of integers. Expected keys are
26
26
  ``"input_ids"`` and optionally ``"attention_mask"``.
27
+ tensor_type : str, optional
28
+ If set, converts lists to tensors. Supported values: ``"pt"`` (PyTorch),
29
+ ``"tf"`` (TensorFlow), ``"np"`` (NumPy).
27
30
 
28
31
  Examples
29
32
  --------
@@ -36,35 +39,26 @@ class BatchEncoding:
36
39
  True
37
40
  """
38
41
 
39
- def __init__(self, data: Dict[str, Any]) -> None:
42
+ def __init__(self, data: Dict[str, Any], tensor_type: Optional[str] = None) -> None:
40
43
  self._data: Dict[str, Any] = data
44
+ if tensor_type is not None:
45
+ self.convert_to_tensors(tensor_type)
41
46
 
42
47
  # ------------------------------------------------------------------
43
48
  # Attribute access
44
49
  # ------------------------------------------------------------------
45
50
 
46
51
  @property
47
- def input_ids(self) -> List[int]:
52
+ def input_ids(self) -> Any:
48
53
  """
49
- List of token IDs.
50
-
51
- Returns
52
- -------
53
- list of int
54
- The encoded token IDs for the input text.
54
+ Token IDs (list or tensor depending on return_tensors).
55
55
  """
56
56
  return self._data["input_ids"]
57
57
 
58
58
  @property
59
- def attention_mask(self) -> Optional[List[int]]:
59
+ def attention_mask(self) -> Any:
60
60
  """
61
61
  Attention mask (1 for real tokens, 0 for padding).
62
-
63
- Returns
64
- -------
65
- list of int or None
66
- ``1`` for each real token position, ``0`` for padding.
67
- ``None`` when the tokenizer was called without padding.
68
62
  """
69
63
  return self._data.get("attention_mask")
70
64
 
@@ -103,6 +97,51 @@ class BatchEncoding:
103
97
  # Conversion
104
98
  # ------------------------------------------------------------------
105
99
 
100
+ def convert_to_tensors(self, tensor_type: str) -> None:
101
+ """
102
+ Convert the internal input lists to tensors.
103
+ """
104
+ if tensor_type == "pt":
105
+ import torch
106
+ for key, val in self._data.items():
107
+ if val is not None:
108
+ try:
109
+ self._data[key] = torch.tensor(val)
110
+ except ValueError as e:
111
+ raise ValueError(
112
+ f"Failed to convert {key} to PyTorch tensor. "
113
+ f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
114
+ f"Error: {e}"
115
+ ) from e
116
+ elif tensor_type == "tf":
117
+ import tensorflow as tf
118
+ for key, val in self._data.items():
119
+ if val is not None:
120
+ try:
121
+ self._data[key] = tf.convert_to_tensor(val)
122
+ except (ValueError, TypeError) as e:
123
+ raise ValueError(
124
+ f"Failed to convert {key} to TensorFlow tensor. "
125
+ f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
126
+ f"Error: {e}"
127
+ ) from e
128
+ elif tensor_type == "np":
129
+ import numpy as np
130
+ for key, val in self._data.items():
131
+ if val is not None:
132
+ try:
133
+ self._data[key] = np.array(val)
134
+ except ValueError as e:
135
+ raise ValueError(
136
+ f"Failed to convert {key} to NumPy array. "
137
+ f"Ensure sequences are rectangular (uniform length) or padding is enabled. "
138
+ f"Error: {e}"
139
+ ) from e
140
+ else:
141
+ raise ValueError(
142
+ f"Unsupported tensor_type '{tensor_type}'. Use 'pt', 'tf', or 'np'."
143
+ )
144
+
106
145
  def to_dict(self) -> Dict[str, Any]:
107
146
  """
108
147
  Convert to a plain Python dictionary.
@@ -11,7 +11,7 @@ from numpy.typing import NDArray
11
11
 
12
12
  from .tokenizer import Tokenizer
13
13
  from .utils.chars import ALL_SINHALA_CHARACTERS, NUMBERS_AND_PUNCTUATION
14
- from .utils.preprocessing import load_char_mapper, remove_non_printable
14
+ from .utils.preprocessing import load_char_mapper, remove_non_printable, process_text
15
15
 
16
16
 
17
17
  class Romanizer:
@@ -39,8 +39,7 @@ class Romanizer:
39
39
  tokenizer_path: Path to tokenizer vocabulary file
40
40
  """
41
41
  self.char_mapper = load_char_mapper()
42
- self.tokenizer = Tokenizer(max_length=None)
43
- self.tokenizer.load_from_pretrained(file_path=None, load_default_tokenizer=True)
42
+ self.tokenizer = Tokenizer.from_pretrained("Ransaka/sinlib")
44
43
 
45
44
  def __call__(self, text: Union[str, List[str]]) -> Union[str, List[str]]:
46
45
  """
@@ -67,32 +66,21 @@ class Romanizer:
67
66
  Romanized version of the input text
68
67
  """
69
68
  text = remove_non_printable(text)
70
- chars: NDArray = np.array(list(text))
71
-
72
- # Create mask for Sinhala characters and allowed punctuation
73
- sinhala_mask = [
74
- char in ALL_SINHALA_CHARACTERS + list(NUMBERS_AND_PUNCTUATION) + [" "]
75
- for char in chars
76
- ]
77
-
78
- # Extract Sinhala text
79
- sinhala_text = "".join(chars[sinhala_mask]).strip()
80
-
81
- # Tokenize and decode
82
- encodings = self.tokenizer(sinhala_text, truncate_and_pad=False)
83
- decoded_chars = [
84
- self.tokenizer.token_id_to_token_map[c] for c in encodings
85
- ]
86
-
87
- # Convert to Roman characters
88
- romanized_chars = [
89
- self.char_mapper.get(ch, ch if ch in NUMBERS_AND_PUNCTUATION.union({" "}) else '')
90
- for ch in decoded_chars
91
- ]
92
- romanized_sinhala = "".join(romanized_chars)
93
-
94
- # Create word mapping and apply to full text
95
- word_mapping = dict(zip(sinhala_text.split(), romanized_sinhala.split()))
96
- romanized_words = [word_mapping.get(word, word) for word in text.split()]
97
-
98
- return " ".join(romanized_words)
69
+ if not text:
70
+ return ""
71
+
72
+ # Fallback mappings for diacritics/modifiers
73
+ char_map = dict(self.char_mapper)
74
+ char_map.setdefault('ං', 'n')
75
+ char_map.setdefault('ඃ', 'h')
76
+
77
+ tokens = process_text(text)
78
+ res = []
79
+ for tok in tokens:
80
+ clean = tok.strip()
81
+ roman = char_map.get(clean, clean)
82
+ if tok.endswith(' ') and not roman.endswith(' '):
83
+ roman += ' '
84
+ res.append(roman)
85
+
86
+ return "".join(res)