hmaraniam 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hmaraniam-0.1.0/LICENSE +21 -0
- hmaraniam-0.1.0/PKG-INFO +199 -0
- hmaraniam-0.1.0/README.md +173 -0
- hmaraniam-0.1.0/hmaraniam/__init__.py +19 -0
- hmaraniam-0.1.0/hmaraniam/data/shards/unigrams_set_001.json +1 -0
- hmaraniam-0.1.0/hmaraniam/data/sibling_zo_stopwords.json +61 -0
- hmaraniam-0.1.0/hmaraniam/data/stopwords.json +1 -0
- hmaraniam-0.1.0/hmaraniam/detector.py +399 -0
- hmaraniam-0.1.0/hmaraniam.egg-info/PKG-INFO +199 -0
- hmaraniam-0.1.0/hmaraniam.egg-info/SOURCES.txt +13 -0
- hmaraniam-0.1.0/hmaraniam.egg-info/dependency_links.txt +1 -0
- hmaraniam-0.1.0/hmaraniam.egg-info/top_level.txt +1 -0
- hmaraniam-0.1.0/pyproject.toml +44 -0
- hmaraniam-0.1.0/setup.cfg +4 -0
- hmaraniam-0.1.0/tests/test_detector.py +99 -0
hmaraniam-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Hmar Heritage Project
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
hmaraniam-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hmaraniam
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: High-precision language identification library for Hmar ('Hmar a ni am?')
|
|
5
|
+
Author-email: Hmar Heritage Project <info@hmarheritage.org>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/hmar-heritage-org/hmaraniam
|
|
8
|
+
Project-URL: Repository, https://github.com/hmar-heritage-org/hmaraniam
|
|
9
|
+
Project-URL: Bug Tracker, https://github.com/hmar-heritage-org/hmaraniam/issues
|
|
10
|
+
Keywords: hmar,language-identification,nlp,kuki-chin,zo-languages,language-detection,hmaraniam
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
22
|
+
Requires-Python: >=3.8
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# hmaraniam 🇲z
|
|
28
|
+
|
|
29
|
+
**High-precision, zero-dependency language identification library for Hmar.**
|
|
30
|
+
|
|
31
|
+
> *"Hmar a ni am?"* — *"Is it Hmar?"*
|
|
32
|
+
|
|
33
|
+
`hmaraniam` is a lightweight Python library designed to accurately distinguish Hmar text from English and other Kuki-Chin / Zo languages (Mizo, Kuki, Paite, Vaiphei).
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## Key Features
|
|
38
|
+
|
|
39
|
+
- **Microsecond Speed:** $O(1)$ dictionary lookups with no heavy ML dependencies (PyTorch/TensorFlow free).
|
|
40
|
+
- **Dual-Lens Diacritic Engine:** Reports both `casual_hmar_ratio` (ASCII-normalized for standard QWERTY typing) and `formal_hmar_ratio` (exact diacritic matching for formal literary text).
|
|
41
|
+
- **Dual Continuous Confidence Metrics:** Disambiguates language metrics into `hmar_confidence` (permanent metric answering *"How confident are we that this text is Hmar?"*) and `detected_language_confidence` (confidence in the overall classification choice).
|
|
42
|
+
- **Standardized Schema:** Guarantees an immutable JSON output structure across all calls, including `non_hmar_words_count` and `diacritic_words_count`.
|
|
43
|
+
- **Extensible & Customizable:** Allows developers to supply `custom_unigrams`, `extra_unigrams`, `custom_stopwords`, or disable default stopwords.
|
|
44
|
+
- **Dual Offline/CDN Architecture:** Automatically syncs with the live `hmar-heritage-org/hmaraniam` unigram dataset via jsDelivr CDN, with automatic local disk caching and bundled fallback.
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install hmaraniam
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
---
|
|
55
|
+
|
|
56
|
+
## Standard Output Schema
|
|
57
|
+
|
|
58
|
+
```json
|
|
59
|
+
{
|
|
60
|
+
"language": "hmar",
|
|
61
|
+
"hmar_confidence": 0.9842,
|
|
62
|
+
"detected_language_confidence": 0.9842,
|
|
63
|
+
"mode": "basic",
|
|
64
|
+
"scores": {
|
|
65
|
+
"casual_hmar_ratio": 0.9524,
|
|
66
|
+
"formal_hmar_ratio": 0.8095,
|
|
67
|
+
"english_stopword_ratio": 0.0000,
|
|
68
|
+
"sibling_zo_stopword_ratio": 0.0000,
|
|
69
|
+
"unknown_words_ratio": 0.0476,
|
|
70
|
+
"total_words": 21,
|
|
71
|
+
"hmar_words_count": 20,
|
|
72
|
+
"non_hmar_words_count": 1,
|
|
73
|
+
"unknown_words_count": 1,
|
|
74
|
+
"english_stopwords_count": 0,
|
|
75
|
+
"sibling_zo_stopwords_count": 0,
|
|
76
|
+
"hmar_diacritic_words_count": 17,
|
|
77
|
+
"non_hmar_diacritic_words_count": 0,
|
|
78
|
+
"total_diacritic_words_count": 17
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
---
|
|
84
|
+
|
|
85
|
+
## Usage
|
|
86
|
+
|
|
87
|
+
### Quick Start
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
import hmaraniam
|
|
91
|
+
|
|
92
|
+
# Authentic text from L. Keivom archive
|
|
93
|
+
sample_text = "Khawvel fe dan phung ei en chun, ram le hnam damna thuruk chu lien lema intel le insung khawm a nih."
|
|
94
|
+
|
|
95
|
+
result = hmaraniam.detect(sample_text)
|
|
96
|
+
|
|
97
|
+
print(result)
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
### Custom Unigrams & Stopwords
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from hmaraniam import Detector
|
|
104
|
+
|
|
105
|
+
# Provide custom unigrams or extra domain vocabulary
|
|
106
|
+
detector = Detector(
|
|
107
|
+
mode="basic",
|
|
108
|
+
extra_unigrams=["customworda", "customwordb"],
|
|
109
|
+
custom_stopwords=["and", "the", "with"],
|
|
110
|
+
disable_default_stopwords=False
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
result = detector.detect("Khawvel fe dan phung...")
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
### Modes & Advanced Options
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
from hmaraniam import Detector
|
|
120
|
+
|
|
121
|
+
# Basic Mode (Active default ~30k core unigrams)
|
|
122
|
+
basic_detector = Detector(mode="basic")
|
|
123
|
+
|
|
124
|
+
# High Mode (Scans data/shards/ and loads all available unigram shards, falling back seamlessly to basic)
|
|
125
|
+
high_detector = Detector(mode="high")
|
|
126
|
+
|
|
127
|
+
# Offline-only mode (uses cached/bundled dataset without network calls)
|
|
128
|
+
offline_detector = Detector(offline_only=True)
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
---
|
|
132
|
+
|
|
133
|
+
## Empirical Benchmarks & Performance
|
|
134
|
+
|
|
135
|
+
`hmaraniam` has been empirically validated across both **controlled parallel corpus datasets** (Parallel Zo Bibles across multiple literary genres) and **unfiltered real-world web archives** (~1,300 scraped articles and raw HTML pages across 5 major publishers).
|
|
136
|
+
|
|
137
|
+
### 1. Controlled Parallel Zo Bible Benchmark
|
|
138
|
+
|
|
139
|
+
Evaluated across parallel chapters (Genesis 1, Exodus 20, Matthew 5, Luke 2, Romans 8, Revelation 21) across 10 parallel Bible translations in 8 Zo languages + English:
|
|
140
|
+
|
|
141
|
+
| Language | Edition / Source | Evaluated Passages | Target Class | Engine Assigned Label | Classification Accuracy | Avg Hmar Confidence | Key Distinguishing Metrics |
|
|
142
|
+
| :--- | :--- | :--- | :---: | :---: | :---: | :---: | :--- |
|
|
143
|
+
| **Hmar** | CLB (Contemporary) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `hmar` | `hmar` | **100%** | **1.0000** | `casual_hmar_ratio` $\ge 0.95$, `unknown_words_ratio` $\le 0.05$ |
|
|
144
|
+
| **Hmar** | OV (Old Version) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `hmar` | `hmar` | **100%** | **1.0000** | Formal diacritic match + `hmar_diacritic_words_count` $>0$ |
|
|
145
|
+
| **Mizo** | OV (Mizo Bible) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathian`, `hnenah`, `avangin`, `tichuan`) & `unknown_words_ratio` ($\approx 24\%$) |
|
|
146
|
+
| **Paite** | Paite Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pasian`, `tungah`, `simhuai`) & `unknown_words_ratio` ($\approx 51\%$) |
|
|
147
|
+
| **Vaiphei** | Vaiphei Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathian`, `tiu-in`, `apat`) & `unknown_words_ratio` ($\approx 38\%$) |
|
|
148
|
+
| **Gangte** | Gangte Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathen`, `hepa`, `dih-in`) & `unknown_words_ratio` ($\approx 40\%$) |
|
|
149
|
+
| **Zou** | Zou Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pasian`, `a-in`, `leh-in`) & `unknown_words_ratio` ($\approx 51\%$) |
|
|
150
|
+
| **Thadou** | Thadou-Kuki Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathen`, `chun`, `tichun`) & `unknown_words_ratio` ($\approx 67\%$) |
|
|
151
|
+
| **English** | WEB (World English) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `english` | `english` | **100%** | **0.0000** | `english_stopword_ratio` ($>0.08$) & `unknown_words_ratio` ($>0.70$) |
|
|
152
|
+
|
|
153
|
+
> **Cognate Resolution:** Closely related sibling languages like Mizo share up to 78% unigram overlap with Hmar. `hmaraniam` cleanly resolves sibling Zo languages without requiring massive full dictionaries by combining vocabulary completeness (`unknown_words_ratio` $\le 0.18$) with curated structural markers (`sibling_zo_stopwords`).
|
|
154
|
+
|
|
155
|
+
---
|
|
156
|
+
|
|
157
|
+
### 2. Real-World Web Archive Benchmark (~1,300 Scraped Documents)
|
|
158
|
+
|
|
159
|
+
Evaluated on real-world scraped web archives across 5 major Hmar/Zo web publishers:
|
|
160
|
+
|
|
161
|
+
| Publisher Web Archive | Corpus Source / Format | Evaluated Items | Hmar Posts Detected (%) | English Posts Detected (%) | Other / Mixed Posts (%) | Avg Hmar Confidence | Mean Casual Hmar Ratio | Mean Formal Hmar Ratio | Mean Unknown Words Ratio | Primary Content Profile |
|
|
162
|
+
| :--- | :--- | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :--- |
|
|
163
|
+
| **L. Keivom Archive** (`keivom`) | Blogger API JSON | 300 | **199 (66.3%)** | 32 (10.7%) | 69 (23.0%) | **0.8320** | **79.3%** | 70.0% | 20.7% | Authentic Hmar literary essays & prose |
|
|
164
|
+
| **Inpui Journal** (`inpui`) | Blogger API JSON | 291 | **124 (42.6%)** | 67 (23.0%) | 100 (34.4%) | **0.7400** | **63.7%** | 55.7% | 36.3% | Bilingual news journal & opinion pieces |
|
|
165
|
+
| **HSA Portal** (`hsa`) | WordPress API JSON | 177 | **23 (13.0%)** | 38 (21.5%) | **116 (65.5%)** | 0.5730 | **53.7%** | 46.7% | 46.3% | Student association alerts & mixed posts |
|
|
166
|
+
| **Hmarram.com** (`hmarram`) | WordPress API JSON | 235 | 12 (5.1%) | **209 (88.9%)** | 14 (6.0%) | 0.7762 | 31.2% | 23.8% | 68.8% | Tech articles & English press releases |
|
|
167
|
+
| **Virthli News** (`virthli`) | Scraped Raw HTML | 296 | 5 (1.7%) | **267 (90.2%)** | 24 (8.1%) | 0.8721 | 17.6% | 13.4% | 82.4% | Employment alerts & exam guidelines |
|
|
168
|
+
|
|
169
|
+
- **Literary Archives:** Archives like L. Keivom and Inpui Journal feature rich Hmar prose and opinion articles, achieving high Hmar classification rates ($42.6\% - 66.3\%$) and high token ratios ($63.7\% - 79.3\%$).
|
|
170
|
+
- **Community News & Recruitment Notices:** Community portals like Hmarram and Virthli publish predominantly in English (recruitment notices, exam guidelines, press statements). `hmaraniam` accurately tags these as `"english"` or `"other"` without false-positive over-classification.
|
|
171
|
+
|
|
172
|
+
---
|
|
173
|
+
|
|
174
|
+
## Error Handling
|
|
175
|
+
|
|
176
|
+
`hmaraniam` provides clear, descriptive error messages:
|
|
177
|
+
|
|
178
|
+
```python
|
|
179
|
+
import hmaraniam
|
|
180
|
+
|
|
181
|
+
# Raises ValueError for unsupported modes
|
|
182
|
+
try:
|
|
183
|
+
hmaraniam.detect("Text", mode="ultra")
|
|
184
|
+
except ValueError as e:
|
|
185
|
+
print(e)
|
|
186
|
+
|
|
187
|
+
# Raises TypeError for non-string input
|
|
188
|
+
try:
|
|
189
|
+
hmaraniam.detect(12345)
|
|
190
|
+
except TypeError as e:
|
|
191
|
+
print(e)
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
---
|
|
195
|
+
|
|
196
|
+
## License
|
|
197
|
+
|
|
198
|
+
Published under the MIT License by the **Hmar Heritage Project**.
|
|
199
|
+
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
# hmaraniam 🇲z
|
|
2
|
+
|
|
3
|
+
**High-precision, zero-dependency language identification library for Hmar.**
|
|
4
|
+
|
|
5
|
+
> *"Hmar a ni am?"* — *"Is it Hmar?"*
|
|
6
|
+
|
|
7
|
+
`hmaraniam` is a lightweight Python library designed to accurately distinguish Hmar text from English and other Kuki-Chin / Zo languages (Mizo, Kuki, Paite, Vaiphei).
|
|
8
|
+
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
## Key Features
|
|
12
|
+
|
|
13
|
+
- **Microsecond Speed:** $O(1)$ dictionary lookups with no heavy ML dependencies (PyTorch/TensorFlow free).
|
|
14
|
+
- **Dual-Lens Diacritic Engine:** Reports both `casual_hmar_ratio` (ASCII-normalized for standard QWERTY typing) and `formal_hmar_ratio` (exact diacritic matching for formal literary text).
|
|
15
|
+
- **Dual Continuous Confidence Metrics:** Disambiguates language metrics into `hmar_confidence` (permanent metric answering *"How confident are we that this text is Hmar?"*) and `detected_language_confidence` (confidence in the overall classification choice).
|
|
16
|
+
- **Standardized Schema:** Guarantees an immutable JSON output structure across all calls, including `non_hmar_words_count` and `diacritic_words_count`.
|
|
17
|
+
- **Extensible & Customizable:** Allows developers to supply `custom_unigrams`, `extra_unigrams`, `custom_stopwords`, or disable default stopwords.
|
|
18
|
+
- **Dual Offline/CDN Architecture:** Automatically syncs with the live `hmar-heritage-org/hmaraniam` unigram dataset via jsDelivr CDN, with automatic local disk caching and bundled fallback.
|
|
19
|
+
|
|
20
|
+
---
|
|
21
|
+
|
|
22
|
+
## Installation
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install hmaraniam
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
## Standard Output Schema
|
|
31
|
+
|
|
32
|
+
```json
|
|
33
|
+
{
|
|
34
|
+
"language": "hmar",
|
|
35
|
+
"hmar_confidence": 0.9842,
|
|
36
|
+
"detected_language_confidence": 0.9842,
|
|
37
|
+
"mode": "basic",
|
|
38
|
+
"scores": {
|
|
39
|
+
"casual_hmar_ratio": 0.9524,
|
|
40
|
+
"formal_hmar_ratio": 0.8095,
|
|
41
|
+
"english_stopword_ratio": 0.0000,
|
|
42
|
+
"sibling_zo_stopword_ratio": 0.0000,
|
|
43
|
+
"unknown_words_ratio": 0.0476,
|
|
44
|
+
"total_words": 21,
|
|
45
|
+
"hmar_words_count": 20,
|
|
46
|
+
"non_hmar_words_count": 1,
|
|
47
|
+
"unknown_words_count": 1,
|
|
48
|
+
"english_stopwords_count": 0,
|
|
49
|
+
"sibling_zo_stopwords_count": 0,
|
|
50
|
+
"hmar_diacritic_words_count": 17,
|
|
51
|
+
"non_hmar_diacritic_words_count": 0,
|
|
52
|
+
"total_diacritic_words_count": 17
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
---
|
|
58
|
+
|
|
59
|
+
## Usage
|
|
60
|
+
|
|
61
|
+
### Quick Start
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
import hmaraniam
|
|
65
|
+
|
|
66
|
+
# Authentic text from L. Keivom archive
|
|
67
|
+
sample_text = "Khawvel fe dan phung ei en chun, ram le hnam damna thuruk chu lien lema intel le insung khawm a nih."
|
|
68
|
+
|
|
69
|
+
result = hmaraniam.detect(sample_text)
|
|
70
|
+
|
|
71
|
+
print(result)
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
### Custom Unigrams & Stopwords
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
from hmaraniam import Detector
|
|
78
|
+
|
|
79
|
+
# Provide custom unigrams or extra domain vocabulary
|
|
80
|
+
detector = Detector(
|
|
81
|
+
mode="basic",
|
|
82
|
+
extra_unigrams=["customworda", "customwordb"],
|
|
83
|
+
custom_stopwords=["and", "the", "with"],
|
|
84
|
+
disable_default_stopwords=False
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
result = detector.detect("Khawvel fe dan phung...")
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### Modes & Advanced Options
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
from hmaraniam import Detector
|
|
94
|
+
|
|
95
|
+
# Basic Mode (Active default ~30k core unigrams)
|
|
96
|
+
basic_detector = Detector(mode="basic")
|
|
97
|
+
|
|
98
|
+
# High Mode (Scans data/shards/ and loads all available unigram shards, falling back seamlessly to basic)
|
|
99
|
+
high_detector = Detector(mode="high")
|
|
100
|
+
|
|
101
|
+
# Offline-only mode (uses cached/bundled dataset without network calls)
|
|
102
|
+
offline_detector = Detector(offline_only=True)
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
---
|
|
106
|
+
|
|
107
|
+
## Empirical Benchmarks & Performance
|
|
108
|
+
|
|
109
|
+
`hmaraniam` has been empirically validated across both **controlled parallel corpus datasets** (Parallel Zo Bibles across multiple literary genres) and **unfiltered real-world web archives** (~1,300 scraped articles and raw HTML pages across 5 major publishers).
|
|
110
|
+
|
|
111
|
+
### 1. Controlled Parallel Zo Bible Benchmark
|
|
112
|
+
|
|
113
|
+
Evaluated across parallel chapters (Genesis 1, Exodus 20, Matthew 5, Luke 2, Romans 8, Revelation 21) across 10 parallel Bible translations in 8 Zo languages + English:
|
|
114
|
+
|
|
115
|
+
| Language | Edition / Source | Evaluated Passages | Target Class | Engine Assigned Label | Classification Accuracy | Avg Hmar Confidence | Key Distinguishing Metrics |
|
|
116
|
+
| :--- | :--- | :--- | :---: | :---: | :---: | :---: | :--- |
|
|
117
|
+
| **Hmar** | CLB (Contemporary) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `hmar` | `hmar` | **100%** | **1.0000** | `casual_hmar_ratio` $\ge 0.95$, `unknown_words_ratio` $\le 0.05$ |
|
|
118
|
+
| **Hmar** | OV (Old Version) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `hmar` | `hmar` | **100%** | **1.0000** | Formal diacritic match + `hmar_diacritic_words_count` $>0$ |
|
|
119
|
+
| **Mizo** | OV (Mizo Bible) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathian`, `hnenah`, `avangin`, `tichuan`) & `unknown_words_ratio` ($\approx 24\%$) |
|
|
120
|
+
| **Paite** | Paite Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pasian`, `tungah`, `simhuai`) & `unknown_words_ratio` ($\approx 51\%$) |
|
|
121
|
+
| **Vaiphei** | Vaiphei Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathian`, `tiu-in`, `apat`) & `unknown_words_ratio` ($\approx 38\%$) |
|
|
122
|
+
| **Gangte** | Gangte Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathen`, `hepa`, `dih-in`) & `unknown_words_ratio` ($\approx 40\%$) |
|
|
123
|
+
| **Zou** | Zou Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pasian`, `a-in`, `leh-in`) & `unknown_words_ratio` ($\approx 51\%$) |
|
|
124
|
+
| **Thadou** | Thadou-Kuki Bible | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `other` | `other` | **100%** | **0.0000** | `sibling_zo_stopwords` (`pathen`, `chun`, `tichun`) & `unknown_words_ratio` ($\approx 67\%$) |
|
|
125
|
+
| **English** | WEB (World English) | Gen 1, Ex 20, Matt 5, Luke 2, Rom 8, Rev 21 | `english` | `english` | **100%** | **0.0000** | `english_stopword_ratio` ($>0.08$) & `unknown_words_ratio` ($>0.70$) |
|
|
126
|
+
|
|
127
|
+
> **Cognate Resolution:** Closely related sibling languages like Mizo share up to 78% unigram overlap with Hmar. `hmaraniam` cleanly resolves sibling Zo languages without requiring massive full dictionaries by combining vocabulary completeness (`unknown_words_ratio` $\le 0.18$) with curated structural markers (`sibling_zo_stopwords`).
|
|
128
|
+
|
|
129
|
+
---
|
|
130
|
+
|
|
131
|
+
### 2. Real-World Web Archive Benchmark (~1,300 Scraped Documents)
|
|
132
|
+
|
|
133
|
+
Evaluated on real-world scraped web archives across 5 major Hmar/Zo web publishers:
|
|
134
|
+
|
|
135
|
+
| Publisher Web Archive | Corpus Source / Format | Evaluated Items | Hmar Posts Detected (%) | English Posts Detected (%) | Other / Mixed Posts (%) | Avg Hmar Confidence | Mean Casual Hmar Ratio | Mean Formal Hmar Ratio | Mean Unknown Words Ratio | Primary Content Profile |
|
|
136
|
+
| :--- | :--- | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :--- |
|
|
137
|
+
| **L. Keivom Archive** (`keivom`) | Blogger API JSON | 300 | **199 (66.3%)** | 32 (10.7%) | 69 (23.0%) | **0.8320** | **79.3%** | 70.0% | 20.7% | Authentic Hmar literary essays & prose |
|
|
138
|
+
| **Inpui Journal** (`inpui`) | Blogger API JSON | 291 | **124 (42.6%)** | 67 (23.0%) | 100 (34.4%) | **0.7400** | **63.7%** | 55.7% | 36.3% | Bilingual news journal & opinion pieces |
|
|
139
|
+
| **HSA Portal** (`hsa`) | WordPress API JSON | 177 | **23 (13.0%)** | 38 (21.5%) | **116 (65.5%)** | 0.5730 | **53.7%** | 46.7% | 46.3% | Student association alerts & mixed posts |
|
|
140
|
+
| **Hmarram.com** (`hmarram`) | WordPress API JSON | 235 | 12 (5.1%) | **209 (88.9%)** | 14 (6.0%) | 0.7762 | 31.2% | 23.8% | 68.8% | Tech articles & English press releases |
|
|
141
|
+
| **Virthli News** (`virthli`) | Scraped Raw HTML | 296 | 5 (1.7%) | **267 (90.2%)** | 24 (8.1%) | 0.8721 | 17.6% | 13.4% | 82.4% | Employment alerts & exam guidelines |
|
|
142
|
+
|
|
143
|
+
- **Literary Archives:** Archives like L. Keivom and Inpui Journal feature rich Hmar prose and opinion articles, achieving high Hmar classification rates ($42.6\% - 66.3\%$) and high token ratios ($63.7\% - 79.3\%$).
|
|
144
|
+
- **Community News & Recruitment Notices:** Community portals like Hmarram and Virthli publish predominantly in English (recruitment notices, exam guidelines, press statements). `hmaraniam` accurately tags these as `"english"` or `"other"` without false-positive over-classification.
|
|
145
|
+
|
|
146
|
+
---
|
|
147
|
+
|
|
148
|
+
## Error Handling
|
|
149
|
+
|
|
150
|
+
`hmaraniam` provides clear, descriptive error messages:
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
import hmaraniam
|
|
154
|
+
|
|
155
|
+
# Raises ValueError for unsupported modes
|
|
156
|
+
try:
|
|
157
|
+
hmaraniam.detect("Text", mode="ultra")
|
|
158
|
+
except ValueError as e:
|
|
159
|
+
print(e)
|
|
160
|
+
|
|
161
|
+
# Raises TypeError for non-string input
|
|
162
|
+
try:
|
|
163
|
+
hmaraniam.detect(12345)
|
|
164
|
+
except TypeError as e:
|
|
165
|
+
print(e)
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
---
|
|
169
|
+
|
|
170
|
+
## License
|
|
171
|
+
|
|
172
|
+
Published under the MIT License by the **Hmar Heritage Project**.
|
|
173
|
+
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""
|
|
2
|
+
hmaraniam - High-precision language identification for Hmar.
|
|
3
|
+
"Hmar a ni am?" -> "Is it Hmar?"
|
|
4
|
+
|
|
5
|
+
Pure string language identification engine.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from hmaraniam.detector import Detector, detect, get_default_detector
|
|
9
|
+
|
|
10
|
+
__version__ = "0.1.0"
|
|
11
|
+
__author__ = "Hmar Heritage Project"
|
|
12
|
+
__license__ = "MIT"
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"Detector",
|
|
16
|
+
"detect",
|
|
17
|
+
"get_default_detector",
|
|
18
|
+
"__version__",
|
|
19
|
+
]
|