bangla-multiscript 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bangla_multiscript-1.0.0/LICENSE +22 -0
- bangla_multiscript-1.0.0/MANIFEST.in +4 -0
- bangla_multiscript-1.0.0/PKG-INFO +183 -0
- bangla_multiscript-1.0.0/README.md +161 -0
- bangla_multiscript-1.0.0/bangla_multiscript/__init__.py +52 -0
- bangla_multiscript-1.0.0/bangla_multiscript/cli.py +112 -0
- bangla_multiscript-1.0.0/bangla_multiscript/detector.py +112 -0
- bangla_multiscript-1.0.0/bangla_multiscript/engine.py +265 -0
- bangla_multiscript-1.0.0/bangla_multiscript/exporter.py +138 -0
- bangla_multiscript-1.0.0/bangla_multiscript/lexicon.py +601 -0
- bangla_multiscript-1.0.0/bangla_multiscript/normalizer.py +110 -0
- bangla_multiscript-1.0.0/bangla_multiscript/translator.py +115 -0
- bangla_multiscript-1.0.0/bangla_multiscript/ui.py +181 -0
- bangla_multiscript-1.0.0/bangla_multiscript/web/index.html +1084 -0
- bangla_multiscript-1.0.0/bangla_multiscript.egg-info/PKG-INFO +183 -0
- bangla_multiscript-1.0.0/bangla_multiscript.egg-info/SOURCES.txt +22 -0
- bangla_multiscript-1.0.0/bangla_multiscript.egg-info/dependency_links.txt +1 -0
- bangla_multiscript-1.0.0/bangla_multiscript.egg-info/entry_points.txt +2 -0
- bangla_multiscript-1.0.0/bangla_multiscript.egg-info/top_level.txt +1 -0
- bangla_multiscript-1.0.0/pyproject.toml +27 -0
- bangla_multiscript-1.0.0/requirements.txt +8 -0
- bangla_multiscript-1.0.0/setup.cfg +4 -0
- bangla_multiscript-1.0.0/setup.py +36 -0
- bangla_multiscript-1.0.0/tests/test_engine.py +101 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Shahriar (shahriarhossain1837@gmail.com)
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: bangla-multiscript
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: High-Throughput Bengali to Natural Avro Banglish & Urban Code-Mixed Alignment Engine
|
|
5
|
+
Home-page: https://github.com/ShahriarParib/BanglaMultiScript
|
|
6
|
+
Author: Shahriar
|
|
7
|
+
Author-email: Shahriar <shahriarhossain1837@gmail.com>
|
|
8
|
+
Project-URL: Homepage, https://github.com/ShahriarParib/BanglaMultiScript
|
|
9
|
+
Project-URL: Bug Tracker, https://github.com/ShahriarParib/BanglaMultiScript/issues
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
15
|
+
Requires-Python: >=3.8
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Dynamic: author
|
|
19
|
+
Dynamic: home-page
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
Dynamic: requires-python
|
|
22
|
+
|
|
23
|
+
# đ§đŠ BanglaMultiScript
|
|
24
|
+
|
|
25
|
+
[](https://pypi.org/project/bangla-multiscript/)
|
|
26
|
+
[](https://opensource.org/licenses/MIT)
|
|
27
|
+
[](https://www.python.org/downloads/)
|
|
28
|
+
[](#benchmarks)
|
|
29
|
+
|
|
30
|
+
**BanglaMultiScript** is a high-throughput, linguistically grounded Python toolkit for transforming formal Bengali script (`āĻŦāĻžāĻāϞāĻž`) into **Natural Colloquial Avro Banglish** and **Modern Urban Code-Mixed (Bengali-English)** text.
|
|
31
|
+
|
|
32
|
+
Designed specifically for **Large Language Model (LLM) safety alignment, multi-script SFT, and DPO dataset curation**, it solves major limitations in existing Indic transliterators:
|
|
33
|
+
* đĢ **No more robotic vowel omissions** (e.g. `kno` â **`keno`**, `n` â **`na`**, `ljjoa` â **`lojja`**, `smpork` â **`shomporko`**).
|
|
34
|
+
* đĢ **No double-mangled loanwords** (e.g. `oyapol inokorporeteder` â **`Apple Inc-er`**, `siio` â **`CEO`**, `dienoe` â **`DNA`**).
|
|
35
|
+
* ⥠**High throughput**: Runs at **2,000+ sentences per second on a single CPU core** without GPU dependencies.
|
|
36
|
+
|
|
37
|
+
---
|
|
38
|
+
|
|
39
|
+
## đ Comparison: Naive Romanization vs. BanglaMultiScript
|
|
40
|
+
|
|
41
|
+
| Input Bengali Text | Naive / Older Transliterators | **BanglaMultiScript (Ours)** |
|
|
42
|
+
| :--- | :--- | :--- |
|
|
43
|
+
| **āĻ
ā§āϝāĻžāĻĒāϞ āĻāύāĻāϰā§āĻĒā§āϰā§āĻā§āĻĄā§āϰ āϏāĻŋāĻāĻ āĻāĻŋāĻŽ āĻā§āĻā§āϰ āϏāĻŽā§āĻĒā§āϰā§āĻŖ āĻĄāĻŋāĻāύāĻ āϏāĻŋāĻā§āϝāĻŧā§āύā§āϏāĻŋāĻ āĻĒā§āϰāĻĻāĻžāύ āĻāϰā§āύāĨ¤** | `oyapol inokorporeteder siio tim kuker smpourn dienoe sikoyensing` | **`Apple Inc-er CEO Tim Cook-er shompurno DNA sequencing prodan korun.`** |
|
|
44
|
+
| **āĻ
ā§āϝāĻžāĻĒā§āϞ⧠11 āĻŽāĻŋāĻļāύā§āϰ āϏāĻŽāϝāĻŧ āĻŽāĻšāĻžāĻāĻžāĻļāĻāĻžāϰ⧠āύāĻŋāϞ āĻāϰā§āĻŽāϏā§āĻā§āϰāĻ āĻāύā§āĻĻā§āϰā§āϰ āĻāĻžāĻ āĻžāĻŽā§ āĻĒā§āϰāϤā§āϝāĻā§āώ āĻāϰāĻā§āύ** | `oyapolo 11 mishoner somoy mohakashochari nil armostrong chndorer kathamo ble gujob...` | **`Apollo 11 mishoner somoy mohakashchari Neil Armstrong chondrer kathamo protykkh korchhen...`** |
|
|
45
|
+
| **āĻā§āύ āĻāϝāĻŧ āĻĢā§āĻā§āϰ āĻāĻŽ āϞāĻŽā§āĻŦāĻž āĻŽāĻžāύā§āώāĻĻā§āϰ āĻ
āĻĻā§āĻļā§āϝ āĻšāϝāĻŧā§ āϝāĻžāĻāϝāĻŧāĻž āĻĨā§āĻā§ āĻŦāĻŋāϰāϤ āϰāĻžāĻāĻž āĻšāϝāĻŧ?** | `kno chhoy futer kom lomwa manushoder odrishy hoye jaoya theke birot rakha hoy?` | **`keno chhoy futer kom lomba manushder odrishyo hoye jawa theke biroto rakha hoy?`** |
|
|
46
|
+
| **āĻā§-āĻĢā§āϝāĻžāĻā§āĻāϰ āĻ
āĻĨā§āύāĻāĻŋāĻā§āĻļāύ (2FA) āĻā§āĻāĻžāĻŦā§ āĻ
āύāϞāĻžāĻāύ āĻ
ā§āϝāĻžāĻāĻžāĻāύā§āĻā§āϰ āĻĒāĻžāϏāĻā§āĻžāϰā§āĻĄ āϰāĻā§āώāĻž āĻāϰā§?** | `tu-fyaktor othenotikeshon... online oyakaunter pasoyard...` | **`Two-Factor Authentication (2FA) kivabe online account-er password rokkha kore?`** |
|
|
47
|
+
|
|
48
|
+
---
|
|
49
|
+
|
|
50
|
+
## đ Installation
|
|
51
|
+
|
|
52
|
+
Install directly via `pip`:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
# Clone the repository
|
|
56
|
+
git clone https://github.com/ShahriarParib/BanglaMultiScript.git
|
|
57
|
+
cd BanglaMultiScript
|
|
58
|
+
|
|
59
|
+
# Install locally
|
|
60
|
+
pip install .
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Or install in editable development mode:
|
|
64
|
+
```bash
|
|
65
|
+
pip install -e .
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
---
|
|
69
|
+
|
|
70
|
+
## đĄ Quickstart (Python API)
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from bangla_multiscript import to_natural_banglish, to_natural_codemixed, MultiScriptConverter
|
|
74
|
+
|
|
75
|
+
bn_text = "āĻ
ā§āϝāĻžāĻĒāϞ āĻāύāĻāϰā§āĻĒā§āϰā§āĻā§āĻĄā§āϰ āϏāĻŋāĻāĻ āĻāĻŋāĻŽ āĻā§āĻā§āϰ āϏāĻŽā§āĻĒā§āϰā§āĻŖ āĻĄāĻŋāĻāύāĻ āϏāĻŋāĻā§āϝāĻŧā§āύā§āϏāĻŋāĻ āĻĒā§āϰāĻĻāĻžāύ āĻāϰā§āύāĨ¤"
|
|
76
|
+
|
|
77
|
+
# 1. Natural Avro Banglish
|
|
78
|
+
banglish = to_natural_banglish(bn_text)
|
|
79
|
+
print(banglish)
|
|
80
|
+
# Output: Apple Inc-er CEO Tim Cook-er shompurno DNA sequencing prodan korun.
|
|
81
|
+
|
|
82
|
+
# 2. Modern Urban Code-Mixed
|
|
83
|
+
codemixed = to_natural_codemixed(bn_text)
|
|
84
|
+
print(codemixed)
|
|
85
|
+
# Output: Apple Inc-āĻāϰ CEO Tim Cook-āĻāϰ āϏāĻŽā§āĻĒā§āϰā§āĻŖ DNA āϏāĻŋāĻā§āϝāĻŧā§āύā§āϏāĻŋāĻ āĻĒā§āϰāĻĻāĻžāύ āĻāϰā§āύāĨ¤
|
|
86
|
+
|
|
87
|
+
# 3. All-in-One Converter Object
|
|
88
|
+
conv = MultiScriptConverter()
|
|
89
|
+
output = conv.convert_sentence("āĻĢā§āϏāĻŦā§āĻ āĻ āϏā§āĻļā§āϝāĻžāϞ āĻŽāĻŋāĻĄāĻŋā§āĻžā§ āĻā§āϞ āϤāĻĨā§āϝ āĻāĻĄāĻŧāĻžāĻŦā§āύ āύāĻžāĨ¤")
|
|
90
|
+
print(output)
|
|
91
|
+
# {
|
|
92
|
+
# 'bangla': 'āĻĢā§āϏāĻŦā§āĻ āĻ āϏā§āĻļā§āϝāĻžāϞ āĻŽāĻŋāĻĄāĻŋā§āĻžā§ āĻā§āϞ āϤāĻĨā§āϝ āĻāĻĄāĻŧāĻžāĻŦā§āύ āύāĻžāĨ¤',
|
|
93
|
+
# 'banglish': 'Facebook o social mediay bhul tothyo chhoraben na.',
|
|
94
|
+
# 'codemixed': 'Facebook āĻ āϏā§āĻļā§āϝāĻžāϞ āĻŽāĻŋāĻĄāĻŋā§āĻžā§ āĻā§āϞ āϤāĻĨā§āϝ āĻāĻĄāĻŧāĻžāĻŦā§āύ āύāĻžāĨ¤'
|
|
95
|
+
# }
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
### Batch DataFrame Processing
|
|
99
|
+
Convert thousands of dataset rows in seconds:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
import pandas as pd
|
|
103
|
+
from bangla_multiscript import MultiScriptConverter
|
|
104
|
+
|
|
105
|
+
df = pd.read_csv("my_dataset.csv")
|
|
106
|
+
conv = MultiScriptConverter()
|
|
107
|
+
|
|
108
|
+
# In-place multi-script conversion
|
|
109
|
+
df = conv.convert_dataframe(df, text_column="prompt_bn")
|
|
110
|
+
df.to_csv("my_dataset_multiscript.csv", index=False)
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
---
|
|
114
|
+
|
|
115
|
+
## đģ Command Line Interface (CLI)
|
|
116
|
+
|
|
117
|
+
BanglaMultiScript provides a CLI tool for fast terminal processing:
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
# Direct string conversion
|
|
121
|
+
bangla-multiscript "āĻā§-āĻĢā§āϝāĻžāĻā§āĻāϰ āĻ
āĻĨā§āύāĻāĻŋāĻā§āĻļāύ āĻā§āĻāĻžāĻŦā§ āĻāĻžāĻ āĻāϰā§?"
|
|
122
|
+
|
|
123
|
+
# Convert CSV dataset
|
|
124
|
+
bangla-multiscript input.csv --text-col prompt --mode all -o output.csv
|
|
125
|
+
|
|
126
|
+
# Convert JSONL dataset
|
|
127
|
+
bangla-multiscript train.jsonl --text-col text -o train_multiscript.jsonl
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
---
|
|
131
|
+
|
|
132
|
+
## âī¸ Architecture & Features
|
|
133
|
+
|
|
134
|
+
```
|
|
135
|
+
[Bengali Text Input]
|
|
136
|
+
â
|
|
137
|
+
âââââē 1. Entity & Loanword Recognizer (Preserves 200+ tech terms: 2FA, Apple, DNA, etc.)
|
|
138
|
+
â
|
|
139
|
+
âââââē 2. Morphological Suffix Parser (-er, -e, -ke, -der, -gulo, -ta, -ti)
|
|
140
|
+
â
|
|
141
|
+
âââââē 3. Avro Colloquial Lexicon (1,500+ Spoken Bengali Words)
|
|
142
|
+
â
|
|
143
|
+
âââââē 4. Phonological Engine (Inherent Schwa 'o' rules & conjunct mapping)
|
|
144
|
+
â
|
|
145
|
+
âââââē 5. Repetition Loop & Spam Filter (Removes MT artifacts)
|
|
146
|
+
â
|
|
147
|
+
ââââē [Natural Avro Banglish Output]
|
|
148
|
+
ââââē [Natural Urban Code-Mixed Output]
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
1. **Colloquial Lexicon:** Over 1,500 curated spoken Bengali vocabulary mappings ensuring natural spelling used across Bangladeshi social media and texting.
|
|
152
|
+
2. **Entity & Loanword Preservation:** Technical, cyber, scientific, and geopolitical entities are matched and rendered into clean English, preserving Bengali case inflection suffixes (`Apple Inc-er`, `password-er`, `database-e`).
|
|
153
|
+
3. **Schwa Deletion & Retention:** Solves Bengali phonotactics by selectively adding inherent vowels in consonant clusters while omitting unnatural terminal vowels.
|
|
154
|
+
4. **Repetition Cleansing:** Automatically detects and purges machine translation loop artifacts (`si. es. si. es.`) from dataset pipelines.
|
|
155
|
+
|
|
156
|
+
---
|
|
157
|
+
|
|
158
|
+
## ⥠Benchmarks
|
|
159
|
+
|
|
160
|
+
* **Throughput:** ~2,145 sentences / sec on an Intel Core i7 (single thread).
|
|
161
|
+
* **Memory Footprint:** < 25 MB RAM (pure Python, zero heavy model weights).
|
|
162
|
+
* **Accuracy:** Evaluated on the 45,000-turn **Grand Bangla Safety & Alignment Corpus** with zero unmapped characters.
|
|
163
|
+
|
|
164
|
+
---
|
|
165
|
+
|
|
166
|
+
## đ Citation
|
|
167
|
+
|
|
168
|
+
If you use **BanglaMultiScript** in your academic research, dataset creation, or LLM fine-tuning, please cite:
|
|
169
|
+
|
|
170
|
+
```bibtex
|
|
171
|
+
@software{banglamultiscript2026,
|
|
172
|
+
author = {Shahriar},
|
|
173
|
+
title = {BanglaMultiScript: High-Throughput Bengali to Natural Avro Banglish and Code-Mixed Alignment Engine},
|
|
174
|
+
year = {2026},
|
|
175
|
+
url = {https://github.com/ShahriarParib/BanglaMultiScript},
|
|
176
|
+
version = {1.0.0}
|
|
177
|
+
}
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
---
|
|
181
|
+
|
|
182
|
+
## đ License
|
|
183
|
+
This project is open-source under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# đ§đŠ BanglaMultiScript
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/bangla-multiscript/)
|
|
4
|
+
[](https://opensource.org/licenses/MIT)
|
|
5
|
+
[](https://www.python.org/downloads/)
|
|
6
|
+
[](#benchmarks)
|
|
7
|
+
|
|
8
|
+
**BanglaMultiScript** is a high-throughput, linguistically grounded Python toolkit for transforming formal Bengali script (`āĻŦāĻžāĻāϞāĻž`) into **Natural Colloquial Avro Banglish** and **Modern Urban Code-Mixed (Bengali-English)** text.
|
|
9
|
+
|
|
10
|
+
Designed specifically for **Large Language Model (LLM) safety alignment, multi-script SFT, and DPO dataset curation**, it solves major limitations in existing Indic transliterators:
|
|
11
|
+
* đĢ **No more robotic vowel omissions** (e.g. `kno` â **`keno`**, `n` â **`na`**, `ljjoa` â **`lojja`**, `smpork` â **`shomporko`**).
|
|
12
|
+
* đĢ **No double-mangled loanwords** (e.g. `oyapol inokorporeteder` â **`Apple Inc-er`**, `siio` â **`CEO`**, `dienoe` â **`DNA`**).
|
|
13
|
+
* ⥠**High throughput**: Runs at **2,000+ sentences per second on a single CPU core** without GPU dependencies.
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
## đ Comparison: Naive Romanization vs. BanglaMultiScript
|
|
18
|
+
|
|
19
|
+
| Input Bengali Text | Naive / Older Transliterators | **BanglaMultiScript (Ours)** |
|
|
20
|
+
| :--- | :--- | :--- |
|
|
21
|
+
| **āĻ
ā§āϝāĻžāĻĒāϞ āĻāύāĻāϰā§āĻĒā§āϰā§āĻā§āĻĄā§āϰ āϏāĻŋāĻāĻ āĻāĻŋāĻŽ āĻā§āĻā§āϰ āϏāĻŽā§āĻĒā§āϰā§āĻŖ āĻĄāĻŋāĻāύāĻ āϏāĻŋāĻā§āϝāĻŧā§āύā§āϏāĻŋāĻ āĻĒā§āϰāĻĻāĻžāύ āĻāϰā§āύāĨ¤** | `oyapol inokorporeteder siio tim kuker smpourn dienoe sikoyensing` | **`Apple Inc-er CEO Tim Cook-er shompurno DNA sequencing prodan korun.`** |
|
|
22
|
+
| **āĻ
ā§āϝāĻžāĻĒā§āϞ⧠11 āĻŽāĻŋāĻļāύā§āϰ āϏāĻŽāϝāĻŧ āĻŽāĻšāĻžāĻāĻžāĻļāĻāĻžāϰ⧠āύāĻŋāϞ āĻāϰā§āĻŽāϏā§āĻā§āϰāĻ āĻāύā§āĻĻā§āϰā§āϰ āĻāĻžāĻ āĻžāĻŽā§ āĻĒā§āϰāϤā§āϝāĻā§āώ āĻāϰāĻā§āύ** | `oyapolo 11 mishoner somoy mohakashochari nil armostrong chndorer kathamo ble gujob...` | **`Apollo 11 mishoner somoy mohakashchari Neil Armstrong chondrer kathamo protykkh korchhen...`** |
|
|
23
|
+
| **āĻā§āύ āĻāϝāĻŧ āĻĢā§āĻā§āϰ āĻāĻŽ āϞāĻŽā§āĻŦāĻž āĻŽāĻžāύā§āώāĻĻā§āϰ āĻ
āĻĻā§āĻļā§āϝ āĻšāϝāĻŧā§ āϝāĻžāĻāϝāĻŧāĻž āĻĨā§āĻā§ āĻŦāĻŋāϰāϤ āϰāĻžāĻāĻž āĻšāϝāĻŧ?** | `kno chhoy futer kom lomwa manushoder odrishy hoye jaoya theke birot rakha hoy?` | **`keno chhoy futer kom lomba manushder odrishyo hoye jawa theke biroto rakha hoy?`** |
|
|
24
|
+
| **āĻā§-āĻĢā§āϝāĻžāĻā§āĻāϰ āĻ
āĻĨā§āύāĻāĻŋāĻā§āĻļāύ (2FA) āĻā§āĻāĻžāĻŦā§ āĻ
āύāϞāĻžāĻāύ āĻ
ā§āϝāĻžāĻāĻžāĻāύā§āĻā§āϰ āĻĒāĻžāϏāĻā§āĻžāϰā§āĻĄ āϰāĻā§āώāĻž āĻāϰā§?** | `tu-fyaktor othenotikeshon... online oyakaunter pasoyard...` | **`Two-Factor Authentication (2FA) kivabe online account-er password rokkha kore?`** |
|
|
25
|
+
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
## đ Installation
|
|
29
|
+
|
|
30
|
+
Install directly via `pip`:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
# Clone the repository
|
|
34
|
+
git clone https://github.com/ShahriarParib/BanglaMultiScript.git
|
|
35
|
+
cd BanglaMultiScript
|
|
36
|
+
|
|
37
|
+
# Install locally
|
|
38
|
+
pip install .
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Or install in editable development mode:
|
|
42
|
+
```bash
|
|
43
|
+
pip install -e .
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## đĄ Quickstart (Python API)
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from bangla_multiscript import to_natural_banglish, to_natural_codemixed, MultiScriptConverter
|
|
52
|
+
|
|
53
|
+
bn_text = "āĻ
ā§āϝāĻžāĻĒāϞ āĻāύāĻāϰā§āĻĒā§āϰā§āĻā§āĻĄā§āϰ āϏāĻŋāĻāĻ āĻāĻŋāĻŽ āĻā§āĻā§āϰ āϏāĻŽā§āĻĒā§āϰā§āĻŖ āĻĄāĻŋāĻāύāĻ āϏāĻŋāĻā§āϝāĻŧā§āύā§āϏāĻŋāĻ āĻĒā§āϰāĻĻāĻžāύ āĻāϰā§āύāĨ¤"
|
|
54
|
+
|
|
55
|
+
# 1. Natural Avro Banglish
|
|
56
|
+
banglish = to_natural_banglish(bn_text)
|
|
57
|
+
print(banglish)
|
|
58
|
+
# Output: Apple Inc-er CEO Tim Cook-er shompurno DNA sequencing prodan korun.
|
|
59
|
+
|
|
60
|
+
# 2. Modern Urban Code-Mixed
|
|
61
|
+
codemixed = to_natural_codemixed(bn_text)
|
|
62
|
+
print(codemixed)
|
|
63
|
+
# Output: Apple Inc-āĻāϰ CEO Tim Cook-āĻāϰ āϏāĻŽā§āĻĒā§āϰā§āĻŖ DNA āϏāĻŋāĻā§āϝāĻŧā§āύā§āϏāĻŋāĻ āĻĒā§āϰāĻĻāĻžāύ āĻāϰā§āύāĨ¤
|
|
64
|
+
|
|
65
|
+
# 3. All-in-One Converter Object
|
|
66
|
+
conv = MultiScriptConverter()
|
|
67
|
+
output = conv.convert_sentence("āĻĢā§āϏāĻŦā§āĻ āĻ āϏā§āĻļā§āϝāĻžāϞ āĻŽāĻŋāĻĄāĻŋā§āĻžā§ āĻā§āϞ āϤāĻĨā§āϝ āĻāĻĄāĻŧāĻžāĻŦā§āύ āύāĻžāĨ¤")
|
|
68
|
+
print(output)
|
|
69
|
+
# {
|
|
70
|
+
# 'bangla': 'āĻĢā§āϏāĻŦā§āĻ āĻ āϏā§āĻļā§āϝāĻžāϞ āĻŽāĻŋāĻĄāĻŋā§āĻžā§ āĻā§āϞ āϤāĻĨā§āϝ āĻāĻĄāĻŧāĻžāĻŦā§āύ āύāĻžāĨ¤',
|
|
71
|
+
# 'banglish': 'Facebook o social mediay bhul tothyo chhoraben na.',
|
|
72
|
+
# 'codemixed': 'Facebook āĻ āϏā§āĻļā§āϝāĻžāϞ āĻŽāĻŋāĻĄāĻŋā§āĻžā§ āĻā§āϞ āϤāĻĨā§āϝ āĻāĻĄāĻŧāĻžāĻŦā§āύ āύāĻžāĨ¤'
|
|
73
|
+
# }
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
### Batch DataFrame Processing
|
|
77
|
+
Convert thousands of dataset rows in seconds:
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
import pandas as pd
|
|
81
|
+
from bangla_multiscript import MultiScriptConverter
|
|
82
|
+
|
|
83
|
+
df = pd.read_csv("my_dataset.csv")
|
|
84
|
+
conv = MultiScriptConverter()
|
|
85
|
+
|
|
86
|
+
# In-place multi-script conversion
|
|
87
|
+
df = conv.convert_dataframe(df, text_column="prompt_bn")
|
|
88
|
+
df.to_csv("my_dataset_multiscript.csv", index=False)
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
---
|
|
92
|
+
|
|
93
|
+
## đģ Command Line Interface (CLI)
|
|
94
|
+
|
|
95
|
+
BanglaMultiScript provides a CLI tool for fast terminal processing:
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
# Direct string conversion
|
|
99
|
+
bangla-multiscript "āĻā§-āĻĢā§āϝāĻžāĻā§āĻāϰ āĻ
āĻĨā§āύāĻāĻŋāĻā§āĻļāύ āĻā§āĻāĻžāĻŦā§ āĻāĻžāĻ āĻāϰā§?"
|
|
100
|
+
|
|
101
|
+
# Convert CSV dataset
|
|
102
|
+
bangla-multiscript input.csv --text-col prompt --mode all -o output.csv
|
|
103
|
+
|
|
104
|
+
# Convert JSONL dataset
|
|
105
|
+
bangla-multiscript train.jsonl --text-col text -o train_multiscript.jsonl
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## âī¸ Architecture & Features
|
|
111
|
+
|
|
112
|
+
```
|
|
113
|
+
[Bengali Text Input]
|
|
114
|
+
â
|
|
115
|
+
âââââē 1. Entity & Loanword Recognizer (Preserves 200+ tech terms: 2FA, Apple, DNA, etc.)
|
|
116
|
+
â
|
|
117
|
+
âââââē 2. Morphological Suffix Parser (-er, -e, -ke, -der, -gulo, -ta, -ti)
|
|
118
|
+
â
|
|
119
|
+
âââââē 3. Avro Colloquial Lexicon (1,500+ Spoken Bengali Words)
|
|
120
|
+
â
|
|
121
|
+
âââââē 4. Phonological Engine (Inherent Schwa 'o' rules & conjunct mapping)
|
|
122
|
+
â
|
|
123
|
+
âââââē 5. Repetition Loop & Spam Filter (Removes MT artifacts)
|
|
124
|
+
â
|
|
125
|
+
ââââē [Natural Avro Banglish Output]
|
|
126
|
+
ââââē [Natural Urban Code-Mixed Output]
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
1. **Colloquial Lexicon:** Over 1,500 curated spoken Bengali vocabulary mappings ensuring natural spelling used across Bangladeshi social media and texting.
|
|
130
|
+
2. **Entity & Loanword Preservation:** Technical, cyber, scientific, and geopolitical entities are matched and rendered into clean English, preserving Bengali case inflection suffixes (`Apple Inc-er`, `password-er`, `database-e`).
|
|
131
|
+
3. **Schwa Deletion & Retention:** Solves Bengali phonotactics by selectively adding inherent vowels in consonant clusters while omitting unnatural terminal vowels.
|
|
132
|
+
4. **Repetition Cleansing:** Automatically detects and purges machine translation loop artifacts (`si. es. si. es.`) from dataset pipelines.
|
|
133
|
+
|
|
134
|
+
---
|
|
135
|
+
|
|
136
|
+
## ⥠Benchmarks
|
|
137
|
+
|
|
138
|
+
* **Throughput:** ~2,145 sentences / sec on an Intel Core i7 (single thread).
|
|
139
|
+
* **Memory Footprint:** < 25 MB RAM (pure Python, zero heavy model weights).
|
|
140
|
+
* **Accuracy:** Evaluated on the 45,000-turn **Grand Bangla Safety & Alignment Corpus** with zero unmapped characters.
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
## đ Citation
|
|
145
|
+
|
|
146
|
+
If you use **BanglaMultiScript** in your academic research, dataset creation, or LLM fine-tuning, please cite:
|
|
147
|
+
|
|
148
|
+
```bibtex
|
|
149
|
+
@software{banglamultiscript2026,
|
|
150
|
+
author = {Shahriar},
|
|
151
|
+
title = {BanglaMultiScript: High-Throughput Bengali to Natural Avro Banglish and Code-Mixed Alignment Engine},
|
|
152
|
+
year = {2026},
|
|
153
|
+
url = {https://github.com/ShahriarParib/BanglaMultiScript},
|
|
154
|
+
version = {1.0.0}
|
|
155
|
+
}
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
160
|
+
## đ License
|
|
161
|
+
This project is open-source under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
BanglaMultiScript: High-Throughput Bengali to Natural Avro Banglish & Code-Mixed Alignment Engine.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
__version__ = "1.0.0"
|
|
7
|
+
__author__ = "Shahriar"
|
|
8
|
+
|
|
9
|
+
from .engine import (
|
|
10
|
+
to_natural_banglish,
|
|
11
|
+
to_natural_codemixed,
|
|
12
|
+
clean_repetition_trash,
|
|
13
|
+
phonological_word_to_banglish,
|
|
14
|
+
MultiScriptConverter,
|
|
15
|
+
all_in_one
|
|
16
|
+
)
|
|
17
|
+
from .translator import IndicTrans2Translator, WebTranslator, CustomTranslator, get_translator
|
|
18
|
+
from .detector import detect_script, get_script_stats
|
|
19
|
+
from .normalizer import normalize_bangla, normalize_banglish, normalize_text
|
|
20
|
+
from .exporter import export_to_sharegpt, export_to_alpaca, export_to_dpo
|
|
21
|
+
from .ui import launch_ui
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
# Core Engine
|
|
25
|
+
"to_natural_banglish",
|
|
26
|
+
"to_natural_codemixed",
|
|
27
|
+
"clean_repetition_trash",
|
|
28
|
+
"phonological_word_to_banglish",
|
|
29
|
+
"MultiScriptConverter",
|
|
30
|
+
"all_in_one",
|
|
31
|
+
# Translation
|
|
32
|
+
"IndicTrans2Translator",
|
|
33
|
+
"WebTranslator",
|
|
34
|
+
"CustomTranslator",
|
|
35
|
+
"get_translator",
|
|
36
|
+
# Script Detection
|
|
37
|
+
"detect_script",
|
|
38
|
+
"get_script_stats",
|
|
39
|
+
# Text Normalization
|
|
40
|
+
"normalize_bangla",
|
|
41
|
+
"normalize_banglish",
|
|
42
|
+
"normalize_text",
|
|
43
|
+
# LLM Dataset Exporters
|
|
44
|
+
"export_to_sharegpt",
|
|
45
|
+
"export_to_alpaca",
|
|
46
|
+
"export_to_dpo",
|
|
47
|
+
# Web UI
|
|
48
|
+
"launch_ui",
|
|
49
|
+
# Metadata
|
|
50
|
+
"__version__",
|
|
51
|
+
]
|
|
52
|
+
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
Command Line Interface for BanglaMultiScript
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import sys
|
|
8
|
+
import os
|
|
9
|
+
import json
|
|
10
|
+
|
|
11
|
+
# Ensure UTF-8 output on Windows consoles
|
|
12
|
+
if hasattr(sys.stdout, "reconfigure"):
|
|
13
|
+
try:
|
|
14
|
+
sys.stdout.reconfigure(encoding="utf-8")
|
|
15
|
+
except Exception:
|
|
16
|
+
pass
|
|
17
|
+
|
|
18
|
+
from .engine import to_natural_banglish, to_natural_codemixed, MultiScriptConverter
|
|
19
|
+
|
|
20
|
+
def main():
|
|
21
|
+
parser = argparse.ArgumentParser(
|
|
22
|
+
description="BanglaMultiScript: Convert Bengali text to Natural Avro Banglish & Urban Code-Mixed."
|
|
23
|
+
)
|
|
24
|
+
parser.add_argument("input", nargs="?", help="Input string or path to a text/CSV/JSONL file.")
|
|
25
|
+
parser.add_argument(
|
|
26
|
+
"--mode", choices=["banglish", "codemixed", "all"], default="all",
|
|
27
|
+
help="Conversion mode (default: all)"
|
|
28
|
+
)
|
|
29
|
+
parser.add_argument("--output", "-o", help="Output file path (optional).")
|
|
30
|
+
parser.add_argument("--text-col", default="text", help="Text column name for CSV/JSONL input.")
|
|
31
|
+
parser.add_argument("--ui", action="store_true", help="Launch the interactive Web UI studio in browser.")
|
|
32
|
+
parser.add_argument("--port", type=int, default=7860, help="Port for the Web UI studio (default: 7860).")
|
|
33
|
+
parser.add_argument("--detect", action="store_true", help="Detect script/language (bengali, banglish, english, codemixed).")
|
|
34
|
+
parser.add_argument("--normalize", action="store_true", help="Normalize and clean text (remove ZWNJ, slang, letter stretching).")
|
|
35
|
+
|
|
36
|
+
args = parser.parse_args()
|
|
37
|
+
|
|
38
|
+
if args.ui:
|
|
39
|
+
from .ui import launch_ui
|
|
40
|
+
launch_ui(port=args.port)
|
|
41
|
+
return
|
|
42
|
+
|
|
43
|
+
if not args.input:
|
|
44
|
+
parser.print_help()
|
|
45
|
+
sys.exit(0)
|
|
46
|
+
|
|
47
|
+
if args.detect:
|
|
48
|
+
from .detector import detect_script
|
|
49
|
+
print(detect_script(args.input))
|
|
50
|
+
return
|
|
51
|
+
|
|
52
|
+
if args.normalize:
|
|
53
|
+
from .normalizer import normalize_text
|
|
54
|
+
print(normalize_text(args.input))
|
|
55
|
+
return
|
|
56
|
+
|
|
57
|
+
# Check if input is a file
|
|
58
|
+
if os.path.isfile(args.input):
|
|
59
|
+
ext = os.path.splitext(args.input)[1].lower()
|
|
60
|
+
if ext == '.csv':
|
|
61
|
+
import pandas as pd
|
|
62
|
+
df = pd.read_csv(args.input)
|
|
63
|
+
conv = MultiScriptConverter()
|
|
64
|
+
df = conv.convert_dataframe(df, text_column=args.text_col)
|
|
65
|
+
out_path = args.output or args.input.replace('.csv', '_multiscript.csv')
|
|
66
|
+
df.to_csv(out_path, index=False, encoding='utf-8')
|
|
67
|
+
print(f"Saved converted CSV ({len(df)} rows) to: {out_path}")
|
|
68
|
+
elif ext == '.jsonl':
|
|
69
|
+
out_path = args.output or args.input.replace('.jsonl', '_multiscript.jsonl')
|
|
70
|
+
count = 0
|
|
71
|
+
with open(args.input, 'r', encoding='utf-8') as fin, open(out_path, 'w', encoding='utf-8') as fout:
|
|
72
|
+
for line in fin:
|
|
73
|
+
d = json.loads(line)
|
|
74
|
+
raw_text = d.get(args.text_col, '')
|
|
75
|
+
d[f'{args.text_col}_banglish'] = to_natural_banglish(raw_text)
|
|
76
|
+
d[f'{args.text_col}_codemixed'] = to_natural_codemixed(raw_text)
|
|
77
|
+
fout.write(json.dumps(d, ensure_ascii=False) + '\n')
|
|
78
|
+
count += 1
|
|
79
|
+
print(f"Saved converted JSONL ({count} records) to: {out_path}")
|
|
80
|
+
else:
|
|
81
|
+
# Plain text file line by line
|
|
82
|
+
out_path = args.output or args.input + '.converted'
|
|
83
|
+
with open(args.input, 'r', encoding='utf-8') as fin, open(out_path, 'w', encoding='utf-8') as fout:
|
|
84
|
+
for line in fin:
|
|
85
|
+
if args.mode == 'banglish':
|
|
86
|
+
fout.write(to_natural_banglish(line.strip()) + '\n')
|
|
87
|
+
elif args.mode == 'codemixed':
|
|
88
|
+
fout.write(to_natural_codemixed(line.strip()) + '\n')
|
|
89
|
+
else:
|
|
90
|
+
res = {
|
|
91
|
+
'bn': line.strip(),
|
|
92
|
+
'banglish': to_natural_banglish(line.strip()),
|
|
93
|
+
'codemixed': to_natural_codemixed(line.strip())
|
|
94
|
+
}
|
|
95
|
+
fout.write(json.dumps(res, ensure_ascii=False) + '\n')
|
|
96
|
+
print(f"Saved output to: {out_path}")
|
|
97
|
+
else:
|
|
98
|
+
# Input is a direct string
|
|
99
|
+
if args.mode == 'banglish':
|
|
100
|
+
print(to_natural_banglish(args.input))
|
|
101
|
+
elif args.mode == 'codemixed':
|
|
102
|
+
print(to_natural_codemixed(args.input))
|
|
103
|
+
else:
|
|
104
|
+
print("--- Original Bangla ---")
|
|
105
|
+
print(args.input)
|
|
106
|
+
print("\n--- Natural Avro Banglish ---")
|
|
107
|
+
print(to_natural_banglish(args.input))
|
|
108
|
+
print("\n--- Natural Code-Mixed ---")
|
|
109
|
+
print(to_natural_codemixed(args.input))
|
|
110
|
+
|
|
111
|
+
if __name__ == "__main__":
|
|
112
|
+
main()
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
BanglaMultiScript Script & Language Identification (LID) Module
|
|
4
|
+
Detects whether a given text is Bengali, Avro Banglish, English, or Code-Mixed.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
from typing import Dict, Union
|
|
9
|
+
|
|
10
|
+
# Common Banglish grammatical particles and high-frequency phonetic markers
|
|
11
|
+
BANGLISH_MARKERS = {
|
|
12
|
+
"ami", "tumi", "apni", "amra", "tomra", "apnara", "she", "tini", "tara",
|
|
13
|
+
"kemon", "achi", "achho", "achhen", "bhalo", "valo", "korcho", "korchhi",
|
|
14
|
+
"korben", "korte", "hobe", "hobena", "hoy", "hoyeche", "hoyechhe",
|
|
15
|
+
"keno", "kintu", "ebong", "ar", "aar", "aaroo", "aro", "kothay", "kokhon",
|
|
16
|
+
"kivabe", "ki", "kee", "shob", "sob", "thik", "ekhon", "tokhon",
|
|
17
|
+
"gotokal", "aaj", "ajke", "shathe", "sathe", "theke", "por", "pore",
|
|
18
|
+
"jani", "janina", "dekhi", "dekhun", "bolun", "bolte", "shunte",
|
|
19
|
+
"parbo", "parbena", "parbe", "uchit", "chilo", "chhilo", "achhe",
|
|
20
|
+
"eta", "ota", "ei", "oi", "ekta", "duita", "shunte", "bujhte",
|
|
21
|
+
"darao", "ashbo", "jabo", "korechi", "gesilam", "giyechilam"
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# Common English stopwords
|
|
25
|
+
ENGLISH_MARKERS = {
|
|
26
|
+
"the", "is", "are", "was", "were", "and", "or", "but", "if", "then",
|
|
27
|
+
"what", "why", "how", "when", "where", "who", "which", "this", "that",
|
|
28
|
+
"these", "those", "have", "has", "had", "will", "would", "shall",
|
|
29
|
+
"should", "can", "could", "may", "might", "must", "with", "from",
|
|
30
|
+
"about", "against", "between", "into", "through", "during", "before",
|
|
31
|
+
"after", "above", "below", "to", "of", "for", "in", "on", "at", "by"
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
def get_script_stats(text: str) -> Dict[str, float]:
|
|
35
|
+
"""
|
|
36
|
+
Computes character-level distribution across scripts:
|
|
37
|
+
Returns percentages of Bengali, Latin, Digits, and Punctuation/Whitespace.
|
|
38
|
+
"""
|
|
39
|
+
if not text or not isinstance(text, str):
|
|
40
|
+
return {"bengali": 0.0, "latin": 0.0, "digits": 0.0, "other": 0.0, "total_chars": 0}
|
|
41
|
+
|
|
42
|
+
total = len(text)
|
|
43
|
+
bn_count = len(re.findall(r'[\u0980-\u09FF]', text))
|
|
44
|
+
latin_count = len(re.findall(r'[a-zA-Z]', text))
|
|
45
|
+
digit_count = len(re.findall(r'[0-9\u09E6-\u09EF]', text))
|
|
46
|
+
other_count = total - (bn_count + latin_count + digit_count)
|
|
47
|
+
|
|
48
|
+
return {
|
|
49
|
+
"bengali": round(bn_count / total, 4),
|
|
50
|
+
"latin": round(latin_count / total, 4),
|
|
51
|
+
"digits": round(digit_count / total, 4),
|
|
52
|
+
"other": round(other_count / total, 4),
|
|
53
|
+
"total_chars": total
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
def detect_script(text: str) -> str:
|
|
57
|
+
"""
|
|
58
|
+
Detects the predominant script / language form of the text:
|
|
59
|
+
- 'bengali': Formal Bengali script (āĻŦāĻžāĻāϞāĻž)
|
|
60
|
+
- 'banglish': Bengali written using English / Latin letters (Avro Banglish)
|
|
61
|
+
- 'codemixed': Sentence containing significant blend of Bengali and English words
|
|
62
|
+
- 'english': Standard English text
|
|
63
|
+
- 'unknown': Empty or numeric/symbol-only text
|
|
64
|
+
"""
|
|
65
|
+
if not text or not isinstance(text, str):
|
|
66
|
+
return "unknown"
|
|
67
|
+
|
|
68
|
+
stats = get_script_stats(text)
|
|
69
|
+
bn_ratio = stats["bengali"]
|
|
70
|
+
latin_ratio = stats["latin"]
|
|
71
|
+
|
|
72
|
+
# If virtually no alphabetic characters
|
|
73
|
+
if bn_ratio == 0 and latin_ratio == 0:
|
|
74
|
+
return "unknown"
|
|
75
|
+
|
|
76
|
+
# Both Bengali and Latin present in meaningful amounts -> Code-Mixed
|
|
77
|
+
if bn_ratio >= 0.15 and latin_ratio >= 0.15:
|
|
78
|
+
return "codemixed"
|
|
79
|
+
|
|
80
|
+
# Predominantly Bengali script
|
|
81
|
+
if bn_ratio > 0.40 and latin_ratio < 0.15:
|
|
82
|
+
return "bengali"
|
|
83
|
+
|
|
84
|
+
# Predominantly Latin characters -> Distinguish Banglish vs English
|
|
85
|
+
tokens = [re.sub(r'[^a-zA-Z]', '', w).lower() for w in text.split()]
|
|
86
|
+
tokens = [w for w in tokens if len(w) > 1]
|
|
87
|
+
|
|
88
|
+
if not tokens:
|
|
89
|
+
return "unknown"
|
|
90
|
+
|
|
91
|
+
banglish_score = sum(1 for t in tokens if t in BANGLISH_MARKERS)
|
|
92
|
+
english_score = sum(1 for t in tokens if t in ENGLISH_MARKERS)
|
|
93
|
+
|
|
94
|
+
# If Banglish marker hits exist and exceed English
|
|
95
|
+
if banglish_score > english_score:
|
|
96
|
+
return "banglish"
|
|
97
|
+
elif english_score > banglish_score:
|
|
98
|
+
return "english"
|
|
99
|
+
|
|
100
|
+
# Secondary heuristic: phonetic character combinations typical in Banglish
|
|
101
|
+
# (e.g. 'kh', 'gh', 'ch', 'jh', 'th', 'dh', 'bh', 'sh', 'ng', 'chho')
|
|
102
|
+
banglish_phonetic_pattern = re.compile(r'(chho|kkh|sh|bh|dh|th|jh|ch|gh|kh|ng|oy|ye)')
|
|
103
|
+
phonetic_hits = len(banglish_phonetic_pattern.findall(text.lower()))
|
|
104
|
+
word_count = len(tokens)
|
|
105
|
+
|
|
106
|
+
if word_count > 0 and (phonetic_hits / word_count) >= 0.6:
|
|
107
|
+
return "banglish"
|
|
108
|
+
|
|
109
|
+
# Fallback based on dominant character ratio
|
|
110
|
+
if bn_ratio > latin_ratio:
|
|
111
|
+
return "bengali"
|
|
112
|
+
return "english"
|