bangla-multiscript 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bangla_multiscript/__init__.py +52 -0
- bangla_multiscript/cli.py +112 -0
- bangla_multiscript/detector.py +112 -0
- bangla_multiscript/engine.py +265 -0
- bangla_multiscript/exporter.py +138 -0
- bangla_multiscript/lexicon.py +601 -0
- bangla_multiscript/normalizer.py +110 -0
- bangla_multiscript/translator.py +115 -0
- bangla_multiscript/ui.py +181 -0
- bangla_multiscript/web/index.html +1084 -0
- bangla_multiscript-1.0.0.dist-info/METADATA +183 -0
- bangla_multiscript-1.0.0.dist-info/RECORD +16 -0
- bangla_multiscript-1.0.0.dist-info/WHEEL +5 -0
- bangla_multiscript-1.0.0.dist-info/entry_points.txt +2 -0
- bangla_multiscript-1.0.0.dist-info/licenses/LICENSE +22 -0
- bangla_multiscript-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
BanglaMultiScript: High-Throughput Bengali to Natural Avro Banglish & Code-Mixed Alignment Engine.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
__version__ = "1.0.0"
|
|
7
|
+
__author__ = "Shahriar"
|
|
8
|
+
|
|
9
|
+
from .engine import (
|
|
10
|
+
to_natural_banglish,
|
|
11
|
+
to_natural_codemixed,
|
|
12
|
+
clean_repetition_trash,
|
|
13
|
+
phonological_word_to_banglish,
|
|
14
|
+
MultiScriptConverter,
|
|
15
|
+
all_in_one
|
|
16
|
+
)
|
|
17
|
+
from .translator import IndicTrans2Translator, WebTranslator, CustomTranslator, get_translator
|
|
18
|
+
from .detector import detect_script, get_script_stats
|
|
19
|
+
from .normalizer import normalize_bangla, normalize_banglish, normalize_text
|
|
20
|
+
from .exporter import export_to_sharegpt, export_to_alpaca, export_to_dpo
|
|
21
|
+
from .ui import launch_ui
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
# Core Engine
|
|
25
|
+
"to_natural_banglish",
|
|
26
|
+
"to_natural_codemixed",
|
|
27
|
+
"clean_repetition_trash",
|
|
28
|
+
"phonological_word_to_banglish",
|
|
29
|
+
"MultiScriptConverter",
|
|
30
|
+
"all_in_one",
|
|
31
|
+
# Translation
|
|
32
|
+
"IndicTrans2Translator",
|
|
33
|
+
"WebTranslator",
|
|
34
|
+
"CustomTranslator",
|
|
35
|
+
"get_translator",
|
|
36
|
+
# Script Detection
|
|
37
|
+
"detect_script",
|
|
38
|
+
"get_script_stats",
|
|
39
|
+
# Text Normalization
|
|
40
|
+
"normalize_bangla",
|
|
41
|
+
"normalize_banglish",
|
|
42
|
+
"normalize_text",
|
|
43
|
+
# LLM Dataset Exporters
|
|
44
|
+
"export_to_sharegpt",
|
|
45
|
+
"export_to_alpaca",
|
|
46
|
+
"export_to_dpo",
|
|
47
|
+
# Web UI
|
|
48
|
+
"launch_ui",
|
|
49
|
+
# Metadata
|
|
50
|
+
"__version__",
|
|
51
|
+
]
|
|
52
|
+
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
Command Line Interface for BanglaMultiScript
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import sys
|
|
8
|
+
import os
|
|
9
|
+
import json
|
|
10
|
+
|
|
11
|
+
# Ensure UTF-8 output on Windows consoles
|
|
12
|
+
if hasattr(sys.stdout, "reconfigure"):
|
|
13
|
+
try:
|
|
14
|
+
sys.stdout.reconfigure(encoding="utf-8")
|
|
15
|
+
except Exception:
|
|
16
|
+
pass
|
|
17
|
+
|
|
18
|
+
from .engine import to_natural_banglish, to_natural_codemixed, MultiScriptConverter
|
|
19
|
+
|
|
20
|
+
def main():
|
|
21
|
+
parser = argparse.ArgumentParser(
|
|
22
|
+
description="BanglaMultiScript: Convert Bengali text to Natural Avro Banglish & Urban Code-Mixed."
|
|
23
|
+
)
|
|
24
|
+
parser.add_argument("input", nargs="?", help="Input string or path to a text/CSV/JSONL file.")
|
|
25
|
+
parser.add_argument(
|
|
26
|
+
"--mode", choices=["banglish", "codemixed", "all"], default="all",
|
|
27
|
+
help="Conversion mode (default: all)"
|
|
28
|
+
)
|
|
29
|
+
parser.add_argument("--output", "-o", help="Output file path (optional).")
|
|
30
|
+
parser.add_argument("--text-col", default="text", help="Text column name for CSV/JSONL input.")
|
|
31
|
+
parser.add_argument("--ui", action="store_true", help="Launch the interactive Web UI studio in browser.")
|
|
32
|
+
parser.add_argument("--port", type=int, default=7860, help="Port for the Web UI studio (default: 7860).")
|
|
33
|
+
parser.add_argument("--detect", action="store_true", help="Detect script/language (bengali, banglish, english, codemixed).")
|
|
34
|
+
parser.add_argument("--normalize", action="store_true", help="Normalize and clean text (remove ZWNJ, slang, letter stretching).")
|
|
35
|
+
|
|
36
|
+
args = parser.parse_args()
|
|
37
|
+
|
|
38
|
+
if args.ui:
|
|
39
|
+
from .ui import launch_ui
|
|
40
|
+
launch_ui(port=args.port)
|
|
41
|
+
return
|
|
42
|
+
|
|
43
|
+
if not args.input:
|
|
44
|
+
parser.print_help()
|
|
45
|
+
sys.exit(0)
|
|
46
|
+
|
|
47
|
+
if args.detect:
|
|
48
|
+
from .detector import detect_script
|
|
49
|
+
print(detect_script(args.input))
|
|
50
|
+
return
|
|
51
|
+
|
|
52
|
+
if args.normalize:
|
|
53
|
+
from .normalizer import normalize_text
|
|
54
|
+
print(normalize_text(args.input))
|
|
55
|
+
return
|
|
56
|
+
|
|
57
|
+
# Check if input is a file
|
|
58
|
+
if os.path.isfile(args.input):
|
|
59
|
+
ext = os.path.splitext(args.input)[1].lower()
|
|
60
|
+
if ext == '.csv':
|
|
61
|
+
import pandas as pd
|
|
62
|
+
df = pd.read_csv(args.input)
|
|
63
|
+
conv = MultiScriptConverter()
|
|
64
|
+
df = conv.convert_dataframe(df, text_column=args.text_col)
|
|
65
|
+
out_path = args.output or args.input.replace('.csv', '_multiscript.csv')
|
|
66
|
+
df.to_csv(out_path, index=False, encoding='utf-8')
|
|
67
|
+
print(f"Saved converted CSV ({len(df)} rows) to: {out_path}")
|
|
68
|
+
elif ext == '.jsonl':
|
|
69
|
+
out_path = args.output or args.input.replace('.jsonl', '_multiscript.jsonl')
|
|
70
|
+
count = 0
|
|
71
|
+
with open(args.input, 'r', encoding='utf-8') as fin, open(out_path, 'w', encoding='utf-8') as fout:
|
|
72
|
+
for line in fin:
|
|
73
|
+
d = json.loads(line)
|
|
74
|
+
raw_text = d.get(args.text_col, '')
|
|
75
|
+
d[f'{args.text_col}_banglish'] = to_natural_banglish(raw_text)
|
|
76
|
+
d[f'{args.text_col}_codemixed'] = to_natural_codemixed(raw_text)
|
|
77
|
+
fout.write(json.dumps(d, ensure_ascii=False) + '\n')
|
|
78
|
+
count += 1
|
|
79
|
+
print(f"Saved converted JSONL ({count} records) to: {out_path}")
|
|
80
|
+
else:
|
|
81
|
+
# Plain text file line by line
|
|
82
|
+
out_path = args.output or args.input + '.converted'
|
|
83
|
+
with open(args.input, 'r', encoding='utf-8') as fin, open(out_path, 'w', encoding='utf-8') as fout:
|
|
84
|
+
for line in fin:
|
|
85
|
+
if args.mode == 'banglish':
|
|
86
|
+
fout.write(to_natural_banglish(line.strip()) + '\n')
|
|
87
|
+
elif args.mode == 'codemixed':
|
|
88
|
+
fout.write(to_natural_codemixed(line.strip()) + '\n')
|
|
89
|
+
else:
|
|
90
|
+
res = {
|
|
91
|
+
'bn': line.strip(),
|
|
92
|
+
'banglish': to_natural_banglish(line.strip()),
|
|
93
|
+
'codemixed': to_natural_codemixed(line.strip())
|
|
94
|
+
}
|
|
95
|
+
fout.write(json.dumps(res, ensure_ascii=False) + '\n')
|
|
96
|
+
print(f"Saved output to: {out_path}")
|
|
97
|
+
else:
|
|
98
|
+
# Input is a direct string
|
|
99
|
+
if args.mode == 'banglish':
|
|
100
|
+
print(to_natural_banglish(args.input))
|
|
101
|
+
elif args.mode == 'codemixed':
|
|
102
|
+
print(to_natural_codemixed(args.input))
|
|
103
|
+
else:
|
|
104
|
+
print("--- Original Bangla ---")
|
|
105
|
+
print(args.input)
|
|
106
|
+
print("\n--- Natural Avro Banglish ---")
|
|
107
|
+
print(to_natural_banglish(args.input))
|
|
108
|
+
print("\n--- Natural Code-Mixed ---")
|
|
109
|
+
print(to_natural_codemixed(args.input))
|
|
110
|
+
|
|
111
|
+
if __name__ == "__main__":
|
|
112
|
+
main()
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
BanglaMultiScript Script & Language Identification (LID) Module
|
|
4
|
+
Detects whether a given text is Bengali, Avro Banglish, English, or Code-Mixed.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
from typing import Dict, Union
|
|
9
|
+
|
|
10
|
+
# Common Banglish grammatical particles and high-frequency phonetic markers
|
|
11
|
+
BANGLISH_MARKERS = {
|
|
12
|
+
"ami", "tumi", "apni", "amra", "tomra", "apnara", "she", "tini", "tara",
|
|
13
|
+
"kemon", "achi", "achho", "achhen", "bhalo", "valo", "korcho", "korchhi",
|
|
14
|
+
"korben", "korte", "hobe", "hobena", "hoy", "hoyeche", "hoyechhe",
|
|
15
|
+
"keno", "kintu", "ebong", "ar", "aar", "aaroo", "aro", "kothay", "kokhon",
|
|
16
|
+
"kivabe", "ki", "kee", "shob", "sob", "thik", "ekhon", "tokhon",
|
|
17
|
+
"gotokal", "aaj", "ajke", "shathe", "sathe", "theke", "por", "pore",
|
|
18
|
+
"jani", "janina", "dekhi", "dekhun", "bolun", "bolte", "shunte",
|
|
19
|
+
"parbo", "parbena", "parbe", "uchit", "chilo", "chhilo", "achhe",
|
|
20
|
+
"eta", "ota", "ei", "oi", "ekta", "duita", "shunte", "bujhte",
|
|
21
|
+
"darao", "ashbo", "jabo", "korechi", "gesilam", "giyechilam"
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# Common English stopwords
|
|
25
|
+
ENGLISH_MARKERS = {
|
|
26
|
+
"the", "is", "are", "was", "were", "and", "or", "but", "if", "then",
|
|
27
|
+
"what", "why", "how", "when", "where", "who", "which", "this", "that",
|
|
28
|
+
"these", "those", "have", "has", "had", "will", "would", "shall",
|
|
29
|
+
"should", "can", "could", "may", "might", "must", "with", "from",
|
|
30
|
+
"about", "against", "between", "into", "through", "during", "before",
|
|
31
|
+
"after", "above", "below", "to", "of", "for", "in", "on", "at", "by"
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
def get_script_stats(text: str) -> Dict[str, float]:
|
|
35
|
+
"""
|
|
36
|
+
Computes character-level distribution across scripts:
|
|
37
|
+
Returns percentages of Bengali, Latin, Digits, and Punctuation/Whitespace.
|
|
38
|
+
"""
|
|
39
|
+
if not text or not isinstance(text, str):
|
|
40
|
+
return {"bengali": 0.0, "latin": 0.0, "digits": 0.0, "other": 0.0, "total_chars": 0}
|
|
41
|
+
|
|
42
|
+
total = len(text)
|
|
43
|
+
bn_count = len(re.findall(r'[\u0980-\u09FF]', text))
|
|
44
|
+
latin_count = len(re.findall(r'[a-zA-Z]', text))
|
|
45
|
+
digit_count = len(re.findall(r'[0-9\u09E6-\u09EF]', text))
|
|
46
|
+
other_count = total - (bn_count + latin_count + digit_count)
|
|
47
|
+
|
|
48
|
+
return {
|
|
49
|
+
"bengali": round(bn_count / total, 4),
|
|
50
|
+
"latin": round(latin_count / total, 4),
|
|
51
|
+
"digits": round(digit_count / total, 4),
|
|
52
|
+
"other": round(other_count / total, 4),
|
|
53
|
+
"total_chars": total
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
def detect_script(text: str) -> str:
|
|
57
|
+
"""
|
|
58
|
+
Detects the predominant script / language form of the text:
|
|
59
|
+
- 'bengali': Formal Bengali script (বাংলা)
|
|
60
|
+
- 'banglish': Bengali written using English / Latin letters (Avro Banglish)
|
|
61
|
+
- 'codemixed': Sentence containing significant blend of Bengali and English words
|
|
62
|
+
- 'english': Standard English text
|
|
63
|
+
- 'unknown': Empty or numeric/symbol-only text
|
|
64
|
+
"""
|
|
65
|
+
if not text or not isinstance(text, str):
|
|
66
|
+
return "unknown"
|
|
67
|
+
|
|
68
|
+
stats = get_script_stats(text)
|
|
69
|
+
bn_ratio = stats["bengali"]
|
|
70
|
+
latin_ratio = stats["latin"]
|
|
71
|
+
|
|
72
|
+
# If virtually no alphabetic characters
|
|
73
|
+
if bn_ratio == 0 and latin_ratio == 0:
|
|
74
|
+
return "unknown"
|
|
75
|
+
|
|
76
|
+
# Both Bengali and Latin present in meaningful amounts -> Code-Mixed
|
|
77
|
+
if bn_ratio >= 0.15 and latin_ratio >= 0.15:
|
|
78
|
+
return "codemixed"
|
|
79
|
+
|
|
80
|
+
# Predominantly Bengali script
|
|
81
|
+
if bn_ratio > 0.40 and latin_ratio < 0.15:
|
|
82
|
+
return "bengali"
|
|
83
|
+
|
|
84
|
+
# Predominantly Latin characters -> Distinguish Banglish vs English
|
|
85
|
+
tokens = [re.sub(r'[^a-zA-Z]', '', w).lower() for w in text.split()]
|
|
86
|
+
tokens = [w for w in tokens if len(w) > 1]
|
|
87
|
+
|
|
88
|
+
if not tokens:
|
|
89
|
+
return "unknown"
|
|
90
|
+
|
|
91
|
+
banglish_score = sum(1 for t in tokens if t in BANGLISH_MARKERS)
|
|
92
|
+
english_score = sum(1 for t in tokens if t in ENGLISH_MARKERS)
|
|
93
|
+
|
|
94
|
+
# If Banglish marker hits exist and exceed English
|
|
95
|
+
if banglish_score > english_score:
|
|
96
|
+
return "banglish"
|
|
97
|
+
elif english_score > banglish_score:
|
|
98
|
+
return "english"
|
|
99
|
+
|
|
100
|
+
# Secondary heuristic: phonetic character combinations typical in Banglish
|
|
101
|
+
# (e.g. 'kh', 'gh', 'ch', 'jh', 'th', 'dh', 'bh', 'sh', 'ng', 'chho')
|
|
102
|
+
banglish_phonetic_pattern = re.compile(r'(chho|kkh|sh|bh|dh|th|jh|ch|gh|kh|ng|oy|ye)')
|
|
103
|
+
phonetic_hits = len(banglish_phonetic_pattern.findall(text.lower()))
|
|
104
|
+
word_count = len(tokens)
|
|
105
|
+
|
|
106
|
+
if word_count > 0 and (phonetic_hits / word_count) >= 0.6:
|
|
107
|
+
return "banglish"
|
|
108
|
+
|
|
109
|
+
# Fallback based on dominant character ratio
|
|
110
|
+
if bn_ratio > latin_ratio:
|
|
111
|
+
return "bengali"
|
|
112
|
+
return "english"
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
BanglaMultiScript Core Engine
|
|
4
|
+
Provides high-throughput, linguistically sound conversion from standard Bengali
|
|
5
|
+
to Natural Avro Banglish and Natural Code-Mixed text.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
import unicodedata
|
|
10
|
+
from typing import List, Dict, Union, Optional
|
|
11
|
+
|
|
12
|
+
from .lexicon import (
|
|
13
|
+
LOANWORDS_AND_ENTITIES,
|
|
14
|
+
COLLOQUIAL_LEXICON,
|
|
15
|
+
CODEMIXED_REPLACEMENTS,
|
|
16
|
+
VOWELS_INDEP,
|
|
17
|
+
VOWELS_DEP,
|
|
18
|
+
CONSONANTS,
|
|
19
|
+
CONJUNCT_MAP
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
def clean_repetition_trash(text: str) -> str:
|
|
23
|
+
"""
|
|
24
|
+
Cleans up artificial repetition loops and trailing acronym spam
|
|
25
|
+
common in machine translation outputs (e.g. 'si. es. si. es...').
|
|
26
|
+
"""
|
|
27
|
+
if not text or not isinstance(text, str):
|
|
28
|
+
return ""
|
|
29
|
+
# Match identical short token (e.g. 'si. ' or 'es. ') repeated 3+ times
|
|
30
|
+
cleaned = re.sub(r'(?:(\b[a-zA-Z\u0980-\u09FF]{1,4}\.?\s+)\1{2,})', r'\1', text)
|
|
31
|
+
# Match 2-token phrase repeated 3+ times (e.g. 'si. es. si. es. si. es.')
|
|
32
|
+
cleaned = re.sub(r'(?:(\b\w+\.?\s+\w+\.?\s+)\1{2,})', r'\1', cleaned)
|
|
33
|
+
# Match any word repeated consecutively 3+ times
|
|
34
|
+
cleaned = re.sub(r'\b(\w+)(?:\s+\1){2,}\b', r'\1', cleaned)
|
|
35
|
+
# Clean trailing acronym loops (e.g. trailing 'সি. এস. সি. এস.')
|
|
36
|
+
cleaned = re.sub(r'(?:[a-zA-Z\u0980-\u09FF]{1,3}\.\s*){4,}', ' ', cleaned)
|
|
37
|
+
cleaned = re.sub(r'[,.\s]+$', '.', cleaned)
|
|
38
|
+
return re.sub(r'\s+', ' ', cleaned).strip()
|
|
39
|
+
|
|
40
|
+
def phonological_word_to_banglish(word: str) -> str:
|
|
41
|
+
"""
|
|
42
|
+
Transliterates a single Bengali token into natural Avro Banglish
|
|
43
|
+
respecting inherent schwa vowel deletion, conjunct gemination, and common suffixes.
|
|
44
|
+
"""
|
|
45
|
+
clean = re.sub(r'[^\w\u0980-\u09FF]', '', word)
|
|
46
|
+
if not clean:
|
|
47
|
+
return word
|
|
48
|
+
|
|
49
|
+
# 1. Direct Colloquial Lexicon Lookup
|
|
50
|
+
if clean in COLLOQUIAL_LEXICON:
|
|
51
|
+
return COLLOQUIAL_LEXICON[clean]
|
|
52
|
+
|
|
53
|
+
# 2. Suffix Decomposition
|
|
54
|
+
for suffix, rep in [
|
|
55
|
+
('ের', 'er'), ('দের', 'der'), ('গুলো', 'gulo'), ('গুলি', 'guli'),
|
|
56
|
+
('কে', 'ke'), ('তে', 'te'), ('টা', 'ta'), ('টি', 'ti'),
|
|
57
|
+
('ভাবে', 'bhabe'), ('মূলক', 'mulok'), ('কারী', 'kari'), ('কারিদের', 'karider')
|
|
58
|
+
]:
|
|
59
|
+
if clean.endswith(suffix) and len(clean) > len(suffix) + 1:
|
|
60
|
+
stem = clean[:-len(suffix)]
|
|
61
|
+
if stem in COLLOQUIAL_LEXICON:
|
|
62
|
+
return COLLOQUIAL_LEXICON[stem] + ('-' if suffix in ['টা', 'টি'] else '') + rep
|
|
63
|
+
|
|
64
|
+
# 3. Conjunct Resolution
|
|
65
|
+
w = unicodedata.normalize('NFC', clean)
|
|
66
|
+
for cj, rep in CONJUNCT_MAP.items():
|
|
67
|
+
w = w.replace(cj, rep)
|
|
68
|
+
|
|
69
|
+
# 4. Phonetic Traversal
|
|
70
|
+
res = []
|
|
71
|
+
i = 0
|
|
72
|
+
n = len(w)
|
|
73
|
+
while i < n:
|
|
74
|
+
c = w[i]
|
|
75
|
+
if i + 1 < n and w[i+1] == '়':
|
|
76
|
+
c = c + '়'
|
|
77
|
+
i += 1
|
|
78
|
+
|
|
79
|
+
if c in VOWELS_INDEP:
|
|
80
|
+
res.append(VOWELS_INDEP[c])
|
|
81
|
+
elif c in CONSONANTS:
|
|
82
|
+
res.append(CONSONANTS[c])
|
|
83
|
+
if i + 1 < n:
|
|
84
|
+
nxt = w[i+1]
|
|
85
|
+
if nxt == '্':
|
|
86
|
+
i += 1 # Skip hasanta
|
|
87
|
+
elif nxt in VOWELS_DEP:
|
|
88
|
+
res.append(VOWELS_DEP[nxt])
|
|
89
|
+
i += 1
|
|
90
|
+
elif nxt in CONSONANTS or nxt in VOWELS_INDEP:
|
|
91
|
+
# Inherent vowel rule: insert 'o' between consonants
|
|
92
|
+
res.append('o')
|
|
93
|
+
elif c in VOWELS_DEP:
|
|
94
|
+
res.append(VOWELS_DEP[c])
|
|
95
|
+
else:
|
|
96
|
+
res.append(c)
|
|
97
|
+
i += 1
|
|
98
|
+
|
|
99
|
+
s = ''.join(res)
|
|
100
|
+
# Clean double 'o' or unnatural clusters
|
|
101
|
+
s = re.sub(r'oo+', 'o', s)
|
|
102
|
+
s = re.sub(r'yy+', 'y', s)
|
|
103
|
+
return s
|
|
104
|
+
|
|
105
|
+
def to_natural_banglish(text: str) -> str:
|
|
106
|
+
"""
|
|
107
|
+
Converts a standard Bengali sentence into colloquial Avro Banglish.
|
|
108
|
+
Preserves English loanwords, technical nomenclature, and entity names in proper Latin.
|
|
109
|
+
"""
|
|
110
|
+
if not text or not isinstance(text, str):
|
|
111
|
+
return ""
|
|
112
|
+
|
|
113
|
+
# 1. Map known loanwords and entities
|
|
114
|
+
res = text
|
|
115
|
+
for bn_term, eng_term in LOANWORDS_AND_ENTITIES:
|
|
116
|
+
pattern = re.escape(bn_term)
|
|
117
|
+
res = re.sub(pattern, eng_term, res)
|
|
118
|
+
|
|
119
|
+
# 2. Tokenize and convert remaining Bengali script tokens
|
|
120
|
+
tokens = re.split(r'(\s+|[.,!?;:\"\'\(\)\[\]|।])', res)
|
|
121
|
+
out = []
|
|
122
|
+
for t in tokens:
|
|
123
|
+
if not t:
|
|
124
|
+
continue
|
|
125
|
+
if re.search(r'[\u0980-\u09FF]', t):
|
|
126
|
+
out.append(phonological_word_to_banglish(t))
|
|
127
|
+
elif t == '।':
|
|
128
|
+
out.append('.')
|
|
129
|
+
else:
|
|
130
|
+
out.append(t)
|
|
131
|
+
|
|
132
|
+
final_text = ''.join(out)
|
|
133
|
+
final_text = clean_repetition_trash(final_text)
|
|
134
|
+
|
|
135
|
+
# Common post-cleanup heuristics
|
|
136
|
+
final_text = re.sub(r'\bkno\b', 'keno', final_text, flags=re.IGNORECASE)
|
|
137
|
+
final_text = re.sub(r'\bbaddhy\b', 'baddho', final_text, flags=re.IGNORECASE)
|
|
138
|
+
final_text = re.sub(r'\bsmpork\b', 'shomporko', final_text, flags=re.IGNORECASE)
|
|
139
|
+
final_text = re.sub(r'\bsmpourn\b', 'shompurno', final_text, flags=re.IGNORECASE)
|
|
140
|
+
final_text = re.sub(r'\bchndrer\b', 'chondrer', final_text, flags=re.IGNORECASE)
|
|
141
|
+
final_text = re.sub(r'\bchndro\b', 'chondro', final_text, flags=re.IGNORECASE)
|
|
142
|
+
final_text = re.sub(r'\bkkhma\b', 'khoma', final_text, flags=re.IGNORECASE)
|
|
143
|
+
final_text = re.sub(r'\bljjoa\b', 'lojja', final_text, flags=re.IGNORECASE)
|
|
144
|
+
final_text = re.sub(r'\bodrishy\b', 'odrishyo', final_text, flags=re.IGNORECASE)
|
|
145
|
+
final_text = re.sub(r'\bbouddhik\b', 'boudhik', final_text, flags=re.IGNORECASE)
|
|
146
|
+
final_text = re.sub(r'\bsmbhoabybhabe\b', 'shombhabobhabe', final_text, flags=re.IGNORECASE)
|
|
147
|
+
|
|
148
|
+
return re.sub(r'\s+', ' ', final_text).strip()
|
|
149
|
+
|
|
150
|
+
def to_natural_codemixed(text: str) -> str:
|
|
151
|
+
"""
|
|
152
|
+
Transforms standard Bengali text into natural urban Code-Mixed Bengali-English.
|
|
153
|
+
Replaces technical, conversational, and security concepts with authentic English loanwords
|
|
154
|
+
while preserving Bengali morpho-syntax.
|
|
155
|
+
"""
|
|
156
|
+
if not text or not isinstance(text, str):
|
|
157
|
+
return ""
|
|
158
|
+
res = text
|
|
159
|
+
for pattern, rep in CODEMIXED_REPLACEMENTS:
|
|
160
|
+
res = re.sub(pattern, rep, res)
|
|
161
|
+
return clean_repetition_trash(res)
|
|
162
|
+
|
|
163
|
+
class MultiScriptConverter:
|
|
164
|
+
"""
|
|
165
|
+
High-level class for converting Bengali text, batches, or DataFrames
|
|
166
|
+
into multiple parallel scripts.
|
|
167
|
+
"""
|
|
168
|
+
def __init__(self, cache_size: int = 10000, translator=None):
|
|
169
|
+
self._cache = {}
|
|
170
|
+
self.cache_size = cache_size
|
|
171
|
+
self._translator = translator
|
|
172
|
+
|
|
173
|
+
def convert_sentence(self, text: str, mode: str = 'all', src_lang: str = 'bn') -> Union[str, Dict[str, str]]:
|
|
174
|
+
"""
|
|
175
|
+
Converts a single sentence.
|
|
176
|
+
mode: 'banglish', 'codemixed', or 'all' (returns dict)
|
|
177
|
+
src_lang: 'bn' (Bengali, default) or 'en' (English, translated first)
|
|
178
|
+
"""
|
|
179
|
+
if not text or not isinstance(text, str):
|
|
180
|
+
return "" if mode != 'all' else {'bangla': '', 'banglish': '', 'codemixed': ''}
|
|
181
|
+
|
|
182
|
+
bn_text = text
|
|
183
|
+
if src_lang == 'en':
|
|
184
|
+
if self._translator is None:
|
|
185
|
+
from .translator import get_translator
|
|
186
|
+
self._translator = get_translator("auto")
|
|
187
|
+
bn_text = self._translator.translate(text)
|
|
188
|
+
|
|
189
|
+
if mode == 'banglish':
|
|
190
|
+
return to_natural_banglish(bn_text)
|
|
191
|
+
elif mode == 'codemixed':
|
|
192
|
+
return to_natural_codemixed(bn_text)
|
|
193
|
+
else:
|
|
194
|
+
res = {
|
|
195
|
+
'bangla': bn_text,
|
|
196
|
+
'banglish': to_natural_banglish(bn_text),
|
|
197
|
+
'codemixed': to_natural_codemixed(bn_text)
|
|
198
|
+
}
|
|
199
|
+
if src_lang == 'en':
|
|
200
|
+
res['english'] = text
|
|
201
|
+
return res
|
|
202
|
+
|
|
203
|
+
def convert_batch(self, texts: List[str], mode: str = 'banglish', src_lang: str = 'bn') -> List[Union[str, Dict[str, str]]]:
|
|
204
|
+
"""Converts a list of sentences."""
|
|
205
|
+
return [self.convert_sentence(t, mode=mode, src_lang=src_lang) for t in texts]
|
|
206
|
+
|
|
207
|
+
def convert_dataframe(self, df, text_column: str,
|
|
208
|
+
banglish_col: str = 'text_banglish',
|
|
209
|
+
codemixed_col: str = 'text_codemixed'):
|
|
210
|
+
"""
|
|
211
|
+
Takes a pandas DataFrame and adds natural Banglish and Code-Mixed columns in-place.
|
|
212
|
+
"""
|
|
213
|
+
df[banglish_col] = df[text_column].apply(to_natural_banglish)
|
|
214
|
+
df[codemixed_col] = df[text_column].apply(to_natural_codemixed)
|
|
215
|
+
return df
|
|
216
|
+
|
|
217
|
+
def all_in_one(
|
|
218
|
+
text: str,
|
|
219
|
+
src_lang: str = "auto",
|
|
220
|
+
translator_backend: str = "auto",
|
|
221
|
+
translator_model_path: Optional[str] = None
|
|
222
|
+
) -> Dict[str, str]:
|
|
223
|
+
"""
|
|
224
|
+
All-in-one unified converter.
|
|
225
|
+
Accepts Bengali or English text and outputs all script representations:
|
|
226
|
+
- Bengali (বাংলা)
|
|
227
|
+
- Natural Avro Banglish
|
|
228
|
+
- Modern Urban Code-Mixed
|
|
229
|
+
- (English original if translated)
|
|
230
|
+
|
|
231
|
+
Parameters:
|
|
232
|
+
text (str): Input text in Bengali or English.
|
|
233
|
+
src_lang (str): 'auto', 'bn', or 'en'.
|
|
234
|
+
translator_backend (str): 'auto', 'web', 'indictrans2', or 'none'.
|
|
235
|
+
translator_model_path (str, optional): Custom path or HuggingFace ID for IndicTrans2.
|
|
236
|
+
"""
|
|
237
|
+
if not text or not isinstance(text, str):
|
|
238
|
+
return {"bangla": "", "banglish": "", "codemixed": ""}
|
|
239
|
+
|
|
240
|
+
is_english = False
|
|
241
|
+
if src_lang == "en":
|
|
242
|
+
is_english = True
|
|
243
|
+
elif src_lang == "auto":
|
|
244
|
+
# Detect if text contains Bengali Unicode characters (U+0980 to U+09FF)
|
|
245
|
+
has_bengali = bool(re.search(r'[\u0980-\u09FF]', text))
|
|
246
|
+
if not has_bengali and any(c.isalpha() for c in text):
|
|
247
|
+
is_english = True
|
|
248
|
+
|
|
249
|
+
if is_english and translator_backend != "none":
|
|
250
|
+
from .translator import get_translator
|
|
251
|
+
tr = get_translator(backend=translator_backend, model_path=translator_model_path)
|
|
252
|
+
bn_text = tr.translate(text)
|
|
253
|
+
return {
|
|
254
|
+
"english": text,
|
|
255
|
+
"bangla": bn_text,
|
|
256
|
+
"banglish": to_natural_banglish(bn_text),
|
|
257
|
+
"codemixed": to_natural_codemixed(bn_text)
|
|
258
|
+
}
|
|
259
|
+
else:
|
|
260
|
+
return {
|
|
261
|
+
"bangla": text,
|
|
262
|
+
"banglish": to_natural_banglish(text),
|
|
263
|
+
"codemixed": to_natural_codemixed(text)
|
|
264
|
+
}
|
|
265
|
+
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
BanglaMultiScript Dataset Exporter Module
|
|
4
|
+
Exports parallel datasets into popular LLM fine-tuning schemas:
|
|
5
|
+
- ShareGPT (Conversations schema used by Unsloth, LLaMA-Factory, FastChat)
|
|
6
|
+
- Alpaca (Instruction, Input, Output)
|
|
7
|
+
- DPO (Direct Preference Optimization: prompt, chosen, rejected)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from typing import List, Dict, Union, Optional, Any
|
|
12
|
+
|
|
13
|
+
def _to_records(data: Any) -> List[Dict]:
|
|
14
|
+
"""Helper to convert pandas DataFrame or list of dicts to standard list of dicts."""
|
|
15
|
+
if hasattr(data, 'to_dict'):
|
|
16
|
+
return data.to_dict(orient='records')
|
|
17
|
+
elif isinstance(data, list):
|
|
18
|
+
return data
|
|
19
|
+
raise TypeError(f"Unsupported data type: {type(data)}. Expected list of dicts or pandas DataFrame.")
|
|
20
|
+
|
|
21
|
+
def export_to_sharegpt(
|
|
22
|
+
data: Any,
|
|
23
|
+
output_file: Optional[str] = None,
|
|
24
|
+
prompt_col: str = "prompt",
|
|
25
|
+
response_col: str = "response",
|
|
26
|
+
system_prompt: Optional[str] = None
|
|
27
|
+
) -> List[Dict]:
|
|
28
|
+
"""
|
|
29
|
+
Exports dataset into ShareGPT / FastChat / LLaMA-Factory format:
|
|
30
|
+
[
|
|
31
|
+
{
|
|
32
|
+
"conversations": [
|
|
33
|
+
{"from": "system", "value": ...}, # Optional
|
|
34
|
+
{"from": "human", "value": ...},
|
|
35
|
+
{"from": "gpt", "value": ...}
|
|
36
|
+
]
|
|
37
|
+
}
|
|
38
|
+
]
|
|
39
|
+
"""
|
|
40
|
+
records = _to_records(data)
|
|
41
|
+
sharegpt_list = []
|
|
42
|
+
|
|
43
|
+
for item in records:
|
|
44
|
+
human_msg = str(item.get(prompt_col, "")).strip()
|
|
45
|
+
gpt_msg = str(item.get(response_col, "")).strip()
|
|
46
|
+
|
|
47
|
+
convs = []
|
|
48
|
+
if system_prompt:
|
|
49
|
+
convs.append({"from": "system", "value": system_prompt})
|
|
50
|
+
convs.append({"from": "human", "value": human_msg})
|
|
51
|
+
convs.append({"from": "gpt", "value": gpt_msg})
|
|
52
|
+
|
|
53
|
+
sharegpt_list.append({"conversations": convs})
|
|
54
|
+
|
|
55
|
+
if output_file:
|
|
56
|
+
with open(output_file, 'w', encoding='utf-8') as f:
|
|
57
|
+
json.dump(sharegpt_list, f, ensure_ascii=False, indent=2)
|
|
58
|
+
|
|
59
|
+
return sharegpt_list
|
|
60
|
+
|
|
61
|
+
def export_to_alpaca(
|
|
62
|
+
data: Any,
|
|
63
|
+
output_file: Optional[str] = None,
|
|
64
|
+
instruction_col: str = "prompt",
|
|
65
|
+
output_col: str = "response",
|
|
66
|
+
input_col: Optional[str] = None
|
|
67
|
+
) -> List[Dict]:
|
|
68
|
+
"""
|
|
69
|
+
Exports dataset into Stanford Alpaca format:
|
|
70
|
+
[
|
|
71
|
+
{
|
|
72
|
+
"instruction": ...,
|
|
73
|
+
"input": ...,
|
|
74
|
+
"output": ...
|
|
75
|
+
}
|
|
76
|
+
]
|
|
77
|
+
"""
|
|
78
|
+
records = _to_records(data)
|
|
79
|
+
alpaca_list = []
|
|
80
|
+
|
|
81
|
+
for item in records:
|
|
82
|
+
inst = str(item.get(instruction_col, "")).strip()
|
|
83
|
+
out = str(item.get(output_col, "")).strip()
|
|
84
|
+
inp = str(item.get(input_col, "")).strip() if input_col else ""
|
|
85
|
+
|
|
86
|
+
alpaca_list.append({
|
|
87
|
+
"instruction": inst,
|
|
88
|
+
"input": inp,
|
|
89
|
+
"output": out
|
|
90
|
+
})
|
|
91
|
+
|
|
92
|
+
if output_file:
|
|
93
|
+
with open(output_file, 'w', encoding='utf-8') as f:
|
|
94
|
+
json.dump(alpaca_list, f, ensure_ascii=False, indent=2)
|
|
95
|
+
|
|
96
|
+
return alpaca_list
|
|
97
|
+
|
|
98
|
+
def export_to_dpo(
|
|
99
|
+
data: Any,
|
|
100
|
+
output_file: Optional[str] = None,
|
|
101
|
+
prompt_col: str = "prompt",
|
|
102
|
+
chosen_col: str = "chosen",
|
|
103
|
+
rejected_col: str = "rejected"
|
|
104
|
+
) -> List[Dict]:
|
|
105
|
+
"""
|
|
106
|
+
Exports dataset into DPO (Direct Preference Optimization) format (TRL / Hugging Face format):
|
|
107
|
+
[
|
|
108
|
+
{
|
|
109
|
+
"prompt": ...,
|
|
110
|
+
"chosen": ...,
|
|
111
|
+
"rejected": ...
|
|
112
|
+
}
|
|
113
|
+
]
|
|
114
|
+
"""
|
|
115
|
+
records = _to_records(data)
|
|
116
|
+
dpo_list = []
|
|
117
|
+
|
|
118
|
+
for item in records:
|
|
119
|
+
p = str(item.get(prompt_col, "")).strip()
|
|
120
|
+
c = str(item.get(chosen_col, "")).strip()
|
|
121
|
+
r = str(item.get(rejected_col, "")).strip()
|
|
122
|
+
|
|
123
|
+
dpo_list.append({
|
|
124
|
+
"prompt": p,
|
|
125
|
+
"chosen": c,
|
|
126
|
+
"rejected": r
|
|
127
|
+
})
|
|
128
|
+
|
|
129
|
+
if output_file:
|
|
130
|
+
is_jsonl = output_file.endswith('.jsonl')
|
|
131
|
+
with open(output_file, 'w', encoding='utf-8') as f:
|
|
132
|
+
if is_jsonl:
|
|
133
|
+
for row in dpo_list:
|
|
134
|
+
f.write(json.dumps(row, ensure_ascii=False) + '\n')
|
|
135
|
+
else:
|
|
136
|
+
json.dump(dpo_list, f, ensure_ascii=False, indent=2)
|
|
137
|
+
|
|
138
|
+
return dpo_list
|