bangla-multiscript 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,52 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ BanglaMultiScript: High-Throughput Bengali to Natural Avro Banglish & Code-Mixed Alignment Engine.
4
+ """
5
+
6
+ __version__ = "1.0.0"
7
+ __author__ = "Shahriar"
8
+
9
+ from .engine import (
10
+ to_natural_banglish,
11
+ to_natural_codemixed,
12
+ clean_repetition_trash,
13
+ phonological_word_to_banglish,
14
+ MultiScriptConverter,
15
+ all_in_one
16
+ )
17
+ from .translator import IndicTrans2Translator, WebTranslator, CustomTranslator, get_translator
18
+ from .detector import detect_script, get_script_stats
19
+ from .normalizer import normalize_bangla, normalize_banglish, normalize_text
20
+ from .exporter import export_to_sharegpt, export_to_alpaca, export_to_dpo
21
+ from .ui import launch_ui
22
+
23
+ __all__ = [
24
+ # Core Engine
25
+ "to_natural_banglish",
26
+ "to_natural_codemixed",
27
+ "clean_repetition_trash",
28
+ "phonological_word_to_banglish",
29
+ "MultiScriptConverter",
30
+ "all_in_one",
31
+ # Translation
32
+ "IndicTrans2Translator",
33
+ "WebTranslator",
34
+ "CustomTranslator",
35
+ "get_translator",
36
+ # Script Detection
37
+ "detect_script",
38
+ "get_script_stats",
39
+ # Text Normalization
40
+ "normalize_bangla",
41
+ "normalize_banglish",
42
+ "normalize_text",
43
+ # LLM Dataset Exporters
44
+ "export_to_sharegpt",
45
+ "export_to_alpaca",
46
+ "export_to_dpo",
47
+ # Web UI
48
+ "launch_ui",
49
+ # Metadata
50
+ "__version__",
51
+ ]
52
+
@@ -0,0 +1,112 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ Command Line Interface for BanglaMultiScript
4
+ """
5
+
6
+ import argparse
7
+ import sys
8
+ import os
9
+ import json
10
+
11
+ # Ensure UTF-8 output on Windows consoles
12
+ if hasattr(sys.stdout, "reconfigure"):
13
+ try:
14
+ sys.stdout.reconfigure(encoding="utf-8")
15
+ except Exception:
16
+ pass
17
+
18
+ from .engine import to_natural_banglish, to_natural_codemixed, MultiScriptConverter
19
+
20
+ def main():
21
+ parser = argparse.ArgumentParser(
22
+ description="BanglaMultiScript: Convert Bengali text to Natural Avro Banglish & Urban Code-Mixed."
23
+ )
24
+ parser.add_argument("input", nargs="?", help="Input string or path to a text/CSV/JSONL file.")
25
+ parser.add_argument(
26
+ "--mode", choices=["banglish", "codemixed", "all"], default="all",
27
+ help="Conversion mode (default: all)"
28
+ )
29
+ parser.add_argument("--output", "-o", help="Output file path (optional).")
30
+ parser.add_argument("--text-col", default="text", help="Text column name for CSV/JSONL input.")
31
+ parser.add_argument("--ui", action="store_true", help="Launch the interactive Web UI studio in browser.")
32
+ parser.add_argument("--port", type=int, default=7860, help="Port for the Web UI studio (default: 7860).")
33
+ parser.add_argument("--detect", action="store_true", help="Detect script/language (bengali, banglish, english, codemixed).")
34
+ parser.add_argument("--normalize", action="store_true", help="Normalize and clean text (remove ZWNJ, slang, letter stretching).")
35
+
36
+ args = parser.parse_args()
37
+
38
+ if args.ui:
39
+ from .ui import launch_ui
40
+ launch_ui(port=args.port)
41
+ return
42
+
43
+ if not args.input:
44
+ parser.print_help()
45
+ sys.exit(0)
46
+
47
+ if args.detect:
48
+ from .detector import detect_script
49
+ print(detect_script(args.input))
50
+ return
51
+
52
+ if args.normalize:
53
+ from .normalizer import normalize_text
54
+ print(normalize_text(args.input))
55
+ return
56
+
57
+ # Check if input is a file
58
+ if os.path.isfile(args.input):
59
+ ext = os.path.splitext(args.input)[1].lower()
60
+ if ext == '.csv':
61
+ import pandas as pd
62
+ df = pd.read_csv(args.input)
63
+ conv = MultiScriptConverter()
64
+ df = conv.convert_dataframe(df, text_column=args.text_col)
65
+ out_path = args.output or args.input.replace('.csv', '_multiscript.csv')
66
+ df.to_csv(out_path, index=False, encoding='utf-8')
67
+ print(f"Saved converted CSV ({len(df)} rows) to: {out_path}")
68
+ elif ext == '.jsonl':
69
+ out_path = args.output or args.input.replace('.jsonl', '_multiscript.jsonl')
70
+ count = 0
71
+ with open(args.input, 'r', encoding='utf-8') as fin, open(out_path, 'w', encoding='utf-8') as fout:
72
+ for line in fin:
73
+ d = json.loads(line)
74
+ raw_text = d.get(args.text_col, '')
75
+ d[f'{args.text_col}_banglish'] = to_natural_banglish(raw_text)
76
+ d[f'{args.text_col}_codemixed'] = to_natural_codemixed(raw_text)
77
+ fout.write(json.dumps(d, ensure_ascii=False) + '\n')
78
+ count += 1
79
+ print(f"Saved converted JSONL ({count} records) to: {out_path}")
80
+ else:
81
+ # Plain text file line by line
82
+ out_path = args.output or args.input + '.converted'
83
+ with open(args.input, 'r', encoding='utf-8') as fin, open(out_path, 'w', encoding='utf-8') as fout:
84
+ for line in fin:
85
+ if args.mode == 'banglish':
86
+ fout.write(to_natural_banglish(line.strip()) + '\n')
87
+ elif args.mode == 'codemixed':
88
+ fout.write(to_natural_codemixed(line.strip()) + '\n')
89
+ else:
90
+ res = {
91
+ 'bn': line.strip(),
92
+ 'banglish': to_natural_banglish(line.strip()),
93
+ 'codemixed': to_natural_codemixed(line.strip())
94
+ }
95
+ fout.write(json.dumps(res, ensure_ascii=False) + '\n')
96
+ print(f"Saved output to: {out_path}")
97
+ else:
98
+ # Input is a direct string
99
+ if args.mode == 'banglish':
100
+ print(to_natural_banglish(args.input))
101
+ elif args.mode == 'codemixed':
102
+ print(to_natural_codemixed(args.input))
103
+ else:
104
+ print("--- Original Bangla ---")
105
+ print(args.input)
106
+ print("\n--- Natural Avro Banglish ---")
107
+ print(to_natural_banglish(args.input))
108
+ print("\n--- Natural Code-Mixed ---")
109
+ print(to_natural_codemixed(args.input))
110
+
111
+ if __name__ == "__main__":
112
+ main()
@@ -0,0 +1,112 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ BanglaMultiScript Script & Language Identification (LID) Module
4
+ Detects whether a given text is Bengali, Avro Banglish, English, or Code-Mixed.
5
+ """
6
+
7
+ import re
8
+ from typing import Dict, Union
9
+
10
+ # Common Banglish grammatical particles and high-frequency phonetic markers
11
+ BANGLISH_MARKERS = {
12
+ "ami", "tumi", "apni", "amra", "tomra", "apnara", "she", "tini", "tara",
13
+ "kemon", "achi", "achho", "achhen", "bhalo", "valo", "korcho", "korchhi",
14
+ "korben", "korte", "hobe", "hobena", "hoy", "hoyeche", "hoyechhe",
15
+ "keno", "kintu", "ebong", "ar", "aar", "aaroo", "aro", "kothay", "kokhon",
16
+ "kivabe", "ki", "kee", "shob", "sob", "thik", "ekhon", "tokhon",
17
+ "gotokal", "aaj", "ajke", "shathe", "sathe", "theke", "por", "pore",
18
+ "jani", "janina", "dekhi", "dekhun", "bolun", "bolte", "shunte",
19
+ "parbo", "parbena", "parbe", "uchit", "chilo", "chhilo", "achhe",
20
+ "eta", "ota", "ei", "oi", "ekta", "duita", "shunte", "bujhte",
21
+ "darao", "ashbo", "jabo", "korechi", "gesilam", "giyechilam"
22
+ }
23
+
24
+ # Common English stopwords
25
+ ENGLISH_MARKERS = {
26
+ "the", "is", "are", "was", "were", "and", "or", "but", "if", "then",
27
+ "what", "why", "how", "when", "where", "who", "which", "this", "that",
28
+ "these", "those", "have", "has", "had", "will", "would", "shall",
29
+ "should", "can", "could", "may", "might", "must", "with", "from",
30
+ "about", "against", "between", "into", "through", "during", "before",
31
+ "after", "above", "below", "to", "of", "for", "in", "on", "at", "by"
32
+ }
33
+
34
+ def get_script_stats(text: str) -> Dict[str, float]:
35
+ """
36
+ Computes character-level distribution across scripts:
37
+ Returns percentages of Bengali, Latin, Digits, and Punctuation/Whitespace.
38
+ """
39
+ if not text or not isinstance(text, str):
40
+ return {"bengali": 0.0, "latin": 0.0, "digits": 0.0, "other": 0.0, "total_chars": 0}
41
+
42
+ total = len(text)
43
+ bn_count = len(re.findall(r'[\u0980-\u09FF]', text))
44
+ latin_count = len(re.findall(r'[a-zA-Z]', text))
45
+ digit_count = len(re.findall(r'[0-9\u09E6-\u09EF]', text))
46
+ other_count = total - (bn_count + latin_count + digit_count)
47
+
48
+ return {
49
+ "bengali": round(bn_count / total, 4),
50
+ "latin": round(latin_count / total, 4),
51
+ "digits": round(digit_count / total, 4),
52
+ "other": round(other_count / total, 4),
53
+ "total_chars": total
54
+ }
55
+
56
+ def detect_script(text: str) -> str:
57
+ """
58
+ Detects the predominant script / language form of the text:
59
+ - 'bengali': Formal Bengali script (বাংলা)
60
+ - 'banglish': Bengali written using English / Latin letters (Avro Banglish)
61
+ - 'codemixed': Sentence containing significant blend of Bengali and English words
62
+ - 'english': Standard English text
63
+ - 'unknown': Empty or numeric/symbol-only text
64
+ """
65
+ if not text or not isinstance(text, str):
66
+ return "unknown"
67
+
68
+ stats = get_script_stats(text)
69
+ bn_ratio = stats["bengali"]
70
+ latin_ratio = stats["latin"]
71
+
72
+ # If virtually no alphabetic characters
73
+ if bn_ratio == 0 and latin_ratio == 0:
74
+ return "unknown"
75
+
76
+ # Both Bengali and Latin present in meaningful amounts -> Code-Mixed
77
+ if bn_ratio >= 0.15 and latin_ratio >= 0.15:
78
+ return "codemixed"
79
+
80
+ # Predominantly Bengali script
81
+ if bn_ratio > 0.40 and latin_ratio < 0.15:
82
+ return "bengali"
83
+
84
+ # Predominantly Latin characters -> Distinguish Banglish vs English
85
+ tokens = [re.sub(r'[^a-zA-Z]', '', w).lower() for w in text.split()]
86
+ tokens = [w for w in tokens if len(w) > 1]
87
+
88
+ if not tokens:
89
+ return "unknown"
90
+
91
+ banglish_score = sum(1 for t in tokens if t in BANGLISH_MARKERS)
92
+ english_score = sum(1 for t in tokens if t in ENGLISH_MARKERS)
93
+
94
+ # If Banglish marker hits exist and exceed English
95
+ if banglish_score > english_score:
96
+ return "banglish"
97
+ elif english_score > banglish_score:
98
+ return "english"
99
+
100
+ # Secondary heuristic: phonetic character combinations typical in Banglish
101
+ # (e.g. 'kh', 'gh', 'ch', 'jh', 'th', 'dh', 'bh', 'sh', 'ng', 'chho')
102
+ banglish_phonetic_pattern = re.compile(r'(chho|kkh|sh|bh|dh|th|jh|ch|gh|kh|ng|oy|ye)')
103
+ phonetic_hits = len(banglish_phonetic_pattern.findall(text.lower()))
104
+ word_count = len(tokens)
105
+
106
+ if word_count > 0 and (phonetic_hits / word_count) >= 0.6:
107
+ return "banglish"
108
+
109
+ # Fallback based on dominant character ratio
110
+ if bn_ratio > latin_ratio:
111
+ return "bengali"
112
+ return "english"
@@ -0,0 +1,265 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ BanglaMultiScript Core Engine
4
+ Provides high-throughput, linguistically sound conversion from standard Bengali
5
+ to Natural Avro Banglish and Natural Code-Mixed text.
6
+ """
7
+
8
+ import re
9
+ import unicodedata
10
+ from typing import List, Dict, Union, Optional
11
+
12
+ from .lexicon import (
13
+ LOANWORDS_AND_ENTITIES,
14
+ COLLOQUIAL_LEXICON,
15
+ CODEMIXED_REPLACEMENTS,
16
+ VOWELS_INDEP,
17
+ VOWELS_DEP,
18
+ CONSONANTS,
19
+ CONJUNCT_MAP
20
+ )
21
+
22
+ def clean_repetition_trash(text: str) -> str:
23
+ """
24
+ Cleans up artificial repetition loops and trailing acronym spam
25
+ common in machine translation outputs (e.g. 'si. es. si. es...').
26
+ """
27
+ if not text or not isinstance(text, str):
28
+ return ""
29
+ # Match identical short token (e.g. 'si. ' or 'es. ') repeated 3+ times
30
+ cleaned = re.sub(r'(?:(\b[a-zA-Z\u0980-\u09FF]{1,4}\.?\s+)\1{2,})', r'\1', text)
31
+ # Match 2-token phrase repeated 3+ times (e.g. 'si. es. si. es. si. es.')
32
+ cleaned = re.sub(r'(?:(\b\w+\.?\s+\w+\.?\s+)\1{2,})', r'\1', cleaned)
33
+ # Match any word repeated consecutively 3+ times
34
+ cleaned = re.sub(r'\b(\w+)(?:\s+\1){2,}\b', r'\1', cleaned)
35
+ # Clean trailing acronym loops (e.g. trailing 'সি. এস. সি. এস.')
36
+ cleaned = re.sub(r'(?:[a-zA-Z\u0980-\u09FF]{1,3}\.\s*){4,}', ' ', cleaned)
37
+ cleaned = re.sub(r'[,.\s]+$', '.', cleaned)
38
+ return re.sub(r'\s+', ' ', cleaned).strip()
39
+
40
+ def phonological_word_to_banglish(word: str) -> str:
41
+ """
42
+ Transliterates a single Bengali token into natural Avro Banglish
43
+ respecting inherent schwa vowel deletion, conjunct gemination, and common suffixes.
44
+ """
45
+ clean = re.sub(r'[^\w\u0980-\u09FF]', '', word)
46
+ if not clean:
47
+ return word
48
+
49
+ # 1. Direct Colloquial Lexicon Lookup
50
+ if clean in COLLOQUIAL_LEXICON:
51
+ return COLLOQUIAL_LEXICON[clean]
52
+
53
+ # 2. Suffix Decomposition
54
+ for suffix, rep in [
55
+ ('ের', 'er'), ('দের', 'der'), ('গুলো', 'gulo'), ('গুলি', 'guli'),
56
+ ('কে', 'ke'), ('তে', 'te'), ('টা', 'ta'), ('টি', 'ti'),
57
+ ('ভাবে', 'bhabe'), ('মূলক', 'mulok'), ('কারী', 'kari'), ('কারিদের', 'karider')
58
+ ]:
59
+ if clean.endswith(suffix) and len(clean) > len(suffix) + 1:
60
+ stem = clean[:-len(suffix)]
61
+ if stem in COLLOQUIAL_LEXICON:
62
+ return COLLOQUIAL_LEXICON[stem] + ('-' if suffix in ['টা', 'টি'] else '') + rep
63
+
64
+ # 3. Conjunct Resolution
65
+ w = unicodedata.normalize('NFC', clean)
66
+ for cj, rep in CONJUNCT_MAP.items():
67
+ w = w.replace(cj, rep)
68
+
69
+ # 4. Phonetic Traversal
70
+ res = []
71
+ i = 0
72
+ n = len(w)
73
+ while i < n:
74
+ c = w[i]
75
+ if i + 1 < n and w[i+1] == '়':
76
+ c = c + '়'
77
+ i += 1
78
+
79
+ if c in VOWELS_INDEP:
80
+ res.append(VOWELS_INDEP[c])
81
+ elif c in CONSONANTS:
82
+ res.append(CONSONANTS[c])
83
+ if i + 1 < n:
84
+ nxt = w[i+1]
85
+ if nxt == '্':
86
+ i += 1 # Skip hasanta
87
+ elif nxt in VOWELS_DEP:
88
+ res.append(VOWELS_DEP[nxt])
89
+ i += 1
90
+ elif nxt in CONSONANTS or nxt in VOWELS_INDEP:
91
+ # Inherent vowel rule: insert 'o' between consonants
92
+ res.append('o')
93
+ elif c in VOWELS_DEP:
94
+ res.append(VOWELS_DEP[c])
95
+ else:
96
+ res.append(c)
97
+ i += 1
98
+
99
+ s = ''.join(res)
100
+ # Clean double 'o' or unnatural clusters
101
+ s = re.sub(r'oo+', 'o', s)
102
+ s = re.sub(r'yy+', 'y', s)
103
+ return s
104
+
105
+ def to_natural_banglish(text: str) -> str:
106
+ """
107
+ Converts a standard Bengali sentence into colloquial Avro Banglish.
108
+ Preserves English loanwords, technical nomenclature, and entity names in proper Latin.
109
+ """
110
+ if not text or not isinstance(text, str):
111
+ return ""
112
+
113
+ # 1. Map known loanwords and entities
114
+ res = text
115
+ for bn_term, eng_term in LOANWORDS_AND_ENTITIES:
116
+ pattern = re.escape(bn_term)
117
+ res = re.sub(pattern, eng_term, res)
118
+
119
+ # 2. Tokenize and convert remaining Bengali script tokens
120
+ tokens = re.split(r'(\s+|[.,!?;:\"\'\(\)\[\]|।])', res)
121
+ out = []
122
+ for t in tokens:
123
+ if not t:
124
+ continue
125
+ if re.search(r'[\u0980-\u09FF]', t):
126
+ out.append(phonological_word_to_banglish(t))
127
+ elif t == '।':
128
+ out.append('.')
129
+ else:
130
+ out.append(t)
131
+
132
+ final_text = ''.join(out)
133
+ final_text = clean_repetition_trash(final_text)
134
+
135
+ # Common post-cleanup heuristics
136
+ final_text = re.sub(r'\bkno\b', 'keno', final_text, flags=re.IGNORECASE)
137
+ final_text = re.sub(r'\bbaddhy\b', 'baddho', final_text, flags=re.IGNORECASE)
138
+ final_text = re.sub(r'\bsmpork\b', 'shomporko', final_text, flags=re.IGNORECASE)
139
+ final_text = re.sub(r'\bsmpourn\b', 'shompurno', final_text, flags=re.IGNORECASE)
140
+ final_text = re.sub(r'\bchndrer\b', 'chondrer', final_text, flags=re.IGNORECASE)
141
+ final_text = re.sub(r'\bchndro\b', 'chondro', final_text, flags=re.IGNORECASE)
142
+ final_text = re.sub(r'\bkkhma\b', 'khoma', final_text, flags=re.IGNORECASE)
143
+ final_text = re.sub(r'\bljjoa\b', 'lojja', final_text, flags=re.IGNORECASE)
144
+ final_text = re.sub(r'\bodrishy\b', 'odrishyo', final_text, flags=re.IGNORECASE)
145
+ final_text = re.sub(r'\bbouddhik\b', 'boudhik', final_text, flags=re.IGNORECASE)
146
+ final_text = re.sub(r'\bsmbhoabybhabe\b', 'shombhabobhabe', final_text, flags=re.IGNORECASE)
147
+
148
+ return re.sub(r'\s+', ' ', final_text).strip()
149
+
150
+ def to_natural_codemixed(text: str) -> str:
151
+ """
152
+ Transforms standard Bengali text into natural urban Code-Mixed Bengali-English.
153
+ Replaces technical, conversational, and security concepts with authentic English loanwords
154
+ while preserving Bengali morpho-syntax.
155
+ """
156
+ if not text or not isinstance(text, str):
157
+ return ""
158
+ res = text
159
+ for pattern, rep in CODEMIXED_REPLACEMENTS:
160
+ res = re.sub(pattern, rep, res)
161
+ return clean_repetition_trash(res)
162
+
163
+ class MultiScriptConverter:
164
+ """
165
+ High-level class for converting Bengali text, batches, or DataFrames
166
+ into multiple parallel scripts.
167
+ """
168
+ def __init__(self, cache_size: int = 10000, translator=None):
169
+ self._cache = {}
170
+ self.cache_size = cache_size
171
+ self._translator = translator
172
+
173
+ def convert_sentence(self, text: str, mode: str = 'all', src_lang: str = 'bn') -> Union[str, Dict[str, str]]:
174
+ """
175
+ Converts a single sentence.
176
+ mode: 'banglish', 'codemixed', or 'all' (returns dict)
177
+ src_lang: 'bn' (Bengali, default) or 'en' (English, translated first)
178
+ """
179
+ if not text or not isinstance(text, str):
180
+ return "" if mode != 'all' else {'bangla': '', 'banglish': '', 'codemixed': ''}
181
+
182
+ bn_text = text
183
+ if src_lang == 'en':
184
+ if self._translator is None:
185
+ from .translator import get_translator
186
+ self._translator = get_translator("auto")
187
+ bn_text = self._translator.translate(text)
188
+
189
+ if mode == 'banglish':
190
+ return to_natural_banglish(bn_text)
191
+ elif mode == 'codemixed':
192
+ return to_natural_codemixed(bn_text)
193
+ else:
194
+ res = {
195
+ 'bangla': bn_text,
196
+ 'banglish': to_natural_banglish(bn_text),
197
+ 'codemixed': to_natural_codemixed(bn_text)
198
+ }
199
+ if src_lang == 'en':
200
+ res['english'] = text
201
+ return res
202
+
203
+ def convert_batch(self, texts: List[str], mode: str = 'banglish', src_lang: str = 'bn') -> List[Union[str, Dict[str, str]]]:
204
+ """Converts a list of sentences."""
205
+ return [self.convert_sentence(t, mode=mode, src_lang=src_lang) for t in texts]
206
+
207
+ def convert_dataframe(self, df, text_column: str,
208
+ banglish_col: str = 'text_banglish',
209
+ codemixed_col: str = 'text_codemixed'):
210
+ """
211
+ Takes a pandas DataFrame and adds natural Banglish and Code-Mixed columns in-place.
212
+ """
213
+ df[banglish_col] = df[text_column].apply(to_natural_banglish)
214
+ df[codemixed_col] = df[text_column].apply(to_natural_codemixed)
215
+ return df
216
+
217
+ def all_in_one(
218
+ text: str,
219
+ src_lang: str = "auto",
220
+ translator_backend: str = "auto",
221
+ translator_model_path: Optional[str] = None
222
+ ) -> Dict[str, str]:
223
+ """
224
+ All-in-one unified converter.
225
+ Accepts Bengali or English text and outputs all script representations:
226
+ - Bengali (বাংলা)
227
+ - Natural Avro Banglish
228
+ - Modern Urban Code-Mixed
229
+ - (English original if translated)
230
+
231
+ Parameters:
232
+ text (str): Input text in Bengali or English.
233
+ src_lang (str): 'auto', 'bn', or 'en'.
234
+ translator_backend (str): 'auto', 'web', 'indictrans2', or 'none'.
235
+ translator_model_path (str, optional): Custom path or HuggingFace ID for IndicTrans2.
236
+ """
237
+ if not text or not isinstance(text, str):
238
+ return {"bangla": "", "banglish": "", "codemixed": ""}
239
+
240
+ is_english = False
241
+ if src_lang == "en":
242
+ is_english = True
243
+ elif src_lang == "auto":
244
+ # Detect if text contains Bengali Unicode characters (U+0980 to U+09FF)
245
+ has_bengali = bool(re.search(r'[\u0980-\u09FF]', text))
246
+ if not has_bengali and any(c.isalpha() for c in text):
247
+ is_english = True
248
+
249
+ if is_english and translator_backend != "none":
250
+ from .translator import get_translator
251
+ tr = get_translator(backend=translator_backend, model_path=translator_model_path)
252
+ bn_text = tr.translate(text)
253
+ return {
254
+ "english": text,
255
+ "bangla": bn_text,
256
+ "banglish": to_natural_banglish(bn_text),
257
+ "codemixed": to_natural_codemixed(bn_text)
258
+ }
259
+ else:
260
+ return {
261
+ "bangla": text,
262
+ "banglish": to_natural_banglish(text),
263
+ "codemixed": to_natural_codemixed(text)
264
+ }
265
+
@@ -0,0 +1,138 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ BanglaMultiScript Dataset Exporter Module
4
+ Exports parallel datasets into popular LLM fine-tuning schemas:
5
+ - ShareGPT (Conversations schema used by Unsloth, LLaMA-Factory, FastChat)
6
+ - Alpaca (Instruction, Input, Output)
7
+ - DPO (Direct Preference Optimization: prompt, chosen, rejected)
8
+ """
9
+
10
+ import json
11
+ from typing import List, Dict, Union, Optional, Any
12
+
13
+ def _to_records(data: Any) -> List[Dict]:
14
+ """Helper to convert pandas DataFrame or list of dicts to standard list of dicts."""
15
+ if hasattr(data, 'to_dict'):
16
+ return data.to_dict(orient='records')
17
+ elif isinstance(data, list):
18
+ return data
19
+ raise TypeError(f"Unsupported data type: {type(data)}. Expected list of dicts or pandas DataFrame.")
20
+
21
+ def export_to_sharegpt(
22
+ data: Any,
23
+ output_file: Optional[str] = None,
24
+ prompt_col: str = "prompt",
25
+ response_col: str = "response",
26
+ system_prompt: Optional[str] = None
27
+ ) -> List[Dict]:
28
+ """
29
+ Exports dataset into ShareGPT / FastChat / LLaMA-Factory format:
30
+ [
31
+ {
32
+ "conversations": [
33
+ {"from": "system", "value": ...}, # Optional
34
+ {"from": "human", "value": ...},
35
+ {"from": "gpt", "value": ...}
36
+ ]
37
+ }
38
+ ]
39
+ """
40
+ records = _to_records(data)
41
+ sharegpt_list = []
42
+
43
+ for item in records:
44
+ human_msg = str(item.get(prompt_col, "")).strip()
45
+ gpt_msg = str(item.get(response_col, "")).strip()
46
+
47
+ convs = []
48
+ if system_prompt:
49
+ convs.append({"from": "system", "value": system_prompt})
50
+ convs.append({"from": "human", "value": human_msg})
51
+ convs.append({"from": "gpt", "value": gpt_msg})
52
+
53
+ sharegpt_list.append({"conversations": convs})
54
+
55
+ if output_file:
56
+ with open(output_file, 'w', encoding='utf-8') as f:
57
+ json.dump(sharegpt_list, f, ensure_ascii=False, indent=2)
58
+
59
+ return sharegpt_list
60
+
61
+ def export_to_alpaca(
62
+ data: Any,
63
+ output_file: Optional[str] = None,
64
+ instruction_col: str = "prompt",
65
+ output_col: str = "response",
66
+ input_col: Optional[str] = None
67
+ ) -> List[Dict]:
68
+ """
69
+ Exports dataset into Stanford Alpaca format:
70
+ [
71
+ {
72
+ "instruction": ...,
73
+ "input": ...,
74
+ "output": ...
75
+ }
76
+ ]
77
+ """
78
+ records = _to_records(data)
79
+ alpaca_list = []
80
+
81
+ for item in records:
82
+ inst = str(item.get(instruction_col, "")).strip()
83
+ out = str(item.get(output_col, "")).strip()
84
+ inp = str(item.get(input_col, "")).strip() if input_col else ""
85
+
86
+ alpaca_list.append({
87
+ "instruction": inst,
88
+ "input": inp,
89
+ "output": out
90
+ })
91
+
92
+ if output_file:
93
+ with open(output_file, 'w', encoding='utf-8') as f:
94
+ json.dump(alpaca_list, f, ensure_ascii=False, indent=2)
95
+
96
+ return alpaca_list
97
+
98
+ def export_to_dpo(
99
+ data: Any,
100
+ output_file: Optional[str] = None,
101
+ prompt_col: str = "prompt",
102
+ chosen_col: str = "chosen",
103
+ rejected_col: str = "rejected"
104
+ ) -> List[Dict]:
105
+ """
106
+ Exports dataset into DPO (Direct Preference Optimization) format (TRL / Hugging Face format):
107
+ [
108
+ {
109
+ "prompt": ...,
110
+ "chosen": ...,
111
+ "rejected": ...
112
+ }
113
+ ]
114
+ """
115
+ records = _to_records(data)
116
+ dpo_list = []
117
+
118
+ for item in records:
119
+ p = str(item.get(prompt_col, "")).strip()
120
+ c = str(item.get(chosen_col, "")).strip()
121
+ r = str(item.get(rejected_col, "")).strip()
122
+
123
+ dpo_list.append({
124
+ "prompt": p,
125
+ "chosen": c,
126
+ "rejected": r
127
+ })
128
+
129
+ if output_file:
130
+ is_jsonl = output_file.endswith('.jsonl')
131
+ with open(output_file, 'w', encoding='utf-8') as f:
132
+ if is_jsonl:
133
+ for row in dpo_list:
134
+ f.write(json.dumps(row, ensure_ascii=False) + '\n')
135
+ else:
136
+ json.dump(dpo_list, f, ensure_ascii=False, indent=2)
137
+
138
+ return dpo_list