mas-asr 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mas_asr-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,236 @@
1
+ Metadata-Version: 2.3
2
+ Name: mas-asr
3
+ Version: 0.1.0
4
+ Summary: Production-grade, modular multilingual ASR library with native script, mixed script, streaming, and fine-tuning support
5
+ Author: ArYan9908
6
+ Author-email: ArYan9908 <9908aryan@gmail.com>
7
+ Requires-Dist: huggingface-hub>=0.23.0
8
+ Requires-Dist: peft>=0.21.2
9
+ Requires-Dist: sentencepiece>=0.2.2
10
+ Requires-Dist: sounddevice>=0.5.6
11
+ Requires-Dist: soundfile>=0.14.0
12
+ Requires-Dist: torch>=2.14.1
13
+ Requires-Dist: torchaudio>=2.11.0
14
+ Requires-Dist: transformers>=5.19.0
15
+ Requires-Dist: triton-windows>=3.8.0.post29
16
+ Requires-Dist: peft>=0.10.0 ; extra == 'finetune'
17
+ Requires-Dist: accelerate>=0.30.0 ; extra == 'finetune'
18
+ Requires-Dist: datasets>=2.19.0 ; extra == 'finetune'
19
+ Requires-Python: >=3.13
20
+ Provides-Extra: finetune
21
+ Description-Content-Type: text/markdown
22
+
23
+ # mas-asr
24
+
25
+ **mas-asr** is a production-grade, modular Python speech recognition library inspired by `openai-whisper` and `faster-whisper`. Built with first-class support for Indic languages, offering transcription in **native scripts**, **mixed scripts (Hinglish/code-mixing)**, and **romanized transliteration**, powered by models like **`bodhan-ai/indic-transcribe-flex`** (1.2B FastConformer + Transformer decoder).
26
+
27
+ ---
28
+
29
+ ## Key Features
30
+
31
+ - **Whisper-like Intuitive API**: Load and transcribe with minimal code (`mas_asr.load_model(...)` & `model.transcribe(...)`).
32
+ - **3 Output Script Modes**:
33
+ - `native`: Standard native script (Devanagari, Gurmukhi, Bengali, Tamil, etc.).
34
+ - `mixed`: Code-mixing & loanwords in Latin script with Indic digits (e.g., modern conversational messaging).
35
+ - `romanized`: Full phonetic transliteration in Latin script.
36
+ - **27+ Indian Languages**: Hindi, Tamil, Telugu, Bengali, Kannada, Marathi, Malayalam, Punjabi, Gujarati, Odia, Urdu, Assamese, and more + English.
37
+ - **Auto Long-Form Audio Segmentation**: Automatically splits audio exceeding 30 seconds at natural pauses using energy-based VAD to prevent context collapse.
38
+ - **Real-Time Streaming**: VAD-buffered sliding window streaming processor for microphone input and WebSockets.
39
+ - **Built-in CLI**: Transcribe directly from terminal with SRT, VTT, JSON, and TXT export.
40
+ - **LoRA & PEFT Fine-Tuning**: Out-of-the-box support for low-rank adapter fine-tuning on domain-specific datasets.
41
+
42
+ ---
43
+
44
+ ## Installation
45
+
46
+ Using `uv`:
47
+
48
+ ```bash
49
+ uv add mas-asr
50
+ ```
51
+
52
+ Or using standard `pip`:
53
+
54
+ ```bash
55
+ pip install mas-asr
56
+ ```
57
+
58
+ To install fine-tuning dependencies:
59
+
60
+ ```bash
61
+ pip install "mas-asr[finetune]"
62
+ ```
63
+
64
+ ---
65
+
66
+ ## Quickstart (Python API)
67
+
68
+ ### Basic Transcription
69
+
70
+ ```python
71
+ import mas_asr
72
+
73
+ # Load model (auto-downloads from Hugging Face or loads local checkpoint directory)
74
+ model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex", device="cuda", compute_type="bfloat16")
75
+
76
+ # Transcribe an audio file
77
+ result = model.transcribe(
78
+ "speech.mp3",
79
+ language="hi", # or None for automatic language detection
80
+ mode="native", # "native", "mixed", or "romanized"
81
+ vad_filter=True, # Auto-chunks long audio > 30s at natural pauses
82
+ )
83
+
84
+ print(result.text)
85
+ print(f"Language: {result.language} (Confidence: {result.language_probability:.2f})")
86
+
87
+ # Timestamped segments
88
+ for seg in result.segments:
89
+ print(f"[{seg.start:.2f}s -> {seg.end:.2f}s] {seg.text}")
90
+ ```
91
+
92
+ ### High-Speed Inference with `torch.compile`
93
+
94
+ Compile the model encoder to reduce kernel overhead and achieve faster transcription:
95
+
96
+ ```python
97
+ # Enable torch.compile for reduced kernel launch overhead
98
+ model = mas_asr.load_model(
99
+ "bodhan-ai/indic-transcribe-flex",
100
+ device="cuda",
101
+ compute_type="bfloat16",
102
+ compile=True, # Compiles FastConformer encoder
103
+ )
104
+ ```
105
+
106
+ ### Loading Trained LoRA Adapters
107
+
108
+ ```python
109
+ # Load base model with a custom domain adapter attached
110
+ model = mas_asr.load_model(
111
+ "bodhan-ai/indic-transcribe-flex",
112
+ adapter_path="./my_domain_adapters",
113
+ )
114
+ ```
115
+
116
+ ### Multi-Mode Transcription in One Go
117
+
118
+ ```python
119
+ # Transcribe into Native, Mixed, and Romanized scripts simultaneously
120
+ outputs = model.all_modes("audio.wav", language="hi")
121
+ print("Native: ", outputs["native"])
122
+ print("Mixed: ", outputs["mixed"])
123
+ print("Romanized:", outputs["romanized"])
124
+ ```
125
+
126
+ ### Export Subtitles (SRT / VTT)
127
+
128
+ ```python
129
+ # Save SubRip Subtitles
130
+ with open("subtitles.srt", "w", encoding="utf-8") as f:
131
+ f.write(result.to_srt())
132
+
133
+ # Save WebVTT Subtitles
134
+ with open("subtitles.vtt", "w", encoding="utf-8") as f:
135
+ f.write(result.to_vtt())
136
+ ```
137
+
138
+ ---
139
+
140
+ ## Real-Time Audio Streaming
141
+
142
+ Use `AudioStreamer` for streaming microphone feeds, audio generator streams, or WebSockets:
143
+
144
+ ```python
145
+ import mas_asr
146
+ from mas_asr import AudioStreamer
147
+
148
+ model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
149
+ streamer = AudioStreamer(model, language="hi", mode="native")
150
+
151
+ # Feed 16kHz audio chunks (torch.Tensor, numpy array, or 16-bit PCM bytes)
152
+ for pcm_chunk in audio_stream:
153
+ for event in streamer.feed(pcm_chunk):
154
+ if event.is_final:
155
+ print(f"\nFinal: {event.text}")
156
+ else:
157
+ print(f"Partial: {event.text}", end="\r")
158
+
159
+ # Flush remaining audio when stream ends
160
+ for event in streamer.flush():
161
+ print(f"\nFinal: {event.text}")
162
+ ```
163
+
164
+ ---
165
+
166
+ ## Command-Line Interface (CLI)
167
+
168
+ ```bash
169
+ # Basic transcription
170
+ mas-asr audio.wav --lang hi --mode native
171
+
172
+ # Auto-detect language and show detection confidence
173
+ mas-asr interview.mp3 --show-lang
174
+
175
+ # Transcribe long audio and export to SRT subtitles
176
+ mas-asr podcast.mp3 --output-format srt --output-dir ./subtitles/
177
+
178
+ # Output all 3 modes (native, mixed, romanized)
179
+ mas-asr clip.wav --all-modes
180
+
181
+ # Batch process multiple audio files
182
+ mas-asr ./recordings/*.wav --output-format json -o ./transcripts/
183
+ ```
184
+
185
+ ---
186
+
187
+ ## Fine-Tuning with LoRA / PEFT
188
+
189
+ Adapt the 1.2B FastConformer model on domain-specific Indian language datasets with consumer GPUs:
190
+
191
+ ```python
192
+ import mas_asr
193
+ from mas_asr.finetune import get_canary_lora_model, ASREngineTrainer
194
+
195
+ # Load base model
196
+ base_model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
197
+
198
+ # Wrap with LoRA adapters
199
+ lora_model = get_canary_lora_model(base_model.model, r=16, lora_alpha=32)
200
+
201
+ # Train using ASREngineTrainer
202
+ trainer = ASREngineTrainer(
203
+ model=lora_model,
204
+ fe=base_model.fe,
205
+ tokenizer=base_model.tokenizer,
206
+ train_dataset=my_hf_dataset,
207
+ batch_size=4,
208
+ learning_rate=1e-4,
209
+ )
210
+
211
+ trainer.train_epoch()
212
+ trainer.save_adapters("./indic_lora_adapters")
213
+ ```
214
+
215
+ See [examples/finetune_indic.py](file:///c:/Users/admin/Desktop/Projects/ITF/examples/finetune_indic.py) for a complete, runnable script.
216
+
217
+ ---
218
+
219
+ ## Supported Languages
220
+
221
+ | Code | Language | Code | Language | Code | Language |
222
+ |------|----------|------|----------|------|----------|
223
+ | `as` | Assamese | `gu` | Gujarati | `mr` | Marathi |
224
+ | `bn` | Bengali | `hi` | Hindi | `ne` | Nepali |
225
+ | `bho`| Bhojpuri | `kn` | Kannada | `or` | Odia |
226
+ | `doi`| Dogri | `kok`| Konkani | `pa` | Punjabi |
227
+ | `en` | English | `ml` | Malayalam| `ta` | Tamil |
228
+ | `ur` | Urdu | `mai`| Maithili | `te` | Telugu |
229
+
230
+ *(And additional scheduled regional languages: `bgc`, `bhb`, `brx`, `hne`, `ks`, `mni`, `sa`, `sat`, `sd`)*
231
+
232
+ ---
233
+
234
+ ## License
235
+
236
+ Apache License 2.0.
@@ -0,0 +1,214 @@
1
+ # mas-asr
2
+
3
+ **mas-asr** is a production-grade, modular Python speech recognition library inspired by `openai-whisper` and `faster-whisper`. Built with first-class support for Indic languages, offering transcription in **native scripts**, **mixed scripts (Hinglish/code-mixing)**, and **romanized transliteration**, powered by models like **`bodhan-ai/indic-transcribe-flex`** (1.2B FastConformer + Transformer decoder).
4
+
5
+ ---
6
+
7
+ ## Key Features
8
+
9
+ - **Whisper-like Intuitive API**: Load and transcribe with minimal code (`mas_asr.load_model(...)` & `model.transcribe(...)`).
10
+ - **3 Output Script Modes**:
11
+ - `native`: Standard native script (Devanagari, Gurmukhi, Bengali, Tamil, etc.).
12
+ - `mixed`: Code-mixing & loanwords in Latin script with Indic digits (e.g., modern conversational messaging).
13
+ - `romanized`: Full phonetic transliteration in Latin script.
14
+ - **27+ Indian Languages**: Hindi, Tamil, Telugu, Bengali, Kannada, Marathi, Malayalam, Punjabi, Gujarati, Odia, Urdu, Assamese, and more + English.
15
+ - **Auto Long-Form Audio Segmentation**: Automatically splits audio exceeding 30 seconds at natural pauses using energy-based VAD to prevent context collapse.
16
+ - **Real-Time Streaming**: VAD-buffered sliding window streaming processor for microphone input and WebSockets.
17
+ - **Built-in CLI**: Transcribe directly from terminal with SRT, VTT, JSON, and TXT export.
18
+ - **LoRA & PEFT Fine-Tuning**: Out-of-the-box support for low-rank adapter fine-tuning on domain-specific datasets.
19
+
20
+ ---
21
+
22
+ ## Installation
23
+
24
+ Using `uv`:
25
+
26
+ ```bash
27
+ uv add mas-asr
28
+ ```
29
+
30
+ Or using standard `pip`:
31
+
32
+ ```bash
33
+ pip install mas-asr
34
+ ```
35
+
36
+ To install fine-tuning dependencies:
37
+
38
+ ```bash
39
+ pip install "mas-asr[finetune]"
40
+ ```
41
+
42
+ ---
43
+
44
+ ## Quickstart (Python API)
45
+
46
+ ### Basic Transcription
47
+
48
+ ```python
49
+ import mas_asr
50
+
51
+ # Load model (auto-downloads from Hugging Face or loads local checkpoint directory)
52
+ model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex", device="cuda", compute_type="bfloat16")
53
+
54
+ # Transcribe an audio file
55
+ result = model.transcribe(
56
+ "speech.mp3",
57
+ language="hi", # or None for automatic language detection
58
+ mode="native", # "native", "mixed", or "romanized"
59
+ vad_filter=True, # Auto-chunks long audio > 30s at natural pauses
60
+ )
61
+
62
+ print(result.text)
63
+ print(f"Language: {result.language} (Confidence: {result.language_probability:.2f})")
64
+
65
+ # Timestamped segments
66
+ for seg in result.segments:
67
+ print(f"[{seg.start:.2f}s -> {seg.end:.2f}s] {seg.text}")
68
+ ```
69
+
70
+ ### High-Speed Inference with `torch.compile`
71
+
72
+ Compile the model encoder to reduce kernel overhead and achieve faster transcription:
73
+
74
+ ```python
75
+ # Enable torch.compile for reduced kernel launch overhead
76
+ model = mas_asr.load_model(
77
+ "bodhan-ai/indic-transcribe-flex",
78
+ device="cuda",
79
+ compute_type="bfloat16",
80
+ compile=True, # Compiles FastConformer encoder
81
+ )
82
+ ```
83
+
84
+ ### Loading Trained LoRA Adapters
85
+
86
+ ```python
87
+ # Load base model with a custom domain adapter attached
88
+ model = mas_asr.load_model(
89
+ "bodhan-ai/indic-transcribe-flex",
90
+ adapter_path="./my_domain_adapters",
91
+ )
92
+ ```
93
+
94
+ ### Multi-Mode Transcription in One Go
95
+
96
+ ```python
97
+ # Transcribe into Native, Mixed, and Romanized scripts simultaneously
98
+ outputs = model.all_modes("audio.wav", language="hi")
99
+ print("Native: ", outputs["native"])
100
+ print("Mixed: ", outputs["mixed"])
101
+ print("Romanized:", outputs["romanized"])
102
+ ```
103
+
104
+ ### Export Subtitles (SRT / VTT)
105
+
106
+ ```python
107
+ # Save SubRip Subtitles
108
+ with open("subtitles.srt", "w", encoding="utf-8") as f:
109
+ f.write(result.to_srt())
110
+
111
+ # Save WebVTT Subtitles
112
+ with open("subtitles.vtt", "w", encoding="utf-8") as f:
113
+ f.write(result.to_vtt())
114
+ ```
115
+
116
+ ---
117
+
118
+ ## Real-Time Audio Streaming
119
+
120
+ Use `AudioStreamer` for streaming microphone feeds, audio generator streams, or WebSockets:
121
+
122
+ ```python
123
+ import mas_asr
124
+ from mas_asr import AudioStreamer
125
+
126
+ model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
127
+ streamer = AudioStreamer(model, language="hi", mode="native")
128
+
129
+ # Feed 16kHz audio chunks (torch.Tensor, numpy array, or 16-bit PCM bytes)
130
+ for pcm_chunk in audio_stream:
131
+ for event in streamer.feed(pcm_chunk):
132
+ if event.is_final:
133
+ print(f"\nFinal: {event.text}")
134
+ else:
135
+ print(f"Partial: {event.text}", end="\r")
136
+
137
+ # Flush remaining audio when stream ends
138
+ for event in streamer.flush():
139
+ print(f"\nFinal: {event.text}")
140
+ ```
141
+
142
+ ---
143
+
144
+ ## Command-Line Interface (CLI)
145
+
146
+ ```bash
147
+ # Basic transcription
148
+ mas-asr audio.wav --lang hi --mode native
149
+
150
+ # Auto-detect language and show detection confidence
151
+ mas-asr interview.mp3 --show-lang
152
+
153
+ # Transcribe long audio and export to SRT subtitles
154
+ mas-asr podcast.mp3 --output-format srt --output-dir ./subtitles/
155
+
156
+ # Output all 3 modes (native, mixed, romanized)
157
+ mas-asr clip.wav --all-modes
158
+
159
+ # Batch process multiple audio files
160
+ mas-asr ./recordings/*.wav --output-format json -o ./transcripts/
161
+ ```
162
+
163
+ ---
164
+
165
+ ## Fine-Tuning with LoRA / PEFT
166
+
167
+ Adapt the 1.2B FastConformer model on domain-specific Indian language datasets with consumer GPUs:
168
+
169
+ ```python
170
+ import mas_asr
171
+ from mas_asr.finetune import get_canary_lora_model, ASREngineTrainer
172
+
173
+ # Load base model
174
+ base_model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
175
+
176
+ # Wrap with LoRA adapters
177
+ lora_model = get_canary_lora_model(base_model.model, r=16, lora_alpha=32)
178
+
179
+ # Train using ASREngineTrainer
180
+ trainer = ASREngineTrainer(
181
+ model=lora_model,
182
+ fe=base_model.fe,
183
+ tokenizer=base_model.tokenizer,
184
+ train_dataset=my_hf_dataset,
185
+ batch_size=4,
186
+ learning_rate=1e-4,
187
+ )
188
+
189
+ trainer.train_epoch()
190
+ trainer.save_adapters("./indic_lora_adapters")
191
+ ```
192
+
193
+ See [examples/finetune_indic.py](file:///c:/Users/admin/Desktop/Projects/ITF/examples/finetune_indic.py) for a complete, runnable script.
194
+
195
+ ---
196
+
197
+ ## Supported Languages
198
+
199
+ | Code | Language | Code | Language | Code | Language |
200
+ |------|----------|------|----------|------|----------|
201
+ | `as` | Assamese | `gu` | Gujarati | `mr` | Marathi |
202
+ | `bn` | Bengali | `hi` | Hindi | `ne` | Nepali |
203
+ | `bho`| Bhojpuri | `kn` | Kannada | `or` | Odia |
204
+ | `doi`| Dogri | `kok`| Konkani | `pa` | Punjabi |
205
+ | `en` | English | `ml` | Malayalam| `ta` | Tamil |
206
+ | `ur` | Urdu | `mai`| Maithili | `te` | Telugu |
207
+
208
+ *(And additional scheduled regional languages: `bgc`, `bhb`, `brx`, `hne`, `ks`, `mni`, `sa`, `sat`, `sd`)*
209
+
210
+ ---
211
+
212
+ ## License
213
+
214
+ Apache License 2.0.
@@ -0,0 +1,48 @@
1
+ [project]
2
+ name = "mas-asr"
3
+ version = "0.1.0"
4
+ description = "Production-grade, modular multilingual ASR library with native script, mixed script, streaming, and fine-tuning support"
5
+ readme = "README.md"
6
+ requires-python = ">=3.13"
7
+ dependencies = [
8
+ "huggingface-hub>=0.23.0",
9
+ "peft>=0.21.2",
10
+ "sentencepiece>=0.2.2",
11
+ "sounddevice>=0.5.6",
12
+ "soundfile>=0.14.0",
13
+ "torch>=2.14.1",
14
+ "torchaudio>=2.11.0",
15
+ "transformers>=5.19.0",
16
+ "triton-windows>=3.8.0.post29",
17
+ ]
18
+
19
+ [[project.authors]]
20
+ name = "ArYan9908"
21
+ email = "9908aryan@gmail.com"
22
+
23
+ [project.optional-dependencies]
24
+ finetune = [
25
+ "peft>=0.10.0",
26
+ "accelerate>=0.30.0",
27
+ "datasets>=2.19.0",
28
+ ]
29
+
30
+ [project.scripts]
31
+ mas-asr = "mas_asr.cli:main"
32
+
33
+ [[tool.uv.index]]
34
+ name = "pytorch-cu130"
35
+ url = "https://download.pytorch.org/whl/cu130"
36
+ explicit = true
37
+
38
+ [[tool.uv.sources.torch]]
39
+ index = "pytorch-cu130"
40
+ marker = "sys_platform == 'linux' or sys_platform == 'win32'"
41
+
42
+ [[tool.uv.sources.torchaudio]]
43
+ index = "pytorch-cu130"
44
+ marker = "sys_platform == 'linux' or sys_platform == 'win32'"
45
+
46
+ [build-system]
47
+ requires = ["uv_build>=0.12.21,<0.13.0"]
48
+ build-backend = "uv_build"
@@ -0,0 +1,47 @@
1
+ [project]
2
+ name = "mas-asr"
3
+ version = "0.1.0"
4
+ description = "Production-grade, modular multilingual ASR library with native script, mixed script, streaming, and fine-tuning support"
5
+ readme = "README.md"
6
+ authors = [
7
+ { name = "ArYan9908", email = "9908aryan@gmail.com" }
8
+ ]
9
+ requires-python = ">=3.13"
10
+ dependencies = [
11
+ "huggingface-hub>=0.23.0",
12
+ "peft>=0.21.2",
13
+ "sentencepiece>=0.2.2",
14
+ "sounddevice>=0.5.6",
15
+ "soundfile>=0.14.0",
16
+ "torch>=2.14.1",
17
+ "torchaudio>=2.11.0",
18
+ "transformers>=5.19.0",
19
+ "triton-windows>=3.8.0.post29",
20
+ ]
21
+
22
+ [project.optional-dependencies]
23
+ finetune = [
24
+ "peft>=0.10.0",
25
+ "accelerate>=0.30.0",
26
+ "datasets>=2.19.0",
27
+ ]
28
+
29
+ [[tool.uv.index]]
30
+ name = "pytorch-cu130"
31
+ url = "https://download.pytorch.org/whl/cu130"
32
+ explicit = true
33
+
34
+ [tool.uv.sources]
35
+ torch = [
36
+ { index = "pytorch-cu130", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
37
+ ]
38
+ torchaudio = [
39
+ { index = "pytorch-cu130", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
40
+ ]
41
+
42
+ [project.scripts]
43
+ mas-asr = "mas_asr.cli:main"
44
+
45
+ [build-system]
46
+ requires = ["uv_build>=0.12.21,<0.13.0"]
47
+ build-backend = "uv_build"
@@ -0,0 +1,29 @@
1
+ """mas-asr: Production-grade, modular multilingual ASR library with native and mixed script support."""
2
+
3
+ from mas_asr.audio import get_audio_duration, load_audio, segment_audio
4
+ from mas_asr.model import load_model, register_model_backend
5
+ from mas_asr.models.indic_canary.lid import RECOMMENDED_LANGS, TRAINED_LANGS
6
+ from mas_asr.models.indic_canary.model import MODES
7
+ from mas_asr.streaming import AudioStreamer, StreamChunkEvent, stream_from_microphone
8
+ from mas_asr.types import LanguageInfo, OutputMode, Segment, TranscriptionResult
9
+
10
+ __version__ = "0.1.0"
11
+
12
+ __all__ = [
13
+ "MODES",
14
+ "RECOMMENDED_LANGS",
15
+ "TRAINED_LANGS",
16
+ "AudioStreamer",
17
+ "LanguageInfo",
18
+ "OutputMode",
19
+ "Segment",
20
+ "StreamChunkEvent",
21
+ "TranscriptionResult",
22
+ "__version__",
23
+ "get_audio_duration",
24
+ "load_audio",
25
+ "load_model",
26
+ "register_model_backend",
27
+ "segment_audio",
28
+ "stream_from_microphone",
29
+ ]