mas-asr 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mas_asr-0.1.0/PKG-INFO +236 -0
- mas_asr-0.1.0/README.md +214 -0
- mas_asr-0.1.0/pyproject.toml +48 -0
- mas_asr-0.1.0/pyproject.toml.orig +47 -0
- mas_asr-0.1.0/src/mas_asr/__init__.py +29 -0
- mas_asr-0.1.0/src/mas_asr/audio.py +137 -0
- mas_asr-0.1.0/src/mas_asr/cli.py +199 -0
- mas_asr-0.1.0/src/mas_asr/finetune/__init__.py +11 -0
- mas_asr-0.1.0/src/mas_asr/finetune/dataset.py +94 -0
- mas_asr-0.1.0/src/mas_asr/finetune/lora.py +47 -0
- mas_asr-0.1.0/src/mas_asr/finetune/trainer.py +68 -0
- mas_asr-0.1.0/src/mas_asr/model.py +101 -0
- mas_asr-0.1.0/src/mas_asr/models/__init__.py +9 -0
- mas_asr-0.1.0/src/mas_asr/models/base.py +76 -0
- mas_asr-0.1.0/src/mas_asr/models/indic_canary/__init__.py +13 -0
- mas_asr-0.1.0/src/mas_asr/models/indic_canary/configuration.py +57 -0
- mas_asr-0.1.0/src/mas_asr/models/indic_canary/feature_extraction.py +121 -0
- mas_asr-0.1.0/src/mas_asr/models/indic_canary/lid.py +88 -0
- mas_asr-0.1.0/src/mas_asr/models/indic_canary/model.py +443 -0
- mas_asr-0.1.0/src/mas_asr/models/indic_canary/modeling.py +604 -0
- mas_asr-0.1.0/src/mas_asr/models/indic_canary/tokenization.py +174 -0
- mas_asr-0.1.0/src/mas_asr/streaming.py +299 -0
- mas_asr-0.1.0/src/mas_asr/types.py +99 -0
mas_asr-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: mas-asr
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Production-grade, modular multilingual ASR library with native script, mixed script, streaming, and fine-tuning support
|
|
5
|
+
Author: ArYan9908
|
|
6
|
+
Author-email: ArYan9908 <9908aryan@gmail.com>
|
|
7
|
+
Requires-Dist: huggingface-hub>=0.23.0
|
|
8
|
+
Requires-Dist: peft>=0.21.2
|
|
9
|
+
Requires-Dist: sentencepiece>=0.2.2
|
|
10
|
+
Requires-Dist: sounddevice>=0.5.6
|
|
11
|
+
Requires-Dist: soundfile>=0.14.0
|
|
12
|
+
Requires-Dist: torch>=2.14.1
|
|
13
|
+
Requires-Dist: torchaudio>=2.11.0
|
|
14
|
+
Requires-Dist: transformers>=5.19.0
|
|
15
|
+
Requires-Dist: triton-windows>=3.8.0.post29
|
|
16
|
+
Requires-Dist: peft>=0.10.0 ; extra == 'finetune'
|
|
17
|
+
Requires-Dist: accelerate>=0.30.0 ; extra == 'finetune'
|
|
18
|
+
Requires-Dist: datasets>=2.19.0 ; extra == 'finetune'
|
|
19
|
+
Requires-Python: >=3.13
|
|
20
|
+
Provides-Extra: finetune
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
|
|
23
|
+
# mas-asr
|
|
24
|
+
|
|
25
|
+
**mas-asr** is a production-grade, modular Python speech recognition library inspired by `openai-whisper` and `faster-whisper`. Built with first-class support for Indic languages, offering transcription in **native scripts**, **mixed scripts (Hinglish/code-mixing)**, and **romanized transliteration**, powered by models like **`bodhan-ai/indic-transcribe-flex`** (1.2B FastConformer + Transformer decoder).
|
|
26
|
+
|
|
27
|
+
---
|
|
28
|
+
|
|
29
|
+
## Key Features
|
|
30
|
+
|
|
31
|
+
- **Whisper-like Intuitive API**: Load and transcribe with minimal code (`mas_asr.load_model(...)` & `model.transcribe(...)`).
|
|
32
|
+
- **3 Output Script Modes**:
|
|
33
|
+
- `native`: Standard native script (Devanagari, Gurmukhi, Bengali, Tamil, etc.).
|
|
34
|
+
- `mixed`: Code-mixing & loanwords in Latin script with Indic digits (e.g., modern conversational messaging).
|
|
35
|
+
- `romanized`: Full phonetic transliteration in Latin script.
|
|
36
|
+
- **27+ Indian Languages**: Hindi, Tamil, Telugu, Bengali, Kannada, Marathi, Malayalam, Punjabi, Gujarati, Odia, Urdu, Assamese, and more + English.
|
|
37
|
+
- **Auto Long-Form Audio Segmentation**: Automatically splits audio exceeding 30 seconds at natural pauses using energy-based VAD to prevent context collapse.
|
|
38
|
+
- **Real-Time Streaming**: VAD-buffered sliding window streaming processor for microphone input and WebSockets.
|
|
39
|
+
- **Built-in CLI**: Transcribe directly from terminal with SRT, VTT, JSON, and TXT export.
|
|
40
|
+
- **LoRA & PEFT Fine-Tuning**: Out-of-the-box support for low-rank adapter fine-tuning on domain-specific datasets.
|
|
41
|
+
|
|
42
|
+
---
|
|
43
|
+
|
|
44
|
+
## Installation
|
|
45
|
+
|
|
46
|
+
Using `uv`:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
uv add mas-asr
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Or using standard `pip`:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install mas-asr
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
To install fine-tuning dependencies:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install "mas-asr[finetune]"
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
---
|
|
65
|
+
|
|
66
|
+
## Quickstart (Python API)
|
|
67
|
+
|
|
68
|
+
### Basic Transcription
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
import mas_asr
|
|
72
|
+
|
|
73
|
+
# Load model (auto-downloads from Hugging Face or loads local checkpoint directory)
|
|
74
|
+
model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex", device="cuda", compute_type="bfloat16")
|
|
75
|
+
|
|
76
|
+
# Transcribe an audio file
|
|
77
|
+
result = model.transcribe(
|
|
78
|
+
"speech.mp3",
|
|
79
|
+
language="hi", # or None for automatic language detection
|
|
80
|
+
mode="native", # "native", "mixed", or "romanized"
|
|
81
|
+
vad_filter=True, # Auto-chunks long audio > 30s at natural pauses
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
print(result.text)
|
|
85
|
+
print(f"Language: {result.language} (Confidence: {result.language_probability:.2f})")
|
|
86
|
+
|
|
87
|
+
# Timestamped segments
|
|
88
|
+
for seg in result.segments:
|
|
89
|
+
print(f"[{seg.start:.2f}s -> {seg.end:.2f}s] {seg.text}")
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
### High-Speed Inference with `torch.compile`
|
|
93
|
+
|
|
94
|
+
Compile the model encoder to reduce kernel overhead and achieve faster transcription:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
# Enable torch.compile for reduced kernel launch overhead
|
|
98
|
+
model = mas_asr.load_model(
|
|
99
|
+
"bodhan-ai/indic-transcribe-flex",
|
|
100
|
+
device="cuda",
|
|
101
|
+
compute_type="bfloat16",
|
|
102
|
+
compile=True, # Compiles FastConformer encoder
|
|
103
|
+
)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### Loading Trained LoRA Adapters
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
# Load base model with a custom domain adapter attached
|
|
110
|
+
model = mas_asr.load_model(
|
|
111
|
+
"bodhan-ai/indic-transcribe-flex",
|
|
112
|
+
adapter_path="./my_domain_adapters",
|
|
113
|
+
)
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
### Multi-Mode Transcription in One Go
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
# Transcribe into Native, Mixed, and Romanized scripts simultaneously
|
|
120
|
+
outputs = model.all_modes("audio.wav", language="hi")
|
|
121
|
+
print("Native: ", outputs["native"])
|
|
122
|
+
print("Mixed: ", outputs["mixed"])
|
|
123
|
+
print("Romanized:", outputs["romanized"])
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
### Export Subtitles (SRT / VTT)
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
# Save SubRip Subtitles
|
|
130
|
+
with open("subtitles.srt", "w", encoding="utf-8") as f:
|
|
131
|
+
f.write(result.to_srt())
|
|
132
|
+
|
|
133
|
+
# Save WebVTT Subtitles
|
|
134
|
+
with open("subtitles.vtt", "w", encoding="utf-8") as f:
|
|
135
|
+
f.write(result.to_vtt())
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
---
|
|
139
|
+
|
|
140
|
+
## Real-Time Audio Streaming
|
|
141
|
+
|
|
142
|
+
Use `AudioStreamer` for streaming microphone feeds, audio generator streams, or WebSockets:
|
|
143
|
+
|
|
144
|
+
```python
|
|
145
|
+
import mas_asr
|
|
146
|
+
from mas_asr import AudioStreamer
|
|
147
|
+
|
|
148
|
+
model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
|
|
149
|
+
streamer = AudioStreamer(model, language="hi", mode="native")
|
|
150
|
+
|
|
151
|
+
# Feed 16kHz audio chunks (torch.Tensor, numpy array, or 16-bit PCM bytes)
|
|
152
|
+
for pcm_chunk in audio_stream:
|
|
153
|
+
for event in streamer.feed(pcm_chunk):
|
|
154
|
+
if event.is_final:
|
|
155
|
+
print(f"\nFinal: {event.text}")
|
|
156
|
+
else:
|
|
157
|
+
print(f"Partial: {event.text}", end="\r")
|
|
158
|
+
|
|
159
|
+
# Flush remaining audio when stream ends
|
|
160
|
+
for event in streamer.flush():
|
|
161
|
+
print(f"\nFinal: {event.text}")
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
---
|
|
165
|
+
|
|
166
|
+
## Command-Line Interface (CLI)
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
# Basic transcription
|
|
170
|
+
mas-asr audio.wav --lang hi --mode native
|
|
171
|
+
|
|
172
|
+
# Auto-detect language and show detection confidence
|
|
173
|
+
mas-asr interview.mp3 --show-lang
|
|
174
|
+
|
|
175
|
+
# Transcribe long audio and export to SRT subtitles
|
|
176
|
+
mas-asr podcast.mp3 --output-format srt --output-dir ./subtitles/
|
|
177
|
+
|
|
178
|
+
# Output all 3 modes (native, mixed, romanized)
|
|
179
|
+
mas-asr clip.wav --all-modes
|
|
180
|
+
|
|
181
|
+
# Batch process multiple audio files
|
|
182
|
+
mas-asr ./recordings/*.wav --output-format json -o ./transcripts/
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
---
|
|
186
|
+
|
|
187
|
+
## Fine-Tuning with LoRA / PEFT
|
|
188
|
+
|
|
189
|
+
Adapt the 1.2B FastConformer model on domain-specific Indian language datasets with consumer GPUs:
|
|
190
|
+
|
|
191
|
+
```python
|
|
192
|
+
import mas_asr
|
|
193
|
+
from mas_asr.finetune import get_canary_lora_model, ASREngineTrainer
|
|
194
|
+
|
|
195
|
+
# Load base model
|
|
196
|
+
base_model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
|
|
197
|
+
|
|
198
|
+
# Wrap with LoRA adapters
|
|
199
|
+
lora_model = get_canary_lora_model(base_model.model, r=16, lora_alpha=32)
|
|
200
|
+
|
|
201
|
+
# Train using ASREngineTrainer
|
|
202
|
+
trainer = ASREngineTrainer(
|
|
203
|
+
model=lora_model,
|
|
204
|
+
fe=base_model.fe,
|
|
205
|
+
tokenizer=base_model.tokenizer,
|
|
206
|
+
train_dataset=my_hf_dataset,
|
|
207
|
+
batch_size=4,
|
|
208
|
+
learning_rate=1e-4,
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
trainer.train_epoch()
|
|
212
|
+
trainer.save_adapters("./indic_lora_adapters")
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
See [examples/finetune_indic.py](file:///c:/Users/admin/Desktop/Projects/ITF/examples/finetune_indic.py) for a complete, runnable script.
|
|
216
|
+
|
|
217
|
+
---
|
|
218
|
+
|
|
219
|
+
## Supported Languages
|
|
220
|
+
|
|
221
|
+
| Code | Language | Code | Language | Code | Language |
|
|
222
|
+
|------|----------|------|----------|------|----------|
|
|
223
|
+
| `as` | Assamese | `gu` | Gujarati | `mr` | Marathi |
|
|
224
|
+
| `bn` | Bengali | `hi` | Hindi | `ne` | Nepali |
|
|
225
|
+
| `bho`| Bhojpuri | `kn` | Kannada | `or` | Odia |
|
|
226
|
+
| `doi`| Dogri | `kok`| Konkani | `pa` | Punjabi |
|
|
227
|
+
| `en` | English | `ml` | Malayalam| `ta` | Tamil |
|
|
228
|
+
| `ur` | Urdu | `mai`| Maithili | `te` | Telugu |
|
|
229
|
+
|
|
230
|
+
*(And additional scheduled regional languages: `bgc`, `bhb`, `brx`, `hne`, `ks`, `mni`, `sa`, `sat`, `sd`)*
|
|
231
|
+
|
|
232
|
+
---
|
|
233
|
+
|
|
234
|
+
## License
|
|
235
|
+
|
|
236
|
+
Apache License 2.0.
|
mas_asr-0.1.0/README.md
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
# mas-asr
|
|
2
|
+
|
|
3
|
+
**mas-asr** is a production-grade, modular Python speech recognition library inspired by `openai-whisper` and `faster-whisper`. Built with first-class support for Indic languages, offering transcription in **native scripts**, **mixed scripts (Hinglish/code-mixing)**, and **romanized transliteration**, powered by models like **`bodhan-ai/indic-transcribe-flex`** (1.2B FastConformer + Transformer decoder).
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## Key Features
|
|
8
|
+
|
|
9
|
+
- **Whisper-like Intuitive API**: Load and transcribe with minimal code (`mas_asr.load_model(...)` & `model.transcribe(...)`).
|
|
10
|
+
- **3 Output Script Modes**:
|
|
11
|
+
- `native`: Standard native script (Devanagari, Gurmukhi, Bengali, Tamil, etc.).
|
|
12
|
+
- `mixed`: Code-mixing & loanwords in Latin script with Indic digits (e.g., modern conversational messaging).
|
|
13
|
+
- `romanized`: Full phonetic transliteration in Latin script.
|
|
14
|
+
- **27+ Indian Languages**: Hindi, Tamil, Telugu, Bengali, Kannada, Marathi, Malayalam, Punjabi, Gujarati, Odia, Urdu, Assamese, and more + English.
|
|
15
|
+
- **Auto Long-Form Audio Segmentation**: Automatically splits audio exceeding 30 seconds at natural pauses using energy-based VAD to prevent context collapse.
|
|
16
|
+
- **Real-Time Streaming**: VAD-buffered sliding window streaming processor for microphone input and WebSockets.
|
|
17
|
+
- **Built-in CLI**: Transcribe directly from terminal with SRT, VTT, JSON, and TXT export.
|
|
18
|
+
- **LoRA & PEFT Fine-Tuning**: Out-of-the-box support for low-rank adapter fine-tuning on domain-specific datasets.
|
|
19
|
+
|
|
20
|
+
---
|
|
21
|
+
|
|
22
|
+
## Installation
|
|
23
|
+
|
|
24
|
+
Using `uv`:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
uv add mas-asr
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Or using standard `pip`:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install mas-asr
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
To install fine-tuning dependencies:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pip install "mas-asr[finetune]"
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
---
|
|
43
|
+
|
|
44
|
+
## Quickstart (Python API)
|
|
45
|
+
|
|
46
|
+
### Basic Transcription
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
import mas_asr
|
|
50
|
+
|
|
51
|
+
# Load model (auto-downloads from Hugging Face or loads local checkpoint directory)
|
|
52
|
+
model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex", device="cuda", compute_type="bfloat16")
|
|
53
|
+
|
|
54
|
+
# Transcribe an audio file
|
|
55
|
+
result = model.transcribe(
|
|
56
|
+
"speech.mp3",
|
|
57
|
+
language="hi", # or None for automatic language detection
|
|
58
|
+
mode="native", # "native", "mixed", or "romanized"
|
|
59
|
+
vad_filter=True, # Auto-chunks long audio > 30s at natural pauses
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
print(result.text)
|
|
63
|
+
print(f"Language: {result.language} (Confidence: {result.language_probability:.2f})")
|
|
64
|
+
|
|
65
|
+
# Timestamped segments
|
|
66
|
+
for seg in result.segments:
|
|
67
|
+
print(f"[{seg.start:.2f}s -> {seg.end:.2f}s] {seg.text}")
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### High-Speed Inference with `torch.compile`
|
|
71
|
+
|
|
72
|
+
Compile the model encoder to reduce kernel overhead and achieve faster transcription:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
# Enable torch.compile for reduced kernel launch overhead
|
|
76
|
+
model = mas_asr.load_model(
|
|
77
|
+
"bodhan-ai/indic-transcribe-flex",
|
|
78
|
+
device="cuda",
|
|
79
|
+
compute_type="bfloat16",
|
|
80
|
+
compile=True, # Compiles FastConformer encoder
|
|
81
|
+
)
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
### Loading Trained LoRA Adapters
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
# Load base model with a custom domain adapter attached
|
|
88
|
+
model = mas_asr.load_model(
|
|
89
|
+
"bodhan-ai/indic-transcribe-flex",
|
|
90
|
+
adapter_path="./my_domain_adapters",
|
|
91
|
+
)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
### Multi-Mode Transcription in One Go
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
# Transcribe into Native, Mixed, and Romanized scripts simultaneously
|
|
98
|
+
outputs = model.all_modes("audio.wav", language="hi")
|
|
99
|
+
print("Native: ", outputs["native"])
|
|
100
|
+
print("Mixed: ", outputs["mixed"])
|
|
101
|
+
print("Romanized:", outputs["romanized"])
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
### Export Subtitles (SRT / VTT)
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
# Save SubRip Subtitles
|
|
108
|
+
with open("subtitles.srt", "w", encoding="utf-8") as f:
|
|
109
|
+
f.write(result.to_srt())
|
|
110
|
+
|
|
111
|
+
# Save WebVTT Subtitles
|
|
112
|
+
with open("subtitles.vtt", "w", encoding="utf-8") as f:
|
|
113
|
+
f.write(result.to_vtt())
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
---
|
|
117
|
+
|
|
118
|
+
## Real-Time Audio Streaming
|
|
119
|
+
|
|
120
|
+
Use `AudioStreamer` for streaming microphone feeds, audio generator streams, or WebSockets:
|
|
121
|
+
|
|
122
|
+
```python
|
|
123
|
+
import mas_asr
|
|
124
|
+
from mas_asr import AudioStreamer
|
|
125
|
+
|
|
126
|
+
model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
|
|
127
|
+
streamer = AudioStreamer(model, language="hi", mode="native")
|
|
128
|
+
|
|
129
|
+
# Feed 16kHz audio chunks (torch.Tensor, numpy array, or 16-bit PCM bytes)
|
|
130
|
+
for pcm_chunk in audio_stream:
|
|
131
|
+
for event in streamer.feed(pcm_chunk):
|
|
132
|
+
if event.is_final:
|
|
133
|
+
print(f"\nFinal: {event.text}")
|
|
134
|
+
else:
|
|
135
|
+
print(f"Partial: {event.text}", end="\r")
|
|
136
|
+
|
|
137
|
+
# Flush remaining audio when stream ends
|
|
138
|
+
for event in streamer.flush():
|
|
139
|
+
print(f"\nFinal: {event.text}")
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
## Command-Line Interface (CLI)
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
# Basic transcription
|
|
148
|
+
mas-asr audio.wav --lang hi --mode native
|
|
149
|
+
|
|
150
|
+
# Auto-detect language and show detection confidence
|
|
151
|
+
mas-asr interview.mp3 --show-lang
|
|
152
|
+
|
|
153
|
+
# Transcribe long audio and export to SRT subtitles
|
|
154
|
+
mas-asr podcast.mp3 --output-format srt --output-dir ./subtitles/
|
|
155
|
+
|
|
156
|
+
# Output all 3 modes (native, mixed, romanized)
|
|
157
|
+
mas-asr clip.wav --all-modes
|
|
158
|
+
|
|
159
|
+
# Batch process multiple audio files
|
|
160
|
+
mas-asr ./recordings/*.wav --output-format json -o ./transcripts/
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
---
|
|
164
|
+
|
|
165
|
+
## Fine-Tuning with LoRA / PEFT
|
|
166
|
+
|
|
167
|
+
Adapt the 1.2B FastConformer model on domain-specific Indian language datasets with consumer GPUs:
|
|
168
|
+
|
|
169
|
+
```python
|
|
170
|
+
import mas_asr
|
|
171
|
+
from mas_asr.finetune import get_canary_lora_model, ASREngineTrainer
|
|
172
|
+
|
|
173
|
+
# Load base model
|
|
174
|
+
base_model = mas_asr.load_model("bodhan-ai/indic-transcribe-flex")
|
|
175
|
+
|
|
176
|
+
# Wrap with LoRA adapters
|
|
177
|
+
lora_model = get_canary_lora_model(base_model.model, r=16, lora_alpha=32)
|
|
178
|
+
|
|
179
|
+
# Train using ASREngineTrainer
|
|
180
|
+
trainer = ASREngineTrainer(
|
|
181
|
+
model=lora_model,
|
|
182
|
+
fe=base_model.fe,
|
|
183
|
+
tokenizer=base_model.tokenizer,
|
|
184
|
+
train_dataset=my_hf_dataset,
|
|
185
|
+
batch_size=4,
|
|
186
|
+
learning_rate=1e-4,
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
trainer.train_epoch()
|
|
190
|
+
trainer.save_adapters("./indic_lora_adapters")
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
See [examples/finetune_indic.py](file:///c:/Users/admin/Desktop/Projects/ITF/examples/finetune_indic.py) for a complete, runnable script.
|
|
194
|
+
|
|
195
|
+
---
|
|
196
|
+
|
|
197
|
+
## Supported Languages
|
|
198
|
+
|
|
199
|
+
| Code | Language | Code | Language | Code | Language |
|
|
200
|
+
|------|----------|------|----------|------|----------|
|
|
201
|
+
| `as` | Assamese | `gu` | Gujarati | `mr` | Marathi |
|
|
202
|
+
| `bn` | Bengali | `hi` | Hindi | `ne` | Nepali |
|
|
203
|
+
| `bho`| Bhojpuri | `kn` | Kannada | `or` | Odia |
|
|
204
|
+
| `doi`| Dogri | `kok`| Konkani | `pa` | Punjabi |
|
|
205
|
+
| `en` | English | `ml` | Malayalam| `ta` | Tamil |
|
|
206
|
+
| `ur` | Urdu | `mai`| Maithili | `te` | Telugu |
|
|
207
|
+
|
|
208
|
+
*(And additional scheduled regional languages: `bgc`, `bhb`, `brx`, `hne`, `ks`, `mni`, `sa`, `sat`, `sd`)*
|
|
209
|
+
|
|
210
|
+
---
|
|
211
|
+
|
|
212
|
+
## License
|
|
213
|
+
|
|
214
|
+
Apache License 2.0.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "mas-asr"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Production-grade, modular multilingual ASR library with native script, mixed script, streaming, and fine-tuning support"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.13"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"huggingface-hub>=0.23.0",
|
|
9
|
+
"peft>=0.21.2",
|
|
10
|
+
"sentencepiece>=0.2.2",
|
|
11
|
+
"sounddevice>=0.5.6",
|
|
12
|
+
"soundfile>=0.14.0",
|
|
13
|
+
"torch>=2.14.1",
|
|
14
|
+
"torchaudio>=2.11.0",
|
|
15
|
+
"transformers>=5.19.0",
|
|
16
|
+
"triton-windows>=3.8.0.post29",
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
[[project.authors]]
|
|
20
|
+
name = "ArYan9908"
|
|
21
|
+
email = "9908aryan@gmail.com"
|
|
22
|
+
|
|
23
|
+
[project.optional-dependencies]
|
|
24
|
+
finetune = [
|
|
25
|
+
"peft>=0.10.0",
|
|
26
|
+
"accelerate>=0.30.0",
|
|
27
|
+
"datasets>=2.19.0",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[project.scripts]
|
|
31
|
+
mas-asr = "mas_asr.cli:main"
|
|
32
|
+
|
|
33
|
+
[[tool.uv.index]]
|
|
34
|
+
name = "pytorch-cu130"
|
|
35
|
+
url = "https://download.pytorch.org/whl/cu130"
|
|
36
|
+
explicit = true
|
|
37
|
+
|
|
38
|
+
[[tool.uv.sources.torch]]
|
|
39
|
+
index = "pytorch-cu130"
|
|
40
|
+
marker = "sys_platform == 'linux' or sys_platform == 'win32'"
|
|
41
|
+
|
|
42
|
+
[[tool.uv.sources.torchaudio]]
|
|
43
|
+
index = "pytorch-cu130"
|
|
44
|
+
marker = "sys_platform == 'linux' or sys_platform == 'win32'"
|
|
45
|
+
|
|
46
|
+
[build-system]
|
|
47
|
+
requires = ["uv_build>=0.12.21,<0.13.0"]
|
|
48
|
+
build-backend = "uv_build"
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "mas-asr"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Production-grade, modular multilingual ASR library with native script, mixed script, streaming, and fine-tuning support"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
authors = [
|
|
7
|
+
{ name = "ArYan9908", email = "9908aryan@gmail.com" }
|
|
8
|
+
]
|
|
9
|
+
requires-python = ">=3.13"
|
|
10
|
+
dependencies = [
|
|
11
|
+
"huggingface-hub>=0.23.0",
|
|
12
|
+
"peft>=0.21.2",
|
|
13
|
+
"sentencepiece>=0.2.2",
|
|
14
|
+
"sounddevice>=0.5.6",
|
|
15
|
+
"soundfile>=0.14.0",
|
|
16
|
+
"torch>=2.14.1",
|
|
17
|
+
"torchaudio>=2.11.0",
|
|
18
|
+
"transformers>=5.19.0",
|
|
19
|
+
"triton-windows>=3.8.0.post29",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
finetune = [
|
|
24
|
+
"peft>=0.10.0",
|
|
25
|
+
"accelerate>=0.30.0",
|
|
26
|
+
"datasets>=2.19.0",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[[tool.uv.index]]
|
|
30
|
+
name = "pytorch-cu130"
|
|
31
|
+
url = "https://download.pytorch.org/whl/cu130"
|
|
32
|
+
explicit = true
|
|
33
|
+
|
|
34
|
+
[tool.uv.sources]
|
|
35
|
+
torch = [
|
|
36
|
+
{ index = "pytorch-cu130", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
|
|
37
|
+
]
|
|
38
|
+
torchaudio = [
|
|
39
|
+
{ index = "pytorch-cu130", marker = "sys_platform == 'linux' or sys_platform == 'win32'" },
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
mas-asr = "mas_asr.cli:main"
|
|
44
|
+
|
|
45
|
+
[build-system]
|
|
46
|
+
requires = ["uv_build>=0.12.21,<0.13.0"]
|
|
47
|
+
build-backend = "uv_build"
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""mas-asr: Production-grade, modular multilingual ASR library with native and mixed script support."""
|
|
2
|
+
|
|
3
|
+
from mas_asr.audio import get_audio_duration, load_audio, segment_audio
|
|
4
|
+
from mas_asr.model import load_model, register_model_backend
|
|
5
|
+
from mas_asr.models.indic_canary.lid import RECOMMENDED_LANGS, TRAINED_LANGS
|
|
6
|
+
from mas_asr.models.indic_canary.model import MODES
|
|
7
|
+
from mas_asr.streaming import AudioStreamer, StreamChunkEvent, stream_from_microphone
|
|
8
|
+
from mas_asr.types import LanguageInfo, OutputMode, Segment, TranscriptionResult
|
|
9
|
+
|
|
10
|
+
__version__ = "0.1.0"
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"MODES",
|
|
14
|
+
"RECOMMENDED_LANGS",
|
|
15
|
+
"TRAINED_LANGS",
|
|
16
|
+
"AudioStreamer",
|
|
17
|
+
"LanguageInfo",
|
|
18
|
+
"OutputMode",
|
|
19
|
+
"Segment",
|
|
20
|
+
"StreamChunkEvent",
|
|
21
|
+
"TranscriptionResult",
|
|
22
|
+
"__version__",
|
|
23
|
+
"get_audio_duration",
|
|
24
|
+
"load_audio",
|
|
25
|
+
"load_model",
|
|
26
|
+
"register_model_backend",
|
|
27
|
+
"segment_audio",
|
|
28
|
+
"stream_from_microphone",
|
|
29
|
+
]
|