omnius 1.0.623 → 1.0.625
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +8105 -7012
- package/dist/scripts/live-voxtral.py +276 -0
- package/dist/update-worker.js +103 -0
- package/docs/DISCOVERY.json +524 -12
- package/docs/DISCOVERY.md +7 -2
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""CUDA-only Voxtral ASR worker managed by Omnius."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import platform
|
|
10
|
+
import sys
|
|
11
|
+
import time
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
MODELS = {
|
|
16
|
+
"voxtral-mini-4b-realtime-2602": {
|
|
17
|
+
"upstream": "mistralai/Voxtral-Mini-4B-Realtime-2602",
|
|
18
|
+
"realtime": True,
|
|
19
|
+
"minimum_vram_gb": 15.0,
|
|
20
|
+
},
|
|
21
|
+
"voxtral-mini-3b-2507": {
|
|
22
|
+
"upstream": "mistralai/Voxtral-Mini-3B-2507",
|
|
23
|
+
"realtime": False,
|
|
24
|
+
"minimum_vram_gb": 9.0,
|
|
25
|
+
},
|
|
26
|
+
"voxtral-small-24b-2507": {
|
|
27
|
+
"upstream": "mistralai/Voxtral-Small-24B-2507",
|
|
28
|
+
"realtime": False,
|
|
29
|
+
"minimum_vram_gb": 54.0,
|
|
30
|
+
},
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def emit(event: dict[str, Any]) -> None:
|
|
35
|
+
sys.stdout.write(json.dumps(event, ensure_ascii=True) + "\n")
|
|
36
|
+
sys.stdout.flush()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def error(message: str) -> None:
|
|
40
|
+
emit({"type": "error", "message": message})
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def model_path(model_id: str) -> Path:
|
|
44
|
+
root = Path(
|
|
45
|
+
os.environ.get(
|
|
46
|
+
"OMNIUS_ASR_MODEL_DIR",
|
|
47
|
+
str(Path.home() / ".omnius" / "models" / "asr" / "voxtral-transformers"),
|
|
48
|
+
)
|
|
49
|
+
)
|
|
50
|
+
return root / model_id
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def select_device(torch: Any, minimum_vram_gb: float) -> tuple[Any, Any, dict[str, Any]]:
|
|
54
|
+
allow_cpu = os.environ.get("OMNIUS_ASR_ALLOW_CPU", "").lower() in (
|
|
55
|
+
"1",
|
|
56
|
+
"true",
|
|
57
|
+
"yes",
|
|
58
|
+
"on",
|
|
59
|
+
)
|
|
60
|
+
if not torch.cuda.is_available():
|
|
61
|
+
if allow_cpu:
|
|
62
|
+
return torch.device("cpu"), torch.float32, {"device": "cpu"}
|
|
63
|
+
raise RuntimeError(
|
|
64
|
+
"CUDA is required for Voxtral ASR; OMNIUS_ASR_ALLOW_CPU=1 is diagnostic only"
|
|
65
|
+
)
|
|
66
|
+
index = int(os.environ.get("OMNIUS_ASR_CUDA_DEVICE", "0") or "0")
|
|
67
|
+
if index < 0 or index >= torch.cuda.device_count():
|
|
68
|
+
raise RuntimeError(
|
|
69
|
+
f"CUDA device {index} is outside the visible range (count={torch.cuda.device_count()})"
|
|
70
|
+
)
|
|
71
|
+
torch.cuda.set_device(index)
|
|
72
|
+
props = torch.cuda.get_device_properties(index)
|
|
73
|
+
total_gb = props.total_memory / 1024**3
|
|
74
|
+
free_bytes, _ = torch.cuda.mem_get_info(index)
|
|
75
|
+
if total_gb < minimum_vram_gb:
|
|
76
|
+
raise RuntimeError(
|
|
77
|
+
f"{props.name} has {total_gb:.1f} GB VRAM; this Voxtral tier requires approximately {minimum_vram_gb:.0f} GB"
|
|
78
|
+
)
|
|
79
|
+
dtype = torch.bfloat16 if torch.cuda.is_bf16_supported() else torch.float16
|
|
80
|
+
device = torch.device(f"cuda:{index}")
|
|
81
|
+
return device, dtype, {
|
|
82
|
+
"device": str(device),
|
|
83
|
+
"name": props.name,
|
|
84
|
+
"freeVramGb": round(free_bytes / 1024**3, 2),
|
|
85
|
+
"totalVramGb": round(total_gb, 2),
|
|
86
|
+
"dtype": str(dtype).replace("torch.", ""),
|
|
87
|
+
"architecture": platform.machine(),
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def pull(model_id: str) -> Path:
|
|
92
|
+
from huggingface_hub import snapshot_download
|
|
93
|
+
|
|
94
|
+
spec = MODELS[model_id]
|
|
95
|
+
target = model_path(model_id)
|
|
96
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
97
|
+
snapshot_download(repo_id=spec["upstream"], local_dir=str(target))
|
|
98
|
+
(target / ".omnius-model.json").write_text(
|
|
99
|
+
json.dumps(
|
|
100
|
+
{
|
|
101
|
+
"engineId": "voxtral-transformers",
|
|
102
|
+
"modelId": model_id,
|
|
103
|
+
"upstreamModelId": spec["upstream"],
|
|
104
|
+
"pulledAt": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
105
|
+
},
|
|
106
|
+
indent=2,
|
|
107
|
+
)
|
|
108
|
+
+ "\n",
|
|
109
|
+
encoding="utf-8",
|
|
110
|
+
)
|
|
111
|
+
return target
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def load(args: argparse.Namespace) -> tuple[Any, Any, Any, dict[str, Any]]:
|
|
115
|
+
import torch
|
|
116
|
+
from transformers import AutoProcessor
|
|
117
|
+
|
|
118
|
+
spec = MODELS[args.model]
|
|
119
|
+
device, dtype, hardware = select_device(torch, spec["minimum_vram_gb"])
|
|
120
|
+
target = model_path(args.model)
|
|
121
|
+
if not (target / "config.json").exists():
|
|
122
|
+
raise RuntimeError(
|
|
123
|
+
f"weights are absent at {target}; call POST /v1/asr/engines/voxtral-transformers/models/{args.model}/pull"
|
|
124
|
+
)
|
|
125
|
+
processor = AutoProcessor.from_pretrained(str(target))
|
|
126
|
+
if spec["realtime"]:
|
|
127
|
+
from transformers import VoxtralRealtimeForConditionalGeneration
|
|
128
|
+
|
|
129
|
+
model_class = VoxtralRealtimeForConditionalGeneration
|
|
130
|
+
else:
|
|
131
|
+
from transformers import VoxtralForConditionalGeneration
|
|
132
|
+
|
|
133
|
+
model_class = VoxtralForConditionalGeneration
|
|
134
|
+
model = model_class.from_pretrained(
|
|
135
|
+
str(target), torch_dtype=dtype, low_cpu_mem_usage=True
|
|
136
|
+
)
|
|
137
|
+
model.to(device)
|
|
138
|
+
model.eval()
|
|
139
|
+
return processor, model, device, hardware
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def transcribe(
|
|
143
|
+
args: argparse.Namespace,
|
|
144
|
+
processor: Any,
|
|
145
|
+
model: Any,
|
|
146
|
+
device: Any,
|
|
147
|
+
audio: Any,
|
|
148
|
+
sample_rate: int,
|
|
149
|
+
) -> str:
|
|
150
|
+
import torch
|
|
151
|
+
|
|
152
|
+
spec = MODELS[args.model]
|
|
153
|
+
if spec["realtime"]:
|
|
154
|
+
inputs = processor(audio, sampling_rate=sample_rate, return_tensors="pt")
|
|
155
|
+
else:
|
|
156
|
+
request: dict[str, Any] = {
|
|
157
|
+
"audio": audio,
|
|
158
|
+
"model_id": spec["upstream"],
|
|
159
|
+
}
|
|
160
|
+
if args.language:
|
|
161
|
+
request["language"] = args.language
|
|
162
|
+
inputs = processor.apply_transcription_request(**request)
|
|
163
|
+
inputs = {name: value.to(device) for name, value in inputs.items()}
|
|
164
|
+
input_length = inputs["input_ids"].shape[-1] if "input_ids" in inputs else 0
|
|
165
|
+
with torch.inference_mode():
|
|
166
|
+
output = model.generate(**inputs, max_new_tokens=1024)
|
|
167
|
+
if input_length:
|
|
168
|
+
output = output[:, input_length:]
|
|
169
|
+
return processor.batch_decode(output, skip_special_tokens=True)[0].strip()
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def transcribe_file(
|
|
173
|
+
args: argparse.Namespace, processor: Any, model: Any, device: Any
|
|
174
|
+
) -> int:
|
|
175
|
+
import librosa
|
|
176
|
+
|
|
177
|
+
started = time.monotonic()
|
|
178
|
+
sample_rate = int(
|
|
179
|
+
getattr(getattr(processor, "feature_extractor", None), "sampling_rate", 16000)
|
|
180
|
+
)
|
|
181
|
+
audio, _ = librosa.load(args.file, sr=sample_rate, mono=True)
|
|
182
|
+
text = transcribe(args, processor, model, device, audio, sample_rate)
|
|
183
|
+
emit(
|
|
184
|
+
{
|
|
185
|
+
"type": "transcript",
|
|
186
|
+
"text": text,
|
|
187
|
+
"rawText": text,
|
|
188
|
+
"isFinal": True,
|
|
189
|
+
"duration": round(time.monotonic() - started, 3),
|
|
190
|
+
"language": args.language,
|
|
191
|
+
"segments": [],
|
|
192
|
+
"engineId": "voxtral-transformers",
|
|
193
|
+
"modelId": args.model,
|
|
194
|
+
}
|
|
195
|
+
)
|
|
196
|
+
return 0
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def stream(
|
|
200
|
+
args: argparse.Namespace, processor: Any, model: Any, device: Any
|
|
201
|
+
) -> int:
|
|
202
|
+
import numpy as np
|
|
203
|
+
|
|
204
|
+
chunk_bytes = max(2, int(args.chunk_seconds * args.sample_rate * 2))
|
|
205
|
+
window_samples = max(1, int(args.window_seconds * args.sample_rate))
|
|
206
|
+
audio_buffer = np.zeros(0, dtype=np.float32)
|
|
207
|
+
last_text = ""
|
|
208
|
+
emit({"type": "ready"})
|
|
209
|
+
while True:
|
|
210
|
+
data = sys.stdin.buffer.read(chunk_bytes)
|
|
211
|
+
if not data:
|
|
212
|
+
break
|
|
213
|
+
samples = np.frombuffer(data, dtype="<i2").astype(np.float32) / 32768.0
|
|
214
|
+
audio_buffer = np.concatenate((audio_buffer, samples))[-window_samples:]
|
|
215
|
+
if len(audio_buffer) < args.sample_rate:
|
|
216
|
+
continue
|
|
217
|
+
text = transcribe(
|
|
218
|
+
args, processor, model, device, audio_buffer, args.sample_rate
|
|
219
|
+
)
|
|
220
|
+
if text and text != last_text:
|
|
221
|
+
last_text = text
|
|
222
|
+
emit({"type": "transcript", "text": text, "isFinal": False})
|
|
223
|
+
if len(audio_buffer) >= args.sample_rate:
|
|
224
|
+
text = transcribe(
|
|
225
|
+
args, processor, model, device, audio_buffer, args.sample_rate
|
|
226
|
+
)
|
|
227
|
+
if text:
|
|
228
|
+
emit({"type": "transcript", "text": text, "isFinal": True})
|
|
229
|
+
return 0
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def main() -> int:
|
|
233
|
+
parser = argparse.ArgumentParser(description="Omnius Voxtral ASR worker")
|
|
234
|
+
parser.add_argument("--model", required=True, choices=sorted(MODELS))
|
|
235
|
+
parser.add_argument("--setup", action="store_true")
|
|
236
|
+
parser.add_argument("--check", action="store_true")
|
|
237
|
+
parser.add_argument("--file")
|
|
238
|
+
parser.add_argument("--language")
|
|
239
|
+
parser.add_argument("--sample-rate", type=int, default=16000)
|
|
240
|
+
parser.add_argument("--chunk-seconds", type=float, default=0.48)
|
|
241
|
+
parser.add_argument("--window-seconds", type=float, default=30.0)
|
|
242
|
+
args = parser.parse_args()
|
|
243
|
+
if args.check:
|
|
244
|
+
emit({"type": "check", "ok": True, "script": str(Path(__file__).resolve())})
|
|
245
|
+
return 0
|
|
246
|
+
if args.setup:
|
|
247
|
+
target = pull(args.model)
|
|
248
|
+
processor, model, device, hardware = load(args)
|
|
249
|
+
del processor, model, device
|
|
250
|
+
emit(
|
|
251
|
+
{
|
|
252
|
+
"type": "ready",
|
|
253
|
+
"engineId": "voxtral-transformers",
|
|
254
|
+
"modelId": args.model,
|
|
255
|
+
"modelPath": str(target),
|
|
256
|
+
"device": hardware["device"],
|
|
257
|
+
"cuda": str(hardware["device"]).startswith("cuda"),
|
|
258
|
+
"hardware": hardware,
|
|
259
|
+
}
|
|
260
|
+
)
|
|
261
|
+
return 0
|
|
262
|
+
processor, model, device, hardware = load(args)
|
|
263
|
+
emit({"type": "status", "message": f"Loaded {args.model} on {hardware['device']}"})
|
|
264
|
+
if args.file:
|
|
265
|
+
return transcribe_file(args, processor, model, device)
|
|
266
|
+
return stream(args, processor, model, device)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
if __name__ == "__main__":
|
|
270
|
+
try:
|
|
271
|
+
sys.exit(main())
|
|
272
|
+
except KeyboardInterrupt:
|
|
273
|
+
sys.exit(0)
|
|
274
|
+
except Exception as exc:
|
|
275
|
+
error(f"{type(exc).__name__}: {exc}")
|
|
276
|
+
sys.exit(1)
|
package/dist/update-worker.js
CHANGED
|
@@ -293431,6 +293431,109 @@ var ASR_ENGINES = Object.freeze([
|
|
|
293431
293431
|
}
|
|
293432
293432
|
]
|
|
293433
293433
|
},
|
|
293434
|
+
{
|
|
293435
|
+
id: "voxtral-transformers",
|
|
293436
|
+
label: "Mistral Voxtral",
|
|
293437
|
+
detail: "Managed CUDA Voxtral ASR for realtime and completed audio",
|
|
293438
|
+
provider: "Mistral AI",
|
|
293439
|
+
runtime: "python",
|
|
293440
|
+
setupMode: "managed",
|
|
293441
|
+
models: [
|
|
293442
|
+
{
|
|
293443
|
+
id: "voxtral-mini-4b-realtime-2602",
|
|
293444
|
+
engineId: "voxtral-transformers",
|
|
293445
|
+
label: "Voxtral Mini 4B Realtime 2602",
|
|
293446
|
+
detail: "Low-latency multilingual streaming and file transcription",
|
|
293447
|
+
upstreamModelId: "mistralai/Voxtral-Mini-4B-Realtime-2602",
|
|
293448
|
+
license: "Apache-2.0",
|
|
293449
|
+
languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi", "ar", "ru", "zh", "ja", "ko"],
|
|
293450
|
+
capabilities: {
|
|
293451
|
+
file: true,
|
|
293452
|
+
pcmStream: true,
|
|
293453
|
+
partials: true,
|
|
293454
|
+
diarization: false,
|
|
293455
|
+
segmentTimestamps: false,
|
|
293456
|
+
wordTimestamps: false,
|
|
293457
|
+
languageDetection: true,
|
|
293458
|
+
languageHints: true,
|
|
293459
|
+
contextPrompt: false,
|
|
293460
|
+
sampleRates: [16e3]
|
|
293461
|
+
},
|
|
293462
|
+
resources: {
|
|
293463
|
+
parameterCount: "4B",
|
|
293464
|
+
minimumGpuMemoryBytes: 15 * 1024 ** 3,
|
|
293465
|
+
cpuSupported: false,
|
|
293466
|
+
architectures: ["x64", "arm64"],
|
|
293467
|
+
notes: [
|
|
293468
|
+
"Transformers 5.2 or newer; 480 ms is the recommended realtime delay.",
|
|
293469
|
+
"Activation is fail-closed and never falls back to CPU or another GPU."
|
|
293470
|
+
]
|
|
293471
|
+
}
|
|
293472
|
+
},
|
|
293473
|
+
{
|
|
293474
|
+
id: "voxtral-mini-3b-2507",
|
|
293475
|
+
engineId: "voxtral-transformers",
|
|
293476
|
+
label: "Voxtral Mini 3B 2507",
|
|
293477
|
+
detail: "Compact multilingual audio understanding and transcription tier",
|
|
293478
|
+
upstreamModelId: "mistralai/Voxtral-Mini-3B-2507",
|
|
293479
|
+
license: "Apache-2.0",
|
|
293480
|
+
languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi"],
|
|
293481
|
+
capabilities: {
|
|
293482
|
+
file: true,
|
|
293483
|
+
pcmStream: false,
|
|
293484
|
+
partials: false,
|
|
293485
|
+
diarization: false,
|
|
293486
|
+
segmentTimestamps: false,
|
|
293487
|
+
wordTimestamps: false,
|
|
293488
|
+
languageDetection: true,
|
|
293489
|
+
languageHints: true,
|
|
293490
|
+
contextPrompt: false,
|
|
293491
|
+
sampleRates: [16e3]
|
|
293492
|
+
},
|
|
293493
|
+
resources: {
|
|
293494
|
+
parameterCount: "3B text model / 5B total",
|
|
293495
|
+
minimumGpuMemoryBytes: 9 * 1024 ** 3,
|
|
293496
|
+
cpuSupported: false,
|
|
293497
|
+
architectures: ["x64", "arm64"],
|
|
293498
|
+
notes: [
|
|
293499
|
+
"Optimized for completed utterances and files rather than incremental PCM.",
|
|
293500
|
+
"Activation is fail-closed and never falls back to CPU or another GPU."
|
|
293501
|
+
]
|
|
293502
|
+
}
|
|
293503
|
+
},
|
|
293504
|
+
{
|
|
293505
|
+
id: "voxtral-small-24b-2507",
|
|
293506
|
+
engineId: "voxtral-transformers",
|
|
293507
|
+
label: "Voxtral Small 24B 2507",
|
|
293508
|
+
detail: "High-capacity multilingual audio understanding and transcription tier",
|
|
293509
|
+
upstreamModelId: "mistralai/Voxtral-Small-24B-2507",
|
|
293510
|
+
license: "Apache-2.0",
|
|
293511
|
+
languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi"],
|
|
293512
|
+
capabilities: {
|
|
293513
|
+
file: true,
|
|
293514
|
+
pcmStream: false,
|
|
293515
|
+
partials: false,
|
|
293516
|
+
diarization: false,
|
|
293517
|
+
segmentTimestamps: false,
|
|
293518
|
+
wordTimestamps: false,
|
|
293519
|
+
languageDetection: true,
|
|
293520
|
+
languageHints: true,
|
|
293521
|
+
contextPrompt: false,
|
|
293522
|
+
sampleRates: [16e3]
|
|
293523
|
+
},
|
|
293524
|
+
resources: {
|
|
293525
|
+
parameterCount: "24B",
|
|
293526
|
+
minimumGpuMemoryBytes: 54 * 1024 ** 3,
|
|
293527
|
+
cpuSupported: false,
|
|
293528
|
+
architectures: ["x64", "arm64"],
|
|
293529
|
+
notes: [
|
|
293530
|
+
"BF16/FP16 deployment requires approximately 55 GB GPU memory and may use tensor parallelism.",
|
|
293531
|
+
"Activation is fail-closed and never falls back to CPU or another GPU."
|
|
293532
|
+
]
|
|
293533
|
+
}
|
|
293534
|
+
}
|
|
293535
|
+
]
|
|
293536
|
+
},
|
|
293434
293537
|
{
|
|
293435
293538
|
id: "vibevoice-transformers",
|
|
293436
293539
|
label: "Microsoft VibeVoice ASR",
|