omnius 1.0.623 → 1.0.625

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,276 @@
1
+ #!/usr/bin/env python3
2
+ """CUDA-only Voxtral ASR worker managed by Omnius."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import json
8
+ import os
9
+ import platform
10
+ import sys
11
+ import time
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ MODELS = {
16
+ "voxtral-mini-4b-realtime-2602": {
17
+ "upstream": "mistralai/Voxtral-Mini-4B-Realtime-2602",
18
+ "realtime": True,
19
+ "minimum_vram_gb": 15.0,
20
+ },
21
+ "voxtral-mini-3b-2507": {
22
+ "upstream": "mistralai/Voxtral-Mini-3B-2507",
23
+ "realtime": False,
24
+ "minimum_vram_gb": 9.0,
25
+ },
26
+ "voxtral-small-24b-2507": {
27
+ "upstream": "mistralai/Voxtral-Small-24B-2507",
28
+ "realtime": False,
29
+ "minimum_vram_gb": 54.0,
30
+ },
31
+ }
32
+
33
+
34
+ def emit(event: dict[str, Any]) -> None:
35
+ sys.stdout.write(json.dumps(event, ensure_ascii=True) + "\n")
36
+ sys.stdout.flush()
37
+
38
+
39
+ def error(message: str) -> None:
40
+ emit({"type": "error", "message": message})
41
+
42
+
43
+ def model_path(model_id: str) -> Path:
44
+ root = Path(
45
+ os.environ.get(
46
+ "OMNIUS_ASR_MODEL_DIR",
47
+ str(Path.home() / ".omnius" / "models" / "asr" / "voxtral-transformers"),
48
+ )
49
+ )
50
+ return root / model_id
51
+
52
+
53
+ def select_device(torch: Any, minimum_vram_gb: float) -> tuple[Any, Any, dict[str, Any]]:
54
+ allow_cpu = os.environ.get("OMNIUS_ASR_ALLOW_CPU", "").lower() in (
55
+ "1",
56
+ "true",
57
+ "yes",
58
+ "on",
59
+ )
60
+ if not torch.cuda.is_available():
61
+ if allow_cpu:
62
+ return torch.device("cpu"), torch.float32, {"device": "cpu"}
63
+ raise RuntimeError(
64
+ "CUDA is required for Voxtral ASR; OMNIUS_ASR_ALLOW_CPU=1 is diagnostic only"
65
+ )
66
+ index = int(os.environ.get("OMNIUS_ASR_CUDA_DEVICE", "0") or "0")
67
+ if index < 0 or index >= torch.cuda.device_count():
68
+ raise RuntimeError(
69
+ f"CUDA device {index} is outside the visible range (count={torch.cuda.device_count()})"
70
+ )
71
+ torch.cuda.set_device(index)
72
+ props = torch.cuda.get_device_properties(index)
73
+ total_gb = props.total_memory / 1024**3
74
+ free_bytes, _ = torch.cuda.mem_get_info(index)
75
+ if total_gb < minimum_vram_gb:
76
+ raise RuntimeError(
77
+ f"{props.name} has {total_gb:.1f} GB VRAM; this Voxtral tier requires approximately {minimum_vram_gb:.0f} GB"
78
+ )
79
+ dtype = torch.bfloat16 if torch.cuda.is_bf16_supported() else torch.float16
80
+ device = torch.device(f"cuda:{index}")
81
+ return device, dtype, {
82
+ "device": str(device),
83
+ "name": props.name,
84
+ "freeVramGb": round(free_bytes / 1024**3, 2),
85
+ "totalVramGb": round(total_gb, 2),
86
+ "dtype": str(dtype).replace("torch.", ""),
87
+ "architecture": platform.machine(),
88
+ }
89
+
90
+
91
+ def pull(model_id: str) -> Path:
92
+ from huggingface_hub import snapshot_download
93
+
94
+ spec = MODELS[model_id]
95
+ target = model_path(model_id)
96
+ target.mkdir(parents=True, exist_ok=True)
97
+ snapshot_download(repo_id=spec["upstream"], local_dir=str(target))
98
+ (target / ".omnius-model.json").write_text(
99
+ json.dumps(
100
+ {
101
+ "engineId": "voxtral-transformers",
102
+ "modelId": model_id,
103
+ "upstreamModelId": spec["upstream"],
104
+ "pulledAt": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
105
+ },
106
+ indent=2,
107
+ )
108
+ + "\n",
109
+ encoding="utf-8",
110
+ )
111
+ return target
112
+
113
+
114
+ def load(args: argparse.Namespace) -> tuple[Any, Any, Any, dict[str, Any]]:
115
+ import torch
116
+ from transformers import AutoProcessor
117
+
118
+ spec = MODELS[args.model]
119
+ device, dtype, hardware = select_device(torch, spec["minimum_vram_gb"])
120
+ target = model_path(args.model)
121
+ if not (target / "config.json").exists():
122
+ raise RuntimeError(
123
+ f"weights are absent at {target}; call POST /v1/asr/engines/voxtral-transformers/models/{args.model}/pull"
124
+ )
125
+ processor = AutoProcessor.from_pretrained(str(target))
126
+ if spec["realtime"]:
127
+ from transformers import VoxtralRealtimeForConditionalGeneration
128
+
129
+ model_class = VoxtralRealtimeForConditionalGeneration
130
+ else:
131
+ from transformers import VoxtralForConditionalGeneration
132
+
133
+ model_class = VoxtralForConditionalGeneration
134
+ model = model_class.from_pretrained(
135
+ str(target), torch_dtype=dtype, low_cpu_mem_usage=True
136
+ )
137
+ model.to(device)
138
+ model.eval()
139
+ return processor, model, device, hardware
140
+
141
+
142
+ def transcribe(
143
+ args: argparse.Namespace,
144
+ processor: Any,
145
+ model: Any,
146
+ device: Any,
147
+ audio: Any,
148
+ sample_rate: int,
149
+ ) -> str:
150
+ import torch
151
+
152
+ spec = MODELS[args.model]
153
+ if spec["realtime"]:
154
+ inputs = processor(audio, sampling_rate=sample_rate, return_tensors="pt")
155
+ else:
156
+ request: dict[str, Any] = {
157
+ "audio": audio,
158
+ "model_id": spec["upstream"],
159
+ }
160
+ if args.language:
161
+ request["language"] = args.language
162
+ inputs = processor.apply_transcription_request(**request)
163
+ inputs = {name: value.to(device) for name, value in inputs.items()}
164
+ input_length = inputs["input_ids"].shape[-1] if "input_ids" in inputs else 0
165
+ with torch.inference_mode():
166
+ output = model.generate(**inputs, max_new_tokens=1024)
167
+ if input_length:
168
+ output = output[:, input_length:]
169
+ return processor.batch_decode(output, skip_special_tokens=True)[0].strip()
170
+
171
+
172
+ def transcribe_file(
173
+ args: argparse.Namespace, processor: Any, model: Any, device: Any
174
+ ) -> int:
175
+ import librosa
176
+
177
+ started = time.monotonic()
178
+ sample_rate = int(
179
+ getattr(getattr(processor, "feature_extractor", None), "sampling_rate", 16000)
180
+ )
181
+ audio, _ = librosa.load(args.file, sr=sample_rate, mono=True)
182
+ text = transcribe(args, processor, model, device, audio, sample_rate)
183
+ emit(
184
+ {
185
+ "type": "transcript",
186
+ "text": text,
187
+ "rawText": text,
188
+ "isFinal": True,
189
+ "duration": round(time.monotonic() - started, 3),
190
+ "language": args.language,
191
+ "segments": [],
192
+ "engineId": "voxtral-transformers",
193
+ "modelId": args.model,
194
+ }
195
+ )
196
+ return 0
197
+
198
+
199
+ def stream(
200
+ args: argparse.Namespace, processor: Any, model: Any, device: Any
201
+ ) -> int:
202
+ import numpy as np
203
+
204
+ chunk_bytes = max(2, int(args.chunk_seconds * args.sample_rate * 2))
205
+ window_samples = max(1, int(args.window_seconds * args.sample_rate))
206
+ audio_buffer = np.zeros(0, dtype=np.float32)
207
+ last_text = ""
208
+ emit({"type": "ready"})
209
+ while True:
210
+ data = sys.stdin.buffer.read(chunk_bytes)
211
+ if not data:
212
+ break
213
+ samples = np.frombuffer(data, dtype="<i2").astype(np.float32) / 32768.0
214
+ audio_buffer = np.concatenate((audio_buffer, samples))[-window_samples:]
215
+ if len(audio_buffer) < args.sample_rate:
216
+ continue
217
+ text = transcribe(
218
+ args, processor, model, device, audio_buffer, args.sample_rate
219
+ )
220
+ if text and text != last_text:
221
+ last_text = text
222
+ emit({"type": "transcript", "text": text, "isFinal": False})
223
+ if len(audio_buffer) >= args.sample_rate:
224
+ text = transcribe(
225
+ args, processor, model, device, audio_buffer, args.sample_rate
226
+ )
227
+ if text:
228
+ emit({"type": "transcript", "text": text, "isFinal": True})
229
+ return 0
230
+
231
+
232
+ def main() -> int:
233
+ parser = argparse.ArgumentParser(description="Omnius Voxtral ASR worker")
234
+ parser.add_argument("--model", required=True, choices=sorted(MODELS))
235
+ parser.add_argument("--setup", action="store_true")
236
+ parser.add_argument("--check", action="store_true")
237
+ parser.add_argument("--file")
238
+ parser.add_argument("--language")
239
+ parser.add_argument("--sample-rate", type=int, default=16000)
240
+ parser.add_argument("--chunk-seconds", type=float, default=0.48)
241
+ parser.add_argument("--window-seconds", type=float, default=30.0)
242
+ args = parser.parse_args()
243
+ if args.check:
244
+ emit({"type": "check", "ok": True, "script": str(Path(__file__).resolve())})
245
+ return 0
246
+ if args.setup:
247
+ target = pull(args.model)
248
+ processor, model, device, hardware = load(args)
249
+ del processor, model, device
250
+ emit(
251
+ {
252
+ "type": "ready",
253
+ "engineId": "voxtral-transformers",
254
+ "modelId": args.model,
255
+ "modelPath": str(target),
256
+ "device": hardware["device"],
257
+ "cuda": str(hardware["device"]).startswith("cuda"),
258
+ "hardware": hardware,
259
+ }
260
+ )
261
+ return 0
262
+ processor, model, device, hardware = load(args)
263
+ emit({"type": "status", "message": f"Loaded {args.model} on {hardware['device']}"})
264
+ if args.file:
265
+ return transcribe_file(args, processor, model, device)
266
+ return stream(args, processor, model, device)
267
+
268
+
269
+ if __name__ == "__main__":
270
+ try:
271
+ sys.exit(main())
272
+ except KeyboardInterrupt:
273
+ sys.exit(0)
274
+ except Exception as exc:
275
+ error(f"{type(exc).__name__}: {exc}")
276
+ sys.exit(1)
@@ -293431,6 +293431,109 @@ var ASR_ENGINES = Object.freeze([
293431
293431
  }
293432
293432
  ]
293433
293433
  },
293434
+ {
293435
+ id: "voxtral-transformers",
293436
+ label: "Mistral Voxtral",
293437
+ detail: "Managed CUDA Voxtral ASR for realtime and completed audio",
293438
+ provider: "Mistral AI",
293439
+ runtime: "python",
293440
+ setupMode: "managed",
293441
+ models: [
293442
+ {
293443
+ id: "voxtral-mini-4b-realtime-2602",
293444
+ engineId: "voxtral-transformers",
293445
+ label: "Voxtral Mini 4B Realtime 2602",
293446
+ detail: "Low-latency multilingual streaming and file transcription",
293447
+ upstreamModelId: "mistralai/Voxtral-Mini-4B-Realtime-2602",
293448
+ license: "Apache-2.0",
293449
+ languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi", "ar", "ru", "zh", "ja", "ko"],
293450
+ capabilities: {
293451
+ file: true,
293452
+ pcmStream: true,
293453
+ partials: true,
293454
+ diarization: false,
293455
+ segmentTimestamps: false,
293456
+ wordTimestamps: false,
293457
+ languageDetection: true,
293458
+ languageHints: true,
293459
+ contextPrompt: false,
293460
+ sampleRates: [16e3]
293461
+ },
293462
+ resources: {
293463
+ parameterCount: "4B",
293464
+ minimumGpuMemoryBytes: 15 * 1024 ** 3,
293465
+ cpuSupported: false,
293466
+ architectures: ["x64", "arm64"],
293467
+ notes: [
293468
+ "Transformers 5.2 or newer; 480 ms is the recommended realtime delay.",
293469
+ "Activation is fail-closed and never falls back to CPU or another GPU."
293470
+ ]
293471
+ }
293472
+ },
293473
+ {
293474
+ id: "voxtral-mini-3b-2507",
293475
+ engineId: "voxtral-transformers",
293476
+ label: "Voxtral Mini 3B 2507",
293477
+ detail: "Compact multilingual audio understanding and transcription tier",
293478
+ upstreamModelId: "mistralai/Voxtral-Mini-3B-2507",
293479
+ license: "Apache-2.0",
293480
+ languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi"],
293481
+ capabilities: {
293482
+ file: true,
293483
+ pcmStream: false,
293484
+ partials: false,
293485
+ diarization: false,
293486
+ segmentTimestamps: false,
293487
+ wordTimestamps: false,
293488
+ languageDetection: true,
293489
+ languageHints: true,
293490
+ contextPrompt: false,
293491
+ sampleRates: [16e3]
293492
+ },
293493
+ resources: {
293494
+ parameterCount: "3B text model / 5B total",
293495
+ minimumGpuMemoryBytes: 9 * 1024 ** 3,
293496
+ cpuSupported: false,
293497
+ architectures: ["x64", "arm64"],
293498
+ notes: [
293499
+ "Optimized for completed utterances and files rather than incremental PCM.",
293500
+ "Activation is fail-closed and never falls back to CPU or another GPU."
293501
+ ]
293502
+ }
293503
+ },
293504
+ {
293505
+ id: "voxtral-small-24b-2507",
293506
+ engineId: "voxtral-transformers",
293507
+ label: "Voxtral Small 24B 2507",
293508
+ detail: "High-capacity multilingual audio understanding and transcription tier",
293509
+ upstreamModelId: "mistralai/Voxtral-Small-24B-2507",
293510
+ license: "Apache-2.0",
293511
+ languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi"],
293512
+ capabilities: {
293513
+ file: true,
293514
+ pcmStream: false,
293515
+ partials: false,
293516
+ diarization: false,
293517
+ segmentTimestamps: false,
293518
+ wordTimestamps: false,
293519
+ languageDetection: true,
293520
+ languageHints: true,
293521
+ contextPrompt: false,
293522
+ sampleRates: [16e3]
293523
+ },
293524
+ resources: {
293525
+ parameterCount: "24B",
293526
+ minimumGpuMemoryBytes: 54 * 1024 ** 3,
293527
+ cpuSupported: false,
293528
+ architectures: ["x64", "arm64"],
293529
+ notes: [
293530
+ "BF16/FP16 deployment requires approximately 55 GB GPU memory and may use tensor parallelism.",
293531
+ "Activation is fail-closed and never falls back to CPU or another GPU."
293532
+ ]
293533
+ }
293534
+ }
293535
+ ]
293536
+ },
293434
293537
  {
293435
293538
  id: "vibevoice-transformers",
293436
293539
  label: "Microsoft VibeVoice ASR",