omnius 1.0.623 → 1.0.624
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +5217 -4604
- package/dist/scripts/live-voxtral.py +276 -0
- package/dist/update-worker.js +103 -0
- package/docs/DISCOVERY.json +184 -6
- package/docs/DISCOVERY.md +3 -1
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""CUDA-only Voxtral ASR worker managed by Omnius."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import platform
|
|
10
|
+
import sys
|
|
11
|
+
import time
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
MODELS = {
|
|
16
|
+
"voxtral-mini-4b-realtime-2602": {
|
|
17
|
+
"upstream": "mistralai/Voxtral-Mini-4B-Realtime-2602",
|
|
18
|
+
"realtime": True,
|
|
19
|
+
"minimum_vram_gb": 15.0,
|
|
20
|
+
},
|
|
21
|
+
"voxtral-mini-3b-2507": {
|
|
22
|
+
"upstream": "mistralai/Voxtral-Mini-3B-2507",
|
|
23
|
+
"realtime": False,
|
|
24
|
+
"minimum_vram_gb": 9.0,
|
|
25
|
+
},
|
|
26
|
+
"voxtral-small-24b-2507": {
|
|
27
|
+
"upstream": "mistralai/Voxtral-Small-24B-2507",
|
|
28
|
+
"realtime": False,
|
|
29
|
+
"minimum_vram_gb": 54.0,
|
|
30
|
+
},
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def emit(event: dict[str, Any]) -> None:
|
|
35
|
+
sys.stdout.write(json.dumps(event, ensure_ascii=True) + "\n")
|
|
36
|
+
sys.stdout.flush()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def error(message: str) -> None:
|
|
40
|
+
emit({"type": "error", "message": message})
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def model_path(model_id: str) -> Path:
|
|
44
|
+
root = Path(
|
|
45
|
+
os.environ.get(
|
|
46
|
+
"OMNIUS_ASR_MODEL_DIR",
|
|
47
|
+
str(Path.home() / ".omnius" / "models" / "asr" / "voxtral-transformers"),
|
|
48
|
+
)
|
|
49
|
+
)
|
|
50
|
+
return root / model_id
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def select_device(torch: Any, minimum_vram_gb: float) -> tuple[Any, Any, dict[str, Any]]:
|
|
54
|
+
allow_cpu = os.environ.get("OMNIUS_ASR_ALLOW_CPU", "").lower() in (
|
|
55
|
+
"1",
|
|
56
|
+
"true",
|
|
57
|
+
"yes",
|
|
58
|
+
"on",
|
|
59
|
+
)
|
|
60
|
+
if not torch.cuda.is_available():
|
|
61
|
+
if allow_cpu:
|
|
62
|
+
return torch.device("cpu"), torch.float32, {"device": "cpu"}
|
|
63
|
+
raise RuntimeError(
|
|
64
|
+
"CUDA is required for Voxtral ASR; OMNIUS_ASR_ALLOW_CPU=1 is diagnostic only"
|
|
65
|
+
)
|
|
66
|
+
index = int(os.environ.get("OMNIUS_ASR_CUDA_DEVICE", "0") or "0")
|
|
67
|
+
if index < 0 or index >= torch.cuda.device_count():
|
|
68
|
+
raise RuntimeError(
|
|
69
|
+
f"CUDA device {index} is outside the visible range (count={torch.cuda.device_count()})"
|
|
70
|
+
)
|
|
71
|
+
torch.cuda.set_device(index)
|
|
72
|
+
props = torch.cuda.get_device_properties(index)
|
|
73
|
+
total_gb = props.total_memory / 1024**3
|
|
74
|
+
free_bytes, _ = torch.cuda.mem_get_info(index)
|
|
75
|
+
if total_gb < minimum_vram_gb:
|
|
76
|
+
raise RuntimeError(
|
|
77
|
+
f"{props.name} has {total_gb:.1f} GB VRAM; this Voxtral tier requires approximately {minimum_vram_gb:.0f} GB"
|
|
78
|
+
)
|
|
79
|
+
dtype = torch.bfloat16 if torch.cuda.is_bf16_supported() else torch.float16
|
|
80
|
+
device = torch.device(f"cuda:{index}")
|
|
81
|
+
return device, dtype, {
|
|
82
|
+
"device": str(device),
|
|
83
|
+
"name": props.name,
|
|
84
|
+
"freeVramGb": round(free_bytes / 1024**3, 2),
|
|
85
|
+
"totalVramGb": round(total_gb, 2),
|
|
86
|
+
"dtype": str(dtype).replace("torch.", ""),
|
|
87
|
+
"architecture": platform.machine(),
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def pull(model_id: str) -> Path:
|
|
92
|
+
from huggingface_hub import snapshot_download
|
|
93
|
+
|
|
94
|
+
spec = MODELS[model_id]
|
|
95
|
+
target = model_path(model_id)
|
|
96
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
97
|
+
snapshot_download(repo_id=spec["upstream"], local_dir=str(target))
|
|
98
|
+
(target / ".omnius-model.json").write_text(
|
|
99
|
+
json.dumps(
|
|
100
|
+
{
|
|
101
|
+
"engineId": "voxtral-transformers",
|
|
102
|
+
"modelId": model_id,
|
|
103
|
+
"upstreamModelId": spec["upstream"],
|
|
104
|
+
"pulledAt": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
105
|
+
},
|
|
106
|
+
indent=2,
|
|
107
|
+
)
|
|
108
|
+
+ "\n",
|
|
109
|
+
encoding="utf-8",
|
|
110
|
+
)
|
|
111
|
+
return target
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def load(args: argparse.Namespace) -> tuple[Any, Any, Any, dict[str, Any]]:
|
|
115
|
+
import torch
|
|
116
|
+
from transformers import AutoProcessor
|
|
117
|
+
|
|
118
|
+
spec = MODELS[args.model]
|
|
119
|
+
device, dtype, hardware = select_device(torch, spec["minimum_vram_gb"])
|
|
120
|
+
target = model_path(args.model)
|
|
121
|
+
if not (target / "config.json").exists():
|
|
122
|
+
raise RuntimeError(
|
|
123
|
+
f"weights are absent at {target}; call POST /v1/asr/engines/voxtral-transformers/models/{args.model}/pull"
|
|
124
|
+
)
|
|
125
|
+
processor = AutoProcessor.from_pretrained(str(target))
|
|
126
|
+
if spec["realtime"]:
|
|
127
|
+
from transformers import VoxtralRealtimeForConditionalGeneration
|
|
128
|
+
|
|
129
|
+
model_class = VoxtralRealtimeForConditionalGeneration
|
|
130
|
+
else:
|
|
131
|
+
from transformers import VoxtralForConditionalGeneration
|
|
132
|
+
|
|
133
|
+
model_class = VoxtralForConditionalGeneration
|
|
134
|
+
model = model_class.from_pretrained(
|
|
135
|
+
str(target), torch_dtype=dtype, low_cpu_mem_usage=True
|
|
136
|
+
)
|
|
137
|
+
model.to(device)
|
|
138
|
+
model.eval()
|
|
139
|
+
return processor, model, device, hardware
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def transcribe(
|
|
143
|
+
args: argparse.Namespace,
|
|
144
|
+
processor: Any,
|
|
145
|
+
model: Any,
|
|
146
|
+
device: Any,
|
|
147
|
+
audio: Any,
|
|
148
|
+
sample_rate: int,
|
|
149
|
+
) -> str:
|
|
150
|
+
import torch
|
|
151
|
+
|
|
152
|
+
spec = MODELS[args.model]
|
|
153
|
+
if spec["realtime"]:
|
|
154
|
+
inputs = processor(audio, sampling_rate=sample_rate, return_tensors="pt")
|
|
155
|
+
else:
|
|
156
|
+
request: dict[str, Any] = {
|
|
157
|
+
"audio": audio,
|
|
158
|
+
"model_id": spec["upstream"],
|
|
159
|
+
}
|
|
160
|
+
if args.language:
|
|
161
|
+
request["language"] = args.language
|
|
162
|
+
inputs = processor.apply_transcription_request(**request)
|
|
163
|
+
inputs = {name: value.to(device) for name, value in inputs.items()}
|
|
164
|
+
input_length = inputs["input_ids"].shape[-1] if "input_ids" in inputs else 0
|
|
165
|
+
with torch.inference_mode():
|
|
166
|
+
output = model.generate(**inputs, max_new_tokens=1024)
|
|
167
|
+
if input_length:
|
|
168
|
+
output = output[:, input_length:]
|
|
169
|
+
return processor.batch_decode(output, skip_special_tokens=True)[0].strip()
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def transcribe_file(
|
|
173
|
+
args: argparse.Namespace, processor: Any, model: Any, device: Any
|
|
174
|
+
) -> int:
|
|
175
|
+
import librosa
|
|
176
|
+
|
|
177
|
+
started = time.monotonic()
|
|
178
|
+
sample_rate = int(
|
|
179
|
+
getattr(getattr(processor, "feature_extractor", None), "sampling_rate", 16000)
|
|
180
|
+
)
|
|
181
|
+
audio, _ = librosa.load(args.file, sr=sample_rate, mono=True)
|
|
182
|
+
text = transcribe(args, processor, model, device, audio, sample_rate)
|
|
183
|
+
emit(
|
|
184
|
+
{
|
|
185
|
+
"type": "transcript",
|
|
186
|
+
"text": text,
|
|
187
|
+
"rawText": text,
|
|
188
|
+
"isFinal": True,
|
|
189
|
+
"duration": round(time.monotonic() - started, 3),
|
|
190
|
+
"language": args.language,
|
|
191
|
+
"segments": [],
|
|
192
|
+
"engineId": "voxtral-transformers",
|
|
193
|
+
"modelId": args.model,
|
|
194
|
+
}
|
|
195
|
+
)
|
|
196
|
+
return 0
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def stream(
|
|
200
|
+
args: argparse.Namespace, processor: Any, model: Any, device: Any
|
|
201
|
+
) -> int:
|
|
202
|
+
import numpy as np
|
|
203
|
+
|
|
204
|
+
chunk_bytes = max(2, int(args.chunk_seconds * args.sample_rate * 2))
|
|
205
|
+
window_samples = max(1, int(args.window_seconds * args.sample_rate))
|
|
206
|
+
audio_buffer = np.zeros(0, dtype=np.float32)
|
|
207
|
+
last_text = ""
|
|
208
|
+
emit({"type": "ready"})
|
|
209
|
+
while True:
|
|
210
|
+
data = sys.stdin.buffer.read(chunk_bytes)
|
|
211
|
+
if not data:
|
|
212
|
+
break
|
|
213
|
+
samples = np.frombuffer(data, dtype="<i2").astype(np.float32) / 32768.0
|
|
214
|
+
audio_buffer = np.concatenate((audio_buffer, samples))[-window_samples:]
|
|
215
|
+
if len(audio_buffer) < args.sample_rate:
|
|
216
|
+
continue
|
|
217
|
+
text = transcribe(
|
|
218
|
+
args, processor, model, device, audio_buffer, args.sample_rate
|
|
219
|
+
)
|
|
220
|
+
if text and text != last_text:
|
|
221
|
+
last_text = text
|
|
222
|
+
emit({"type": "transcript", "text": text, "isFinal": False})
|
|
223
|
+
if len(audio_buffer) >= args.sample_rate:
|
|
224
|
+
text = transcribe(
|
|
225
|
+
args, processor, model, device, audio_buffer, args.sample_rate
|
|
226
|
+
)
|
|
227
|
+
if text:
|
|
228
|
+
emit({"type": "transcript", "text": text, "isFinal": True})
|
|
229
|
+
return 0
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def main() -> int:
|
|
233
|
+
parser = argparse.ArgumentParser(description="Omnius Voxtral ASR worker")
|
|
234
|
+
parser.add_argument("--model", required=True, choices=sorted(MODELS))
|
|
235
|
+
parser.add_argument("--setup", action="store_true")
|
|
236
|
+
parser.add_argument("--check", action="store_true")
|
|
237
|
+
parser.add_argument("--file")
|
|
238
|
+
parser.add_argument("--language")
|
|
239
|
+
parser.add_argument("--sample-rate", type=int, default=16000)
|
|
240
|
+
parser.add_argument("--chunk-seconds", type=float, default=0.48)
|
|
241
|
+
parser.add_argument("--window-seconds", type=float, default=30.0)
|
|
242
|
+
args = parser.parse_args()
|
|
243
|
+
if args.check:
|
|
244
|
+
emit({"type": "check", "ok": True, "script": str(Path(__file__).resolve())})
|
|
245
|
+
return 0
|
|
246
|
+
if args.setup:
|
|
247
|
+
target = pull(args.model)
|
|
248
|
+
processor, model, device, hardware = load(args)
|
|
249
|
+
del processor, model, device
|
|
250
|
+
emit(
|
|
251
|
+
{
|
|
252
|
+
"type": "ready",
|
|
253
|
+
"engineId": "voxtral-transformers",
|
|
254
|
+
"modelId": args.model,
|
|
255
|
+
"modelPath": str(target),
|
|
256
|
+
"device": hardware["device"],
|
|
257
|
+
"cuda": str(hardware["device"]).startswith("cuda"),
|
|
258
|
+
"hardware": hardware,
|
|
259
|
+
}
|
|
260
|
+
)
|
|
261
|
+
return 0
|
|
262
|
+
processor, model, device, hardware = load(args)
|
|
263
|
+
emit({"type": "status", "message": f"Loaded {args.model} on {hardware['device']}"})
|
|
264
|
+
if args.file:
|
|
265
|
+
return transcribe_file(args, processor, model, device)
|
|
266
|
+
return stream(args, processor, model, device)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
if __name__ == "__main__":
|
|
270
|
+
try:
|
|
271
|
+
sys.exit(main())
|
|
272
|
+
except KeyboardInterrupt:
|
|
273
|
+
sys.exit(0)
|
|
274
|
+
except Exception as exc:
|
|
275
|
+
error(f"{type(exc).__name__}: {exc}")
|
|
276
|
+
sys.exit(1)
|
package/dist/update-worker.js
CHANGED
|
@@ -293431,6 +293431,109 @@ var ASR_ENGINES = Object.freeze([
|
|
|
293431
293431
|
}
|
|
293432
293432
|
]
|
|
293433
293433
|
},
|
|
293434
|
+
{
|
|
293435
|
+
id: "voxtral-transformers",
|
|
293436
|
+
label: "Mistral Voxtral",
|
|
293437
|
+
detail: "Managed CUDA Voxtral ASR for realtime and completed audio",
|
|
293438
|
+
provider: "Mistral AI",
|
|
293439
|
+
runtime: "python",
|
|
293440
|
+
setupMode: "managed",
|
|
293441
|
+
models: [
|
|
293442
|
+
{
|
|
293443
|
+
id: "voxtral-mini-4b-realtime-2602",
|
|
293444
|
+
engineId: "voxtral-transformers",
|
|
293445
|
+
label: "Voxtral Mini 4B Realtime 2602",
|
|
293446
|
+
detail: "Low-latency multilingual streaming and file transcription",
|
|
293447
|
+
upstreamModelId: "mistralai/Voxtral-Mini-4B-Realtime-2602",
|
|
293448
|
+
license: "Apache-2.0",
|
|
293449
|
+
languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi", "ar", "ru", "zh", "ja", "ko"],
|
|
293450
|
+
capabilities: {
|
|
293451
|
+
file: true,
|
|
293452
|
+
pcmStream: true,
|
|
293453
|
+
partials: true,
|
|
293454
|
+
diarization: false,
|
|
293455
|
+
segmentTimestamps: false,
|
|
293456
|
+
wordTimestamps: false,
|
|
293457
|
+
languageDetection: true,
|
|
293458
|
+
languageHints: true,
|
|
293459
|
+
contextPrompt: false,
|
|
293460
|
+
sampleRates: [16e3]
|
|
293461
|
+
},
|
|
293462
|
+
resources: {
|
|
293463
|
+
parameterCount: "4B",
|
|
293464
|
+
minimumGpuMemoryBytes: 15 * 1024 ** 3,
|
|
293465
|
+
cpuSupported: false,
|
|
293466
|
+
architectures: ["x64", "arm64"],
|
|
293467
|
+
notes: [
|
|
293468
|
+
"Transformers 5.2 or newer; 480 ms is the recommended realtime delay.",
|
|
293469
|
+
"Activation is fail-closed and never falls back to CPU or another GPU."
|
|
293470
|
+
]
|
|
293471
|
+
}
|
|
293472
|
+
},
|
|
293473
|
+
{
|
|
293474
|
+
id: "voxtral-mini-3b-2507",
|
|
293475
|
+
engineId: "voxtral-transformers",
|
|
293476
|
+
label: "Voxtral Mini 3B 2507",
|
|
293477
|
+
detail: "Compact multilingual audio understanding and transcription tier",
|
|
293478
|
+
upstreamModelId: "mistralai/Voxtral-Mini-3B-2507",
|
|
293479
|
+
license: "Apache-2.0",
|
|
293480
|
+
languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi"],
|
|
293481
|
+
capabilities: {
|
|
293482
|
+
file: true,
|
|
293483
|
+
pcmStream: false,
|
|
293484
|
+
partials: false,
|
|
293485
|
+
diarization: false,
|
|
293486
|
+
segmentTimestamps: false,
|
|
293487
|
+
wordTimestamps: false,
|
|
293488
|
+
languageDetection: true,
|
|
293489
|
+
languageHints: true,
|
|
293490
|
+
contextPrompt: false,
|
|
293491
|
+
sampleRates: [16e3]
|
|
293492
|
+
},
|
|
293493
|
+
resources: {
|
|
293494
|
+
parameterCount: "3B text model / 5B total",
|
|
293495
|
+
minimumGpuMemoryBytes: 9 * 1024 ** 3,
|
|
293496
|
+
cpuSupported: false,
|
|
293497
|
+
architectures: ["x64", "arm64"],
|
|
293498
|
+
notes: [
|
|
293499
|
+
"Optimized for completed utterances and files rather than incremental PCM.",
|
|
293500
|
+
"Activation is fail-closed and never falls back to CPU or another GPU."
|
|
293501
|
+
]
|
|
293502
|
+
}
|
|
293503
|
+
},
|
|
293504
|
+
{
|
|
293505
|
+
id: "voxtral-small-24b-2507",
|
|
293506
|
+
engineId: "voxtral-transformers",
|
|
293507
|
+
label: "Voxtral Small 24B 2507",
|
|
293508
|
+
detail: "High-capacity multilingual audio understanding and transcription tier",
|
|
293509
|
+
upstreamModelId: "mistralai/Voxtral-Small-24B-2507",
|
|
293510
|
+
license: "Apache-2.0",
|
|
293511
|
+
languages: ["en", "fr", "de", "es", "it", "pt", "nl", "hi"],
|
|
293512
|
+
capabilities: {
|
|
293513
|
+
file: true,
|
|
293514
|
+
pcmStream: false,
|
|
293515
|
+
partials: false,
|
|
293516
|
+
diarization: false,
|
|
293517
|
+
segmentTimestamps: false,
|
|
293518
|
+
wordTimestamps: false,
|
|
293519
|
+
languageDetection: true,
|
|
293520
|
+
languageHints: true,
|
|
293521
|
+
contextPrompt: false,
|
|
293522
|
+
sampleRates: [16e3]
|
|
293523
|
+
},
|
|
293524
|
+
resources: {
|
|
293525
|
+
parameterCount: "24B",
|
|
293526
|
+
minimumGpuMemoryBytes: 54 * 1024 ** 3,
|
|
293527
|
+
cpuSupported: false,
|
|
293528
|
+
architectures: ["x64", "arm64"],
|
|
293529
|
+
notes: [
|
|
293530
|
+
"BF16/FP16 deployment requires approximately 55 GB GPU memory and may use tensor parallelism.",
|
|
293531
|
+
"Activation is fail-closed and never falls back to CPU or another GPU."
|
|
293532
|
+
]
|
|
293533
|
+
}
|
|
293534
|
+
}
|
|
293535
|
+
]
|
|
293536
|
+
},
|
|
293434
293537
|
{
|
|
293435
293538
|
id: "vibevoice-transformers",
|
|
293436
293539
|
label: "Microsoft VibeVoice ASR",
|
package/docs/DISCOVERY.json
CHANGED
|
@@ -3407,7 +3407,7 @@
|
|
|
3407
3407
|
"tags": [
|
|
3408
3408
|
"ASR"
|
|
3409
3409
|
],
|
|
3410
|
-
"description": "Optional body: {device}. Downloads model weights into Omnius-managed storage and verifies CUDA placement for Whisper and
|
|
3410
|
+
"description": "Optional body: {device}. Downloads model weights into Omnius-managed storage and verifies CUDA placement for Whisper, Nemotron, and Voxtral. Voxtral model IDs are voxtral-mini-4b-realtime-2602, voxtral-mini-3b-2507, and voxtral-small-24b-2507.",
|
|
3411
3411
|
"parameters": [
|
|
3412
3412
|
{
|
|
3413
3413
|
"name": "engineId",
|
|
@@ -3501,7 +3501,7 @@
|
|
|
3501
3501
|
"tags": [
|
|
3502
3502
|
"ASR"
|
|
3503
3503
|
],
|
|
3504
|
-
"description": "Body: {modelId?, device?}. Installs the managed runtime and pulls the requested model. Whisper and
|
|
3504
|
+
"description": "Body: {modelId?, device?}. Installs the managed runtime and pulls the requested model. Whisper, Nemotron, and Voxtral load the requested weights to validate exact CUDA placement; VibeVoice stores its pinned model and tokenizer snapshot under Omnius' ASR runtime directories.",
|
|
3505
3505
|
"parameters": [
|
|
3506
3506
|
{
|
|
3507
3507
|
"name": "engineId",
|
|
@@ -15276,7 +15276,7 @@
|
|
|
15276
15276
|
"id": "api.v1-voice-models",
|
|
15277
15277
|
"kind": "api",
|
|
15278
15278
|
"title": "/v1/voice/models",
|
|
15279
|
-
"summary": "List TTS voice models with backend
|
|
15279
|
+
"summary": "List TTS voice models with backend metadata and managed readiness",
|
|
15280
15280
|
"aliases": [
|
|
15281
15281
|
"/v1/voice/models"
|
|
15282
15282
|
],
|
|
@@ -15326,13 +15326,191 @@
|
|
|
15326
15326
|
],
|
|
15327
15327
|
"operations": {
|
|
15328
15328
|
"get": {
|
|
15329
|
-
"summary": "List TTS voice models with backend
|
|
15329
|
+
"summary": "List TTS voice models with backend metadata and managed readiness",
|
|
15330
15330
|
"tags": [
|
|
15331
15331
|
"Voice"
|
|
15332
15332
|
],
|
|
15333
15333
|
"responses": {
|
|
15334
15334
|
"200": {
|
|
15335
|
-
"description": "Voice model list"
|
|
15335
|
+
"description": "Voice model list, active selection, and Voxtral pull/deploy state"
|
|
15336
|
+
}
|
|
15337
|
+
}
|
|
15338
|
+
}
|
|
15339
|
+
},
|
|
15340
|
+
"source_of_truth": [
|
|
15341
|
+
"GET /openapi.json",
|
|
15342
|
+
"packages/cli/src/api/openapi.ts"
|
|
15343
|
+
]
|
|
15344
|
+
},
|
|
15345
|
+
{
|
|
15346
|
+
"id": "api.v1-voice-models-model-id-deploy",
|
|
15347
|
+
"kind": "api",
|
|
15348
|
+
"title": "/v1/voice/models/{modelId}/deploy",
|
|
15349
|
+
"summary": "Pull and deploy one managed CUDA TTS model",
|
|
15350
|
+
"aliases": [
|
|
15351
|
+
"/v1/voice/models/{modelId}/deploy"
|
|
15352
|
+
],
|
|
15353
|
+
"keywords": [
|
|
15354
|
+
"rest",
|
|
15355
|
+
"openapi",
|
|
15356
|
+
"POST",
|
|
15357
|
+
"Voice",
|
|
15358
|
+
"v1",
|
|
15359
|
+
"voice",
|
|
15360
|
+
"models",
|
|
15361
|
+
"{modelId}",
|
|
15362
|
+
"deploy"
|
|
15363
|
+
],
|
|
15364
|
+
"maturity": "stable",
|
|
15365
|
+
"layer": "interface",
|
|
15366
|
+
"audiences": [
|
|
15367
|
+
"integrator",
|
|
15368
|
+
"service-agent",
|
|
15369
|
+
"coding-agent"
|
|
15370
|
+
],
|
|
15371
|
+
"interfaces": [
|
|
15372
|
+
{
|
|
15373
|
+
"type": "rest",
|
|
15374
|
+
"target": "POST /v1/voice/models/{modelId}/deploy"
|
|
15375
|
+
},
|
|
15376
|
+
{
|
|
15377
|
+
"type": "openapi",
|
|
15378
|
+
"target": "/openapi.json"
|
|
15379
|
+
}
|
|
15380
|
+
],
|
|
15381
|
+
"references": [
|
|
15382
|
+
{
|
|
15383
|
+
"type": "source",
|
|
15384
|
+
"target": "packages/cli/src/api/openapi.ts",
|
|
15385
|
+
"relation": "openapi-source"
|
|
15386
|
+
},
|
|
15387
|
+
{
|
|
15388
|
+
"type": "documentation",
|
|
15389
|
+
"target": "docs/reference/rest-api.md",
|
|
15390
|
+
"relation": "endpoint-inventory"
|
|
15391
|
+
}
|
|
15392
|
+
],
|
|
15393
|
+
"methods": [
|
|
15394
|
+
"POST"
|
|
15395
|
+
],
|
|
15396
|
+
"tags": [
|
|
15397
|
+
"Voice"
|
|
15398
|
+
],
|
|
15399
|
+
"operations": {
|
|
15400
|
+
"post": {
|
|
15401
|
+
"summary": "Pull and deploy one managed CUDA TTS model",
|
|
15402
|
+
"tags": [
|
|
15403
|
+
"Voice"
|
|
15404
|
+
],
|
|
15405
|
+
"description": "For voxtral-4b-tts-2603 this verifies the selected CUDA device and starts a persistent local vLLM-Omni server. Set OMNIUS_TTS_CUDA_DEVICE to choose the physical GPU.",
|
|
15406
|
+
"parameters": [
|
|
15407
|
+
{
|
|
15408
|
+
"name": "modelId",
|
|
15409
|
+
"in": "path",
|
|
15410
|
+
"required": true,
|
|
15411
|
+
"schema": {
|
|
15412
|
+
"type": "string"
|
|
15413
|
+
}
|
|
15414
|
+
}
|
|
15415
|
+
],
|
|
15416
|
+
"responses": {
|
|
15417
|
+
"200": {
|
|
15418
|
+
"description": "Runtime deployed and ready"
|
|
15419
|
+
},
|
|
15420
|
+
"404": {
|
|
15421
|
+
"description": "No managed deploy adapter"
|
|
15422
|
+
},
|
|
15423
|
+
"500": {
|
|
15424
|
+
"description": "Runtime installation, CUDA preflight, or startup failed"
|
|
15425
|
+
}
|
|
15426
|
+
}
|
|
15427
|
+
}
|
|
15428
|
+
},
|
|
15429
|
+
"source_of_truth": [
|
|
15430
|
+
"GET /openapi.json",
|
|
15431
|
+
"packages/cli/src/api/openapi.ts"
|
|
15432
|
+
]
|
|
15433
|
+
},
|
|
15434
|
+
{
|
|
15435
|
+
"id": "api.v1-voice-models-model-id-pull",
|
|
15436
|
+
"kind": "api",
|
|
15437
|
+
"title": "/v1/voice/models/{modelId}/pull",
|
|
15438
|
+
"summary": "Install runtime prerequisites and pull one managed TTS model",
|
|
15439
|
+
"aliases": [
|
|
15440
|
+
"/v1/voice/models/{modelId}/pull"
|
|
15441
|
+
],
|
|
15442
|
+
"keywords": [
|
|
15443
|
+
"rest",
|
|
15444
|
+
"openapi",
|
|
15445
|
+
"POST",
|
|
15446
|
+
"Voice",
|
|
15447
|
+
"v1",
|
|
15448
|
+
"voice",
|
|
15449
|
+
"models",
|
|
15450
|
+
"{modelId}",
|
|
15451
|
+
"pull"
|
|
15452
|
+
],
|
|
15453
|
+
"maturity": "stable",
|
|
15454
|
+
"layer": "interface",
|
|
15455
|
+
"audiences": [
|
|
15456
|
+
"integrator",
|
|
15457
|
+
"service-agent",
|
|
15458
|
+
"coding-agent"
|
|
15459
|
+
],
|
|
15460
|
+
"interfaces": [
|
|
15461
|
+
{
|
|
15462
|
+
"type": "rest",
|
|
15463
|
+
"target": "POST /v1/voice/models/{modelId}/pull"
|
|
15464
|
+
},
|
|
15465
|
+
{
|
|
15466
|
+
"type": "openapi",
|
|
15467
|
+
"target": "/openapi.json"
|
|
15468
|
+
}
|
|
15469
|
+
],
|
|
15470
|
+
"references": [
|
|
15471
|
+
{
|
|
15472
|
+
"type": "source",
|
|
15473
|
+
"target": "packages/cli/src/api/openapi.ts",
|
|
15474
|
+
"relation": "openapi-source"
|
|
15475
|
+
},
|
|
15476
|
+
{
|
|
15477
|
+
"type": "documentation",
|
|
15478
|
+
"target": "docs/reference/rest-api.md",
|
|
15479
|
+
"relation": "endpoint-inventory"
|
|
15480
|
+
}
|
|
15481
|
+
],
|
|
15482
|
+
"methods": [
|
|
15483
|
+
"POST"
|
|
15484
|
+
],
|
|
15485
|
+
"tags": [
|
|
15486
|
+
"Voice"
|
|
15487
|
+
],
|
|
15488
|
+
"operations": {
|
|
15489
|
+
"post": {
|
|
15490
|
+
"summary": "Install runtime prerequisites and pull one managed TTS model",
|
|
15491
|
+
"tags": [
|
|
15492
|
+
"Voice"
|
|
15493
|
+
],
|
|
15494
|
+
"description": "For voxtral-4b-tts-2603 this creates an Omnius-managed environment and snapshots mistralai/Voxtral-4B-TTS-2603. It does not require shell setup and does not start inference.",
|
|
15495
|
+
"parameters": [
|
|
15496
|
+
{
|
|
15497
|
+
"name": "modelId",
|
|
15498
|
+
"in": "path",
|
|
15499
|
+
"required": true,
|
|
15500
|
+
"schema": {
|
|
15501
|
+
"type": "string"
|
|
15502
|
+
}
|
|
15503
|
+
}
|
|
15504
|
+
],
|
|
15505
|
+
"responses": {
|
|
15506
|
+
"200": {
|
|
15507
|
+
"description": "Weights ready"
|
|
15508
|
+
},
|
|
15509
|
+
"404": {
|
|
15510
|
+
"description": "No managed pull adapter"
|
|
15511
|
+
},
|
|
15512
|
+
"500": {
|
|
15513
|
+
"description": "Dependency installation or download failed"
|
|
15336
15514
|
}
|
|
15337
15515
|
}
|
|
15338
15516
|
}
|
|
@@ -16017,7 +16195,7 @@
|
|
|
16017
16195
|
"tags": [
|
|
16018
16196
|
"Voice"
|
|
16019
16197
|
],
|
|
16020
|
-
"description": "Returns synthesized audio and automatically enables/warms the
|
|
16198
|
+
"description": "Returns synthesized audio and automatically enables/warms the selected runtime. Format `wav` (default) ships a complete WAV with header; `pcm` ships raw Int16 LE PCM. Set model=voxtral-4b-tts-2603 and voice=<preset> for the managed CUDA Voxtral renderer; the OpenAI-compatible alias also accepts model=mistralai/Voxtral-4B-TTS-2603. Other model transactions remain serialized and restore the persistent renderer. Headers report X-Voice-Model, X-Voice-Backend, and X-Sample-Rate.",
|
|
16021
16199
|
"responses": {
|
|
16022
16200
|
"200": {
|
|
16023
16201
|
"description": "Audio bytes (audio/wav or audio/L16). Headers: X-Voice-Model, X-Sample-Rate."
|
package/docs/DISCOVERY.md
CHANGED
|
@@ -219,7 +219,9 @@ Daemon equivalents are `GET /v1/discovery/bootstrap`, `GET /v1/discovery?q=<inte
|
|
|
219
219
|
| `api.v1-voice-clone-refs-filename-rename` | /v1/voice/clone-refs/{filename}/rename | Rename the friendly name (filename unchanged) — body {name} |
|
|
220
220
|
| `api.v1-voice-clone-refs-from-url` | /v1/voice/clone-refs/from-url | Fetch a voice clone reference from a URL |
|
|
221
221
|
| `api.v1-voice-clone-refs-upload` | /v1/voice/clone-refs/upload | Upload a voice clone reference (raw bytes) |
|
|
222
|
-
| `api.v1-voice-models` | /v1/voice/models | List TTS voice models with backend
|
|
222
|
+
| `api.v1-voice-models` | /v1/voice/models | List TTS voice models with backend metadata and managed readiness |
|
|
223
|
+
| `api.v1-voice-models-model-id-deploy` | /v1/voice/models/{modelId}/deploy | Pull and deploy one managed CUDA TTS model |
|
|
224
|
+
| `api.v1-voice-models-model-id-pull` | /v1/voice/models/{modelId}/pull | Install runtime prerequisites and pull one managed TTS model |
|
|
223
225
|
| `api.v1-voice-models-switch` | /v1/voice/models/switch | Switch and enable the active TTS model — body {modelId, enable?} |
|
|
224
226
|
| `api.v1-voice-speak` | /v1/voice/speak | Synthesize text and broadcast to /v1/voicechat/ws clients (no audio in response body) |
|
|
225
227
|
| `api.v1-voice-start` | /v1/voice/start | Enable and warm the daemon voice runtime; optional body {modelId\|model\|voice} |
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "omnius",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.624",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "omnius",
|
|
9
|
-
"version": "1.0.
|
|
9
|
+
"version": "1.0.624",
|
|
10
10
|
"bundleDependencies": [
|
|
11
11
|
"image-to-ascii"
|
|
12
12
|
],
|
package/package.json
CHANGED