omnius 1.0.627 → 1.0.628
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/dist/index.js +9483 -8913
- package/dist/scripts/audio-yamnet-tensorrt-worker.py +335 -0
- package/dist/update-worker.js +30 -19
- package/docs/DISCOVERY.json +232 -2
- package/docs/DISCOVERY.md +3 -0
- package/docs/reference/rest-api.md +3 -0
- package/docs/rest/endpoints/voice-vision.md +41 -0
- package/npm-shrinkwrap.json +14 -14
- package/package.json +2 -2
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Persistent CUDA/TensorRT YAMNet worker for Omnius audio classification.
|
|
3
|
+
|
|
4
|
+
The process deliberately accepts only already-captured WAV paths. It never
|
|
5
|
+
opens an ALSA device and never downloads a model. Omnius bootstrap stages and
|
|
6
|
+
checksums the ONNX model and TensorRT plan before this worker is launched.
|
|
7
|
+
|
|
8
|
+
Protocol: JSON Lines on stdin/stdout. stdout is protocol-only; diagnostics go
|
|
9
|
+
to stderr so callers cannot confuse logs with inference results.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import argparse
|
|
14
|
+
import csv
|
|
15
|
+
import hashlib
|
|
16
|
+
import json
|
|
17
|
+
import os
|
|
18
|
+
import sys
|
|
19
|
+
import time
|
|
20
|
+
import traceback
|
|
21
|
+
import wave
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def emit(payload: dict) -> None:
|
|
25
|
+
sys.stdout.write(json.dumps(payload, separators=(",", ":")) + "\n")
|
|
26
|
+
sys.stdout.flush()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def fail(message: str) -> None:
|
|
30
|
+
emit({"type": "fatal", "error": message})
|
|
31
|
+
raise RuntimeError(message)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def cuda_check(result, label: str):
|
|
35
|
+
"""cuda-python returns either an error enum or (error enum, value)."""
|
|
36
|
+
if isinstance(result, tuple):
|
|
37
|
+
status, value = result[0], result[1:]
|
|
38
|
+
else:
|
|
39
|
+
status, value = result, ()
|
|
40
|
+
if int(status) != 0:
|
|
41
|
+
raise RuntimeError(f"CUDA {label} failed: {status}")
|
|
42
|
+
if len(value) == 0:
|
|
43
|
+
return None
|
|
44
|
+
return value[0] if len(value) == 1 else value
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class CudaPythonRuntime:
|
|
48
|
+
"""Thin adapter over JetPack's cuda-python bindings."""
|
|
49
|
+
def __init__(self, cudart):
|
|
50
|
+
self.cudart = cudart
|
|
51
|
+
|
|
52
|
+
def stream_create(self):
|
|
53
|
+
return cuda_check(self.cudart.cudaStreamCreate(), "stream create")
|
|
54
|
+
|
|
55
|
+
def alloc(self, size, label):
|
|
56
|
+
return cuda_check(self.cudart.cudaMalloc(size), f"{label} allocation")
|
|
57
|
+
|
|
58
|
+
def h2d(self, device, host, stream, label):
|
|
59
|
+
cuda_check(
|
|
60
|
+
self.cudart.cudaMemcpyAsync(
|
|
61
|
+
device,
|
|
62
|
+
host.ctypes.data,
|
|
63
|
+
host.nbytes,
|
|
64
|
+
self.cudart.cudaMemcpyKind.cudaMemcpyHostToDevice,
|
|
65
|
+
stream,
|
|
66
|
+
),
|
|
67
|
+
f"{label} copy",
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
def d2h(self, host, device, stream, label):
|
|
71
|
+
cuda_check(
|
|
72
|
+
self.cudart.cudaMemcpyAsync(
|
|
73
|
+
host.ctypes.data,
|
|
74
|
+
device,
|
|
75
|
+
host.nbytes,
|
|
76
|
+
self.cudart.cudaMemcpyKind.cudaMemcpyDeviceToHost,
|
|
77
|
+
stream,
|
|
78
|
+
),
|
|
79
|
+
f"{label} copy",
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
def synchronize(self, stream):
|
|
83
|
+
cuda_check(self.cudart.cudaStreamSynchronize(stream), "stream synchronize")
|
|
84
|
+
|
|
85
|
+
def free(self, allocation):
|
|
86
|
+
self.cudart.cudaFree(allocation)
|
|
87
|
+
|
|
88
|
+
def device_metadata(self):
|
|
89
|
+
device = cuda_check(self.cudart.cudaGetDevice(), "get device")
|
|
90
|
+
props = cuda_check(self.cudart.cudaGetDeviceProperties(device), "get device properties")
|
|
91
|
+
name = getattr(props, "name", "CUDA device")
|
|
92
|
+
if isinstance(name, bytes):
|
|
93
|
+
name = name.split(b"\0", 1)[0].decode("utf8", "replace")
|
|
94
|
+
return str(name), f"{getattr(props, 'major', 0)}.{getattr(props, 'minor', 0)}"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class PyCudaRuntime:
|
|
98
|
+
"""JetPack also ships python3-pycuda on supported L4T images.
|
|
99
|
+
|
|
100
|
+
Keeping this fallback avoids a generic PyPI CUDA runtime wheel: both
|
|
101
|
+
backends call the CUDA libraries supplied by the installed JetPack image.
|
|
102
|
+
"""
|
|
103
|
+
def __init__(self, cuda):
|
|
104
|
+
cuda.init()
|
|
105
|
+
self.cuda = cuda
|
|
106
|
+
self.device = cuda.Device(0)
|
|
107
|
+
self.context = self.device.make_context()
|
|
108
|
+
|
|
109
|
+
def stream_create(self):
|
|
110
|
+
return self.cuda.Stream()
|
|
111
|
+
|
|
112
|
+
def alloc(self, size, _label):
|
|
113
|
+
return self.cuda.mem_alloc(size)
|
|
114
|
+
|
|
115
|
+
def h2d(self, device, host, stream, _label):
|
|
116
|
+
self.cuda.memcpy_htod_async(device, host, stream)
|
|
117
|
+
|
|
118
|
+
def d2h(self, host, device, stream, _label):
|
|
119
|
+
self.cuda.memcpy_dtoh_async(host, device, stream)
|
|
120
|
+
|
|
121
|
+
def synchronize(self, stream):
|
|
122
|
+
stream.synchronize()
|
|
123
|
+
|
|
124
|
+
def free(self, allocation):
|
|
125
|
+
allocation.free()
|
|
126
|
+
|
|
127
|
+
def stream_handle(self, stream):
|
|
128
|
+
return stream.handle
|
|
129
|
+
|
|
130
|
+
def device_metadata(self):
|
|
131
|
+
major, minor = self.device.compute_capability()
|
|
132
|
+
return self.device.name(), f"{major}.{minor}"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def read_wav(path: str, np):
|
|
136
|
+
started = time.perf_counter()
|
|
137
|
+
with wave.open(path, "rb") as reader:
|
|
138
|
+
channels = reader.getnchannels()
|
|
139
|
+
sample_width = reader.getsampwidth()
|
|
140
|
+
sample_rate = reader.getframerate()
|
|
141
|
+
frames = reader.getnframes()
|
|
142
|
+
compression = reader.getcomptype()
|
|
143
|
+
raw = reader.readframes(frames)
|
|
144
|
+
if compression != "NONE":
|
|
145
|
+
raise RuntimeError("Only uncompressed RIFF/WAV input is supported")
|
|
146
|
+
if channels != 1:
|
|
147
|
+
raise RuntimeError(f"Expected mono WAV from the caller, received {channels} channels")
|
|
148
|
+
if sample_width != 2:
|
|
149
|
+
raise RuntimeError(f"Expected PCM16 WAV from the caller, received {sample_width * 8}-bit samples")
|
|
150
|
+
if sample_rate != 16000:
|
|
151
|
+
raise RuntimeError(f"Expected 16 kHz WAV from the caller, received {sample_rate} Hz")
|
|
152
|
+
waveform = np.frombuffer(raw, dtype="<i2").astype(np.float32) / 32768.0
|
|
153
|
+
return waveform, sample_rate, (time.perf_counter() - started) * 1000
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class YAMNetTensorRT:
|
|
157
|
+
def __init__(self, engine_path: str, class_map_path: str):
|
|
158
|
+
started = time.perf_counter()
|
|
159
|
+
import numpy as np
|
|
160
|
+
import tensorrt as trt
|
|
161
|
+
try:
|
|
162
|
+
from cuda import cudart
|
|
163
|
+
cuda = CudaPythonRuntime(cudart)
|
|
164
|
+
except ImportError:
|
|
165
|
+
import pycuda.driver as pycuda
|
|
166
|
+
cuda = PyCudaRuntime(pycuda)
|
|
167
|
+
|
|
168
|
+
self.np = np
|
|
169
|
+
self.trt = trt
|
|
170
|
+
self.cuda = cuda
|
|
171
|
+
self.logger = trt.Logger(trt.Logger.ERROR)
|
|
172
|
+
with open(engine_path, "rb") as handle:
|
|
173
|
+
serialized = handle.read()
|
|
174
|
+
self.runtime = trt.Runtime(self.logger)
|
|
175
|
+
self.engine = self.runtime.deserialize_cuda_engine(serialized)
|
|
176
|
+
if self.engine is None:
|
|
177
|
+
raise RuntimeError("TensorRT could not deserialize the YAMNet engine")
|
|
178
|
+
self.context = self.engine.create_execution_context()
|
|
179
|
+
if self.context is None:
|
|
180
|
+
raise RuntimeError("TensorRT could not create a YAMNet execution context")
|
|
181
|
+
self.stream = cuda.stream_create()
|
|
182
|
+
self.input_name = None
|
|
183
|
+
self.output_names = []
|
|
184
|
+
for index in range(self.engine.num_io_tensors):
|
|
185
|
+
name = self.engine.get_tensor_name(index)
|
|
186
|
+
if self.engine.get_tensor_mode(name) == trt.TensorIOMode.INPUT:
|
|
187
|
+
self.input_name = name
|
|
188
|
+
else:
|
|
189
|
+
self.output_names.append(name)
|
|
190
|
+
if self.input_name is None or not self.output_names:
|
|
191
|
+
raise RuntimeError("YAMNet TensorRT engine did not expose input and output tensors")
|
|
192
|
+
with open(class_map_path, newline="", encoding="utf8") as handle:
|
|
193
|
+
self.classes = [row["display_name"] for row in csv.DictReader(handle)]
|
|
194
|
+
if len(self.classes) != 521:
|
|
195
|
+
raise RuntimeError(f"YAMNet class map must contain 521 classes; found {len(self.classes)}")
|
|
196
|
+
self.device_name, self.compute_capability = self._device_metadata()
|
|
197
|
+
self.model_load_ms = (time.perf_counter() - started) * 1000
|
|
198
|
+
|
|
199
|
+
def _device_metadata(self):
|
|
200
|
+
# TensorRT uses the CUDA-visible ordinal. Omnius preflight limits that
|
|
201
|
+
# namespace to exactly the approved Jetson GPU before process launch.
|
|
202
|
+
return self.cuda.device_metadata()
|
|
203
|
+
|
|
204
|
+
def warm(self):
|
|
205
|
+
# YAMNet needs >=0.975 seconds. This is a model-load probe, not a
|
|
206
|
+
# captured-audio classification, and it verifies the TensorRT path is
|
|
207
|
+
# actually executable before readiness is reported.
|
|
208
|
+
self.infer(self.np.zeros(15600, dtype=self.np.float32), 1)
|
|
209
|
+
|
|
210
|
+
def infer(self, waveform, top_k: int):
|
|
211
|
+
started = time.perf_counter()
|
|
212
|
+
np = self.np
|
|
213
|
+
if waveform.ndim != 1 or waveform.size < 15600:
|
|
214
|
+
# YAMNet's framing kernel requires 0.975 s; pad silence only to
|
|
215
|
+
# satisfy the model shape, without altering any supplied samples.
|
|
216
|
+
waveform = np.pad(waveform.reshape(-1), (0, max(0, 15600 - waveform.size)))
|
|
217
|
+
waveform = np.ascontiguousarray(waveform.astype(np.float32, copy=False))
|
|
218
|
+
if not self.context.set_input_shape(self.input_name, tuple(waveform.shape)):
|
|
219
|
+
raise RuntimeError(f"TensorRT rejected YAMNet input shape {tuple(waveform.shape)}")
|
|
220
|
+
|
|
221
|
+
allocations = []
|
|
222
|
+
host_outputs = {}
|
|
223
|
+
try:
|
|
224
|
+
input_bytes = waveform.nbytes
|
|
225
|
+
input_device = self.cuda.alloc(input_bytes, "input")
|
|
226
|
+
allocations.append(input_device)
|
|
227
|
+
self.context.set_tensor_address(self.input_name, int(input_device))
|
|
228
|
+
for name in self.output_names:
|
|
229
|
+
shape = tuple(self.context.get_tensor_shape(name))
|
|
230
|
+
if any(int(dim) < 0 for dim in shape):
|
|
231
|
+
raise RuntimeError(f"TensorRT did not resolve output shape for {name}: {shape}")
|
|
232
|
+
dtype = self.trt.nptype(self.engine.get_tensor_dtype(name))
|
|
233
|
+
host = np.empty(shape, dtype=dtype)
|
|
234
|
+
device = self.cuda.alloc(host.nbytes, name)
|
|
235
|
+
allocations.append(device)
|
|
236
|
+
host_outputs[name] = (host, device)
|
|
237
|
+
self.context.set_tensor_address(name, int(device))
|
|
238
|
+
self.cuda.h2d(input_device, waveform, self.stream, "input")
|
|
239
|
+
stream_handle = self.cuda.stream_handle(self.stream) if hasattr(self.cuda, "stream_handle") else self.stream
|
|
240
|
+
if not self.context.execute_async_v3(stream_handle):
|
|
241
|
+
raise RuntimeError("TensorRT YAMNet execute_async_v3 returned false")
|
|
242
|
+
for name, (host, device) in host_outputs.items():
|
|
243
|
+
self.cuda.d2h(host, device, self.stream, name)
|
|
244
|
+
self.cuda.synchronize(self.stream)
|
|
245
|
+
# The converted Google model exposes output_0 with AudioSet scores.
|
|
246
|
+
scores = next((value[0] for name, value in host_outputs.items() if value[0].shape[-1] == 521), None)
|
|
247
|
+
if scores is None:
|
|
248
|
+
raise RuntimeError("TensorRT YAMNet engine did not produce a 521-class score tensor")
|
|
249
|
+
mean_scores = np.mean(scores, axis=0)
|
|
250
|
+
indices = np.argsort(mean_scores)[-top_k:][::-1]
|
|
251
|
+
return [
|
|
252
|
+
{"label": self.classes[int(index)], "confidence": float(mean_scores[int(index)])}
|
|
253
|
+
for index in indices
|
|
254
|
+
], (time.perf_counter() - started) * 1000
|
|
255
|
+
finally:
|
|
256
|
+
for allocation in allocations:
|
|
257
|
+
try:
|
|
258
|
+
self.cuda.free(allocation)
|
|
259
|
+
except Exception:
|
|
260
|
+
pass
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def file_digest(path: str) -> str:
|
|
264
|
+
digest = hashlib.sha256()
|
|
265
|
+
with open(path, "rb") as handle:
|
|
266
|
+
while True:
|
|
267
|
+
chunk = handle.read(1024 * 1024)
|
|
268
|
+
if not chunk:
|
|
269
|
+
break
|
|
270
|
+
digest.update(chunk)
|
|
271
|
+
return "sha256:" + digest.hexdigest()
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def main() -> int:
|
|
275
|
+
parser = argparse.ArgumentParser()
|
|
276
|
+
parser.add_argument("--engine", required=True)
|
|
277
|
+
parser.add_argument("--class-map", required=True)
|
|
278
|
+
parser.add_argument("--model-digest", required=True)
|
|
279
|
+
args = parser.parse_args()
|
|
280
|
+
# CUDA_VISIBLE_DEVICES must be a single approved accelerator before any
|
|
281
|
+
# CUDA/TensorRT import. Jetson's integrated GPU is logical device 0.
|
|
282
|
+
visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip()
|
|
283
|
+
if visible != "0":
|
|
284
|
+
fail("Audio TensorRT worker requires exactly CUDA_VISIBLE_DEVICES=0 on Jetson")
|
|
285
|
+
worker = YAMNetTensorRT(args.engine, args.class_map)
|
|
286
|
+
worker.warm()
|
|
287
|
+
emit({
|
|
288
|
+
"type": "ready",
|
|
289
|
+
"pid": os.getpid(),
|
|
290
|
+
"backend": "tensorrt-fp16",
|
|
291
|
+
"device": worker.device_name,
|
|
292
|
+
"compute_capability": worker.compute_capability,
|
|
293
|
+
"cuda_visible_devices": visible,
|
|
294
|
+
"model": "yamnet",
|
|
295
|
+
"model_digest": args.model_digest,
|
|
296
|
+
"taxonomy": "AudioSet-521",
|
|
297
|
+
"model_load_ms": round(worker.model_load_ms, 3),
|
|
298
|
+
"warmed": True,
|
|
299
|
+
})
|
|
300
|
+
for line in sys.stdin:
|
|
301
|
+
try:
|
|
302
|
+
request = json.loads(line)
|
|
303
|
+
if request.get("type") == "shutdown":
|
|
304
|
+
emit({"type": "stopped"})
|
|
305
|
+
return 0
|
|
306
|
+
if request.get("type") != "classify":
|
|
307
|
+
raise RuntimeError("unknown request type")
|
|
308
|
+
request_id = str(request.get("id") or "")
|
|
309
|
+
file_path = str(request.get("file") or "")
|
|
310
|
+
if not request_id or not file_path:
|
|
311
|
+
raise RuntimeError("classify requires id and file")
|
|
312
|
+
waveform, sample_rate, decode_ms = read_wav(file_path, worker.np)
|
|
313
|
+
classifications, inference_ms = worker.infer(waveform, max(1, min(int(request.get("top_k", 5)), 25)))
|
|
314
|
+
emit({
|
|
315
|
+
"type": "result",
|
|
316
|
+
"id": request_id,
|
|
317
|
+
"success": True,
|
|
318
|
+
"sample_rate_hz": sample_rate,
|
|
319
|
+
"duration_seconds": round(float(waveform.size) / sample_rate, 6),
|
|
320
|
+
"classifications": classifications,
|
|
321
|
+
"timings_ms": {"decode": round(decode_ms, 3), "inference": round(inference_ms, 3)},
|
|
322
|
+
})
|
|
323
|
+
except Exception as exc: # keep the persistent worker alive per request
|
|
324
|
+
emit({
|
|
325
|
+
"type": "result",
|
|
326
|
+
"id": str(locals().get("request", {}).get("id", "")),
|
|
327
|
+
"success": False,
|
|
328
|
+
"error": str(exc),
|
|
329
|
+
})
|
|
330
|
+
print(traceback.format_exc(), file=sys.stderr, flush=True)
|
|
331
|
+
return 0
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
if __name__ == "__main__":
|
|
335
|
+
raise SystemExit(main())
|
package/dist/update-worker.js
CHANGED
|
@@ -52,6 +52,13 @@ var init_model_broker = __esm({
|
|
|
52
52
|
}
|
|
53
53
|
});
|
|
54
54
|
|
|
55
|
+
// packages/execution/dist/process-async.js
|
|
56
|
+
var init_process_async = __esm({
|
|
57
|
+
"packages/execution/dist/process-async.js"() {
|
|
58
|
+
"use strict";
|
|
59
|
+
}
|
|
60
|
+
});
|
|
61
|
+
|
|
55
62
|
// packages/execution/dist/venv-paths.js
|
|
56
63
|
var isWin;
|
|
57
64
|
var init_venv_paths = __esm({
|
|
@@ -61,6 +68,16 @@ var init_venv_paths = __esm({
|
|
|
61
68
|
}
|
|
62
69
|
});
|
|
63
70
|
|
|
71
|
+
// packages/execution/dist/tools/model-store.js
|
|
72
|
+
import { homedir, platform } from "node:os";
|
|
73
|
+
var platformId;
|
|
74
|
+
var init_model_store = __esm({
|
|
75
|
+
"packages/execution/dist/tools/model-store.js"() {
|
|
76
|
+
"use strict";
|
|
77
|
+
platformId = platform();
|
|
78
|
+
}
|
|
79
|
+
});
|
|
80
|
+
|
|
64
81
|
// packages/execution/dist/process-kill.js
|
|
65
82
|
import { execSync } from "node:child_process";
|
|
66
83
|
function killProcessTree(pid, signal = "SIGKILL") {
|
|
@@ -104,7 +121,7 @@ var init_process_kill = __esm({
|
|
|
104
121
|
// packages/execution/dist/process-lifecycle.js
|
|
105
122
|
import { randomUUID, createHash } from "node:crypto";
|
|
106
123
|
import { appendFileSync, existsSync, mkdirSync, readFileSync, readlinkSync, renameSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
107
|
-
import { homedir } from "node:os";
|
|
124
|
+
import { homedir as homedir2 } from "node:os";
|
|
108
125
|
import { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
109
126
|
function nowFrom(system) {
|
|
110
127
|
return system?.now?.() ?? Date.now();
|
|
@@ -113,7 +130,7 @@ function realSleep(ms) {
|
|
|
113
130
|
return new Promise((resolveSleep) => setTimeout(resolveSleep, ms));
|
|
114
131
|
}
|
|
115
132
|
function registryDir(options) {
|
|
116
|
-
return options?.storeDir || process.env["OMNIUS_PROCESS_REGISTRY_DIR"] || join(process.env["OMNIUS_HOME"] || join(
|
|
133
|
+
return options?.storeDir || process.env["OMNIUS_PROCESS_REGISTRY_DIR"] || join(process.env["OMNIUS_HOME"] || join(homedir2(), ".omnius"), "processes");
|
|
117
134
|
}
|
|
118
135
|
function ensureStoreDir(options) {
|
|
119
136
|
const dir = registryDir(options);
|
|
@@ -499,13 +516,6 @@ var init_system_deps = __esm({
|
|
|
499
516
|
}
|
|
500
517
|
});
|
|
501
518
|
|
|
502
|
-
// packages/execution/dist/process-async.js
|
|
503
|
-
var init_process_async = __esm({
|
|
504
|
-
"packages/execution/dist/process-async.js"() {
|
|
505
|
-
"use strict";
|
|
506
|
-
}
|
|
507
|
-
});
|
|
508
|
-
|
|
509
519
|
// packages/execution/dist/tools/cuda-device-filter.js
|
|
510
520
|
var init_cuda_device_filter = __esm({
|
|
511
521
|
"packages/execution/dist/tools/cuda-device-filter.js"() {
|
|
@@ -514,16 +524,6 @@ var init_cuda_device_filter = __esm({
|
|
|
514
524
|
}
|
|
515
525
|
});
|
|
516
526
|
|
|
517
|
-
// packages/execution/dist/tools/model-store.js
|
|
518
|
-
import { homedir as homedir2, platform } from "node:os";
|
|
519
|
-
var platformId;
|
|
520
|
-
var init_model_store = __esm({
|
|
521
|
-
"packages/execution/dist/tools/model-store.js"() {
|
|
522
|
-
"use strict";
|
|
523
|
-
platformId = platform();
|
|
524
|
-
}
|
|
525
|
-
});
|
|
526
|
-
|
|
527
527
|
// packages/execution/dist/tools/camera-capture.js
|
|
528
528
|
var DEFAULT_CAMERA_PROFILE, CAMERA_PROFILES;
|
|
529
529
|
var init_camera_capture = __esm({
|
|
@@ -245069,10 +245069,21 @@ import { dirname as dirname8 } from "node:path";
|
|
|
245069
245069
|
init_model_broker();
|
|
245070
245070
|
init_jetson_monitor();
|
|
245071
245071
|
|
|
245072
|
+
// packages/execution/dist/audio-classifier-runtime.js
|
|
245073
|
+
init_process_async();
|
|
245074
|
+
|
|
245072
245075
|
// packages/execution/dist/python-cuda-runtime.js
|
|
245073
245076
|
init_jetson_monitor();
|
|
245074
245077
|
init_venv_paths();
|
|
245075
245078
|
|
|
245079
|
+
// packages/execution/dist/audio-classifier-runtime.js
|
|
245080
|
+
init_jetson_monitor();
|
|
245081
|
+
init_model_store();
|
|
245082
|
+
var MODEL_REVISION = "ac2ca3bd45d12ec1f19f1144205ea529b4e9dedf";
|
|
245083
|
+
var MODEL_URL = `https://huggingface.co/zeropointnine/yamnet-onnx/resolve/${MODEL_REVISION}/yamnet.onnx`;
|
|
245084
|
+
var CLASS_MAP_URL = `https://huggingface.co/zeropointnine/yamnet-onnx/resolve/${MODEL_REVISION}/yamnet_class_map.csv`;
|
|
245085
|
+
var SETUP_TIMEOUT_MS = 20 * 6e4;
|
|
245086
|
+
|
|
245076
245087
|
// packages/execution/dist/broker-mediated-backend.js
|
|
245077
245088
|
init_model_broker();
|
|
245078
245089
|
|
package/docs/DISCOVERY.json
CHANGED
|
@@ -3832,7 +3832,7 @@
|
|
|
3832
3832
|
"tags": [
|
|
3833
3833
|
"ASR"
|
|
3834
3834
|
],
|
|
3835
|
-
"description": "Body: {modelId?, device?}. Installs the managed runtime and pulls the requested model. Whisper, Nemotron, and Voxtral load the requested weights to validate exact CUDA placement; VibeVoice stores its pinned model and tokenizer snapshot under Omnius' ASR runtime directories.",
|
|
3835
|
+
"description": "Body: {modelId?, device?}. Installs the managed runtime and pulls the requested model. Whisper, Nemotron, and Voxtral load the requested weights to validate exact CUDA placement; VibeVoice stores its pinned microsoft/VibeVoice-ASR model and tokenizer snapshot under Omnius' ASR runtime directories.",
|
|
3836
3836
|
"parameters": [
|
|
3837
3837
|
{
|
|
3838
3838
|
"name": "engineId",
|
|
@@ -3936,7 +3936,7 @@
|
|
|
3936
3936
|
"tags": [
|
|
3937
3937
|
"ASR"
|
|
3938
3938
|
],
|
|
3939
|
-
"description": "Body: {engineId, modelId?, setup?, device?}. Selection is validated against the registry. setup=true pulls the exact managed model and verifies CUDA placement before activation. Hardware evidence uses nvidia-smi on discrete Linux and NVIDIA's documented tegrastats plus a CUDA Torch property probe on Jetson/L4T.",
|
|
3939
|
+
"description": "Body: {engineId, modelId?, setup?, device?}. Selection is validated against the registry and fails fail-closed: no unavailable engine, missing weights, or unverified CUDA placement can silently replace the active selection. setup=true pulls the exact managed model and verifies CUDA placement before activation. Hardware evidence uses nvidia-smi on discrete Linux and NVIDIA's documented tegrastats plus a CUDA Torch property probe on Jetson/L4T.",
|
|
3940
3940
|
"responses": {
|
|
3941
3941
|
"200": {
|
|
3942
3942
|
"description": "Exact selection activated"
|
|
@@ -4184,6 +4184,236 @@
|
|
|
4184
4184
|
"packages/cli/src/api/openapi.ts"
|
|
4185
4185
|
]
|
|
4186
4186
|
},
|
|
4187
|
+
{
|
|
4188
|
+
"id": "api.v1-audio-classify",
|
|
4189
|
+
"kind": "api",
|
|
4190
|
+
"title": "/v1/audio/classify",
|
|
4191
|
+
"summary": "Classify a supplied WAV with persistent CUDA/TensorRT YAMNet",
|
|
4192
|
+
"aliases": [
|
|
4193
|
+
"/v1/audio/classify"
|
|
4194
|
+
],
|
|
4195
|
+
"keywords": [
|
|
4196
|
+
"rest",
|
|
4197
|
+
"openapi",
|
|
4198
|
+
"POST",
|
|
4199
|
+
"Audio",
|
|
4200
|
+
"v1",
|
|
4201
|
+
"audio",
|
|
4202
|
+
"classify"
|
|
4203
|
+
],
|
|
4204
|
+
"maturity": "stable",
|
|
4205
|
+
"layer": "interface",
|
|
4206
|
+
"audiences": [
|
|
4207
|
+
"integrator",
|
|
4208
|
+
"service-agent",
|
|
4209
|
+
"coding-agent"
|
|
4210
|
+
],
|
|
4211
|
+
"interfaces": [
|
|
4212
|
+
{
|
|
4213
|
+
"type": "rest",
|
|
4214
|
+
"target": "POST /v1/audio/classify"
|
|
4215
|
+
},
|
|
4216
|
+
{
|
|
4217
|
+
"type": "openapi",
|
|
4218
|
+
"target": "/openapi.json"
|
|
4219
|
+
}
|
|
4220
|
+
],
|
|
4221
|
+
"references": [
|
|
4222
|
+
{
|
|
4223
|
+
"type": "source",
|
|
4224
|
+
"target": "packages/cli/src/api/openapi.ts",
|
|
4225
|
+
"relation": "openapi-source"
|
|
4226
|
+
},
|
|
4227
|
+
{
|
|
4228
|
+
"type": "documentation",
|
|
4229
|
+
"target": "docs/reference/rest-api.md",
|
|
4230
|
+
"relation": "endpoint-inventory"
|
|
4231
|
+
}
|
|
4232
|
+
],
|
|
4233
|
+
"methods": [
|
|
4234
|
+
"POST"
|
|
4235
|
+
],
|
|
4236
|
+
"tags": [
|
|
4237
|
+
"Audio"
|
|
4238
|
+
],
|
|
4239
|
+
"operations": {
|
|
4240
|
+
"post": {
|
|
4241
|
+
"summary": "Classify a supplied WAV with persistent CUDA/TensorRT YAMNet",
|
|
4242
|
+
"tags": [
|
|
4243
|
+
"Audio"
|
|
4244
|
+
],
|
|
4245
|
+
"description": "Alias of POST /v1/tools/audio_analyze/call. Body: {args:{action:'classify',file:'/absolute/path.wav',top_k:5},timeout_ms:90000}. Input must be caller-supplied mono PCM16/16 kHz WAV; no microphone capture, dependency install, or model download occurs on this path. Returns result.data.classifications.",
|
|
4246
|
+
"responses": {
|
|
4247
|
+
"200": {
|
|
4248
|
+
"description": "Structured AudioSet-521 classification"
|
|
4249
|
+
},
|
|
4250
|
+
"409": {
|
|
4251
|
+
"description": "One request runs and one waits; additional work is busy"
|
|
4252
|
+
},
|
|
4253
|
+
"503": {
|
|
4254
|
+
"description": "Classifier not ready"
|
|
4255
|
+
}
|
|
4256
|
+
}
|
|
4257
|
+
}
|
|
4258
|
+
},
|
|
4259
|
+
"source_of_truth": [
|
|
4260
|
+
"GET /openapi.json",
|
|
4261
|
+
"packages/cli/src/api/openapi.ts"
|
|
4262
|
+
]
|
|
4263
|
+
},
|
|
4264
|
+
{
|
|
4265
|
+
"id": "api.v1-audio-classify-health",
|
|
4266
|
+
"kind": "api",
|
|
4267
|
+
"title": "/v1/audio/classify/health",
|
|
4268
|
+
"summary": "Jetson CUDA/TensorRT YAMNet readiness",
|
|
4269
|
+
"aliases": [
|
|
4270
|
+
"/v1/audio/classify/health"
|
|
4271
|
+
],
|
|
4272
|
+
"keywords": [
|
|
4273
|
+
"rest",
|
|
4274
|
+
"openapi",
|
|
4275
|
+
"GET",
|
|
4276
|
+
"Audio",
|
|
4277
|
+
"v1",
|
|
4278
|
+
"audio",
|
|
4279
|
+
"classify",
|
|
4280
|
+
"health"
|
|
4281
|
+
],
|
|
4282
|
+
"maturity": "stable",
|
|
4283
|
+
"layer": "interface",
|
|
4284
|
+
"audiences": [
|
|
4285
|
+
"integrator",
|
|
4286
|
+
"service-agent",
|
|
4287
|
+
"coding-agent"
|
|
4288
|
+
],
|
|
4289
|
+
"interfaces": [
|
|
4290
|
+
{
|
|
4291
|
+
"type": "rest",
|
|
4292
|
+
"target": "GET /v1/audio/classify/health"
|
|
4293
|
+
},
|
|
4294
|
+
{
|
|
4295
|
+
"type": "openapi",
|
|
4296
|
+
"target": "/openapi.json"
|
|
4297
|
+
}
|
|
4298
|
+
],
|
|
4299
|
+
"references": [
|
|
4300
|
+
{
|
|
4301
|
+
"type": "source",
|
|
4302
|
+
"target": "packages/cli/src/api/openapi.ts",
|
|
4303
|
+
"relation": "openapi-source"
|
|
4304
|
+
},
|
|
4305
|
+
{
|
|
4306
|
+
"type": "documentation",
|
|
4307
|
+
"target": "docs/reference/rest-api.md",
|
|
4308
|
+
"relation": "endpoint-inventory"
|
|
4309
|
+
}
|
|
4310
|
+
],
|
|
4311
|
+
"methods": [
|
|
4312
|
+
"GET"
|
|
4313
|
+
],
|
|
4314
|
+
"tags": [
|
|
4315
|
+
"Audio"
|
|
4316
|
+
],
|
|
4317
|
+
"operations": {
|
|
4318
|
+
"get": {
|
|
4319
|
+
"summary": "Jetson CUDA/TensorRT YAMNet readiness",
|
|
4320
|
+
"tags": [
|
|
4321
|
+
"Audio"
|
|
4322
|
+
],
|
|
4323
|
+
"description": "Reports real pinned-model, TensorRT worker, CUDA placement, warmup, queue, and error state. General /health is not an audio readiness signal.",
|
|
4324
|
+
"responses": {
|
|
4325
|
+
"200": {
|
|
4326
|
+
"description": "Warm audio classifier readiness"
|
|
4327
|
+
},
|
|
4328
|
+
"503": {
|
|
4329
|
+
"description": "Audio classifier is installing, stopped, or unavailable"
|
|
4330
|
+
}
|
|
4331
|
+
}
|
|
4332
|
+
}
|
|
4333
|
+
},
|
|
4334
|
+
"source_of_truth": [
|
|
4335
|
+
"GET /openapi.json",
|
|
4336
|
+
"packages/cli/src/api/openapi.ts"
|
|
4337
|
+
]
|
|
4338
|
+
},
|
|
4339
|
+
{
|
|
4340
|
+
"id": "api.v1-audio-classify-setup",
|
|
4341
|
+
"kind": "api",
|
|
4342
|
+
"title": "/v1/audio/classify/setup",
|
|
4343
|
+
"summary": "Provision and warm Jetson CUDA/TensorRT YAMNet",
|
|
4344
|
+
"aliases": [
|
|
4345
|
+
"/v1/audio/classify/setup"
|
|
4346
|
+
],
|
|
4347
|
+
"keywords": [
|
|
4348
|
+
"rest",
|
|
4349
|
+
"openapi",
|
|
4350
|
+
"POST",
|
|
4351
|
+
"Audio",
|
|
4352
|
+
"v1",
|
|
4353
|
+
"audio",
|
|
4354
|
+
"classify",
|
|
4355
|
+
"setup"
|
|
4356
|
+
],
|
|
4357
|
+
"maturity": "stable",
|
|
4358
|
+
"layer": "interface",
|
|
4359
|
+
"audiences": [
|
|
4360
|
+
"integrator",
|
|
4361
|
+
"service-agent",
|
|
4362
|
+
"coding-agent"
|
|
4363
|
+
],
|
|
4364
|
+
"interfaces": [
|
|
4365
|
+
{
|
|
4366
|
+
"type": "rest",
|
|
4367
|
+
"target": "POST /v1/audio/classify/setup"
|
|
4368
|
+
},
|
|
4369
|
+
{
|
|
4370
|
+
"type": "openapi",
|
|
4371
|
+
"target": "/openapi.json"
|
|
4372
|
+
}
|
|
4373
|
+
],
|
|
4374
|
+
"references": [
|
|
4375
|
+
{
|
|
4376
|
+
"type": "source",
|
|
4377
|
+
"target": "packages/cli/src/api/openapi.ts",
|
|
4378
|
+
"relation": "openapi-source"
|
|
4379
|
+
},
|
|
4380
|
+
{
|
|
4381
|
+
"type": "documentation",
|
|
4382
|
+
"target": "docs/reference/rest-api.md",
|
|
4383
|
+
"relation": "endpoint-inventory"
|
|
4384
|
+
}
|
|
4385
|
+
],
|
|
4386
|
+
"methods": [
|
|
4387
|
+
"POST"
|
|
4388
|
+
],
|
|
4389
|
+
"tags": [
|
|
4390
|
+
"Audio"
|
|
4391
|
+
],
|
|
4392
|
+
"operations": {
|
|
4393
|
+
"post": {
|
|
4394
|
+
"summary": "Provision and warm Jetson CUDA/TensorRT YAMNet",
|
|
4395
|
+
"tags": [
|
|
4396
|
+
"Audio"
|
|
4397
|
+
],
|
|
4398
|
+
"description": "Admin-only. Downloads checksummed ONNX/class-map artifacts once, builds a JetPack-specific FP16 TensorRT engine, verifies CUDA placement, and warms the persistent worker. It never installs a generic CUDA/Torch wheel.",
|
|
4399
|
+
"responses": {
|
|
4400
|
+
"200": {
|
|
4401
|
+
"description": "Audio classifier installed and warm"
|
|
4402
|
+
},
|
|
4403
|
+
"403": {
|
|
4404
|
+
"description": "Admin scope required"
|
|
4405
|
+
},
|
|
4406
|
+
"500": {
|
|
4407
|
+
"description": "JetPack/TensorRT/CUDA preflight or setup failed"
|
|
4408
|
+
}
|
|
4409
|
+
}
|
|
4410
|
+
}
|
|
4411
|
+
},
|
|
4412
|
+
"source_of_truth": [
|
|
4413
|
+
"GET /openapi.json",
|
|
4414
|
+
"packages/cli/src/api/openapi.ts"
|
|
4415
|
+
]
|
|
4416
|
+
},
|
|
4187
4417
|
{
|
|
4188
4418
|
"id": "api.v1-audio-embed",
|
|
4189
4419
|
"kind": "api",
|