omnius 1.0.627 → 1.0.628

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,335 @@
1
+ #!/usr/bin/env python3
2
+ """Persistent CUDA/TensorRT YAMNet worker for Omnius audio classification.
3
+
4
+ The process deliberately accepts only already-captured WAV paths. It never
5
+ opens an ALSA device and never downloads a model. Omnius bootstrap stages and
6
+ checksums the ONNX model and TensorRT plan before this worker is launched.
7
+
8
+ Protocol: JSON Lines on stdin/stdout. stdout is protocol-only; diagnostics go
9
+ to stderr so callers cannot confuse logs with inference results.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import argparse
14
+ import csv
15
+ import hashlib
16
+ import json
17
+ import os
18
+ import sys
19
+ import time
20
+ import traceback
21
+ import wave
22
+
23
+
24
+ def emit(payload: dict) -> None:
25
+ sys.stdout.write(json.dumps(payload, separators=(",", ":")) + "\n")
26
+ sys.stdout.flush()
27
+
28
+
29
+ def fail(message: str) -> None:
30
+ emit({"type": "fatal", "error": message})
31
+ raise RuntimeError(message)
32
+
33
+
34
+ def cuda_check(result, label: str):
35
+ """cuda-python returns either an error enum or (error enum, value)."""
36
+ if isinstance(result, tuple):
37
+ status, value = result[0], result[1:]
38
+ else:
39
+ status, value = result, ()
40
+ if int(status) != 0:
41
+ raise RuntimeError(f"CUDA {label} failed: {status}")
42
+ if len(value) == 0:
43
+ return None
44
+ return value[0] if len(value) == 1 else value
45
+
46
+
47
+ class CudaPythonRuntime:
48
+ """Thin adapter over JetPack's cuda-python bindings."""
49
+ def __init__(self, cudart):
50
+ self.cudart = cudart
51
+
52
+ def stream_create(self):
53
+ return cuda_check(self.cudart.cudaStreamCreate(), "stream create")
54
+
55
+ def alloc(self, size, label):
56
+ return cuda_check(self.cudart.cudaMalloc(size), f"{label} allocation")
57
+
58
+ def h2d(self, device, host, stream, label):
59
+ cuda_check(
60
+ self.cudart.cudaMemcpyAsync(
61
+ device,
62
+ host.ctypes.data,
63
+ host.nbytes,
64
+ self.cudart.cudaMemcpyKind.cudaMemcpyHostToDevice,
65
+ stream,
66
+ ),
67
+ f"{label} copy",
68
+ )
69
+
70
+ def d2h(self, host, device, stream, label):
71
+ cuda_check(
72
+ self.cudart.cudaMemcpyAsync(
73
+ host.ctypes.data,
74
+ device,
75
+ host.nbytes,
76
+ self.cudart.cudaMemcpyKind.cudaMemcpyDeviceToHost,
77
+ stream,
78
+ ),
79
+ f"{label} copy",
80
+ )
81
+
82
+ def synchronize(self, stream):
83
+ cuda_check(self.cudart.cudaStreamSynchronize(stream), "stream synchronize")
84
+
85
+ def free(self, allocation):
86
+ self.cudart.cudaFree(allocation)
87
+
88
+ def device_metadata(self):
89
+ device = cuda_check(self.cudart.cudaGetDevice(), "get device")
90
+ props = cuda_check(self.cudart.cudaGetDeviceProperties(device), "get device properties")
91
+ name = getattr(props, "name", "CUDA device")
92
+ if isinstance(name, bytes):
93
+ name = name.split(b"\0", 1)[0].decode("utf8", "replace")
94
+ return str(name), f"{getattr(props, 'major', 0)}.{getattr(props, 'minor', 0)}"
95
+
96
+
97
+ class PyCudaRuntime:
98
+ """JetPack also ships python3-pycuda on supported L4T images.
99
+
100
+ Keeping this fallback avoids a generic PyPI CUDA runtime wheel: both
101
+ backends call the CUDA libraries supplied by the installed JetPack image.
102
+ """
103
+ def __init__(self, cuda):
104
+ cuda.init()
105
+ self.cuda = cuda
106
+ self.device = cuda.Device(0)
107
+ self.context = self.device.make_context()
108
+
109
+ def stream_create(self):
110
+ return self.cuda.Stream()
111
+
112
+ def alloc(self, size, _label):
113
+ return self.cuda.mem_alloc(size)
114
+
115
+ def h2d(self, device, host, stream, _label):
116
+ self.cuda.memcpy_htod_async(device, host, stream)
117
+
118
+ def d2h(self, host, device, stream, _label):
119
+ self.cuda.memcpy_dtoh_async(host, device, stream)
120
+
121
+ def synchronize(self, stream):
122
+ stream.synchronize()
123
+
124
+ def free(self, allocation):
125
+ allocation.free()
126
+
127
+ def stream_handle(self, stream):
128
+ return stream.handle
129
+
130
+ def device_metadata(self):
131
+ major, minor = self.device.compute_capability()
132
+ return self.device.name(), f"{major}.{minor}"
133
+
134
+
135
+ def read_wav(path: str, np):
136
+ started = time.perf_counter()
137
+ with wave.open(path, "rb") as reader:
138
+ channels = reader.getnchannels()
139
+ sample_width = reader.getsampwidth()
140
+ sample_rate = reader.getframerate()
141
+ frames = reader.getnframes()
142
+ compression = reader.getcomptype()
143
+ raw = reader.readframes(frames)
144
+ if compression != "NONE":
145
+ raise RuntimeError("Only uncompressed RIFF/WAV input is supported")
146
+ if channels != 1:
147
+ raise RuntimeError(f"Expected mono WAV from the caller, received {channels} channels")
148
+ if sample_width != 2:
149
+ raise RuntimeError(f"Expected PCM16 WAV from the caller, received {sample_width * 8}-bit samples")
150
+ if sample_rate != 16000:
151
+ raise RuntimeError(f"Expected 16 kHz WAV from the caller, received {sample_rate} Hz")
152
+ waveform = np.frombuffer(raw, dtype="<i2").astype(np.float32) / 32768.0
153
+ return waveform, sample_rate, (time.perf_counter() - started) * 1000
154
+
155
+
156
+ class YAMNetTensorRT:
157
+ def __init__(self, engine_path: str, class_map_path: str):
158
+ started = time.perf_counter()
159
+ import numpy as np
160
+ import tensorrt as trt
161
+ try:
162
+ from cuda import cudart
163
+ cuda = CudaPythonRuntime(cudart)
164
+ except ImportError:
165
+ import pycuda.driver as pycuda
166
+ cuda = PyCudaRuntime(pycuda)
167
+
168
+ self.np = np
169
+ self.trt = trt
170
+ self.cuda = cuda
171
+ self.logger = trt.Logger(trt.Logger.ERROR)
172
+ with open(engine_path, "rb") as handle:
173
+ serialized = handle.read()
174
+ self.runtime = trt.Runtime(self.logger)
175
+ self.engine = self.runtime.deserialize_cuda_engine(serialized)
176
+ if self.engine is None:
177
+ raise RuntimeError("TensorRT could not deserialize the YAMNet engine")
178
+ self.context = self.engine.create_execution_context()
179
+ if self.context is None:
180
+ raise RuntimeError("TensorRT could not create a YAMNet execution context")
181
+ self.stream = cuda.stream_create()
182
+ self.input_name = None
183
+ self.output_names = []
184
+ for index in range(self.engine.num_io_tensors):
185
+ name = self.engine.get_tensor_name(index)
186
+ if self.engine.get_tensor_mode(name) == trt.TensorIOMode.INPUT:
187
+ self.input_name = name
188
+ else:
189
+ self.output_names.append(name)
190
+ if self.input_name is None or not self.output_names:
191
+ raise RuntimeError("YAMNet TensorRT engine did not expose input and output tensors")
192
+ with open(class_map_path, newline="", encoding="utf8") as handle:
193
+ self.classes = [row["display_name"] for row in csv.DictReader(handle)]
194
+ if len(self.classes) != 521:
195
+ raise RuntimeError(f"YAMNet class map must contain 521 classes; found {len(self.classes)}")
196
+ self.device_name, self.compute_capability = self._device_metadata()
197
+ self.model_load_ms = (time.perf_counter() - started) * 1000
198
+
199
+ def _device_metadata(self):
200
+ # TensorRT uses the CUDA-visible ordinal. Omnius preflight limits that
201
+ # namespace to exactly the approved Jetson GPU before process launch.
202
+ return self.cuda.device_metadata()
203
+
204
+ def warm(self):
205
+ # YAMNet needs >=0.975 seconds. This is a model-load probe, not a
206
+ # captured-audio classification, and it verifies the TensorRT path is
207
+ # actually executable before readiness is reported.
208
+ self.infer(self.np.zeros(15600, dtype=self.np.float32), 1)
209
+
210
+ def infer(self, waveform, top_k: int):
211
+ started = time.perf_counter()
212
+ np = self.np
213
+ if waveform.ndim != 1 or waveform.size < 15600:
214
+ # YAMNet's framing kernel requires 0.975 s; pad silence only to
215
+ # satisfy the model shape, without altering any supplied samples.
216
+ waveform = np.pad(waveform.reshape(-1), (0, max(0, 15600 - waveform.size)))
217
+ waveform = np.ascontiguousarray(waveform.astype(np.float32, copy=False))
218
+ if not self.context.set_input_shape(self.input_name, tuple(waveform.shape)):
219
+ raise RuntimeError(f"TensorRT rejected YAMNet input shape {tuple(waveform.shape)}")
220
+
221
+ allocations = []
222
+ host_outputs = {}
223
+ try:
224
+ input_bytes = waveform.nbytes
225
+ input_device = self.cuda.alloc(input_bytes, "input")
226
+ allocations.append(input_device)
227
+ self.context.set_tensor_address(self.input_name, int(input_device))
228
+ for name in self.output_names:
229
+ shape = tuple(self.context.get_tensor_shape(name))
230
+ if any(int(dim) < 0 for dim in shape):
231
+ raise RuntimeError(f"TensorRT did not resolve output shape for {name}: {shape}")
232
+ dtype = self.trt.nptype(self.engine.get_tensor_dtype(name))
233
+ host = np.empty(shape, dtype=dtype)
234
+ device = self.cuda.alloc(host.nbytes, name)
235
+ allocations.append(device)
236
+ host_outputs[name] = (host, device)
237
+ self.context.set_tensor_address(name, int(device))
238
+ self.cuda.h2d(input_device, waveform, self.stream, "input")
239
+ stream_handle = self.cuda.stream_handle(self.stream) if hasattr(self.cuda, "stream_handle") else self.stream
240
+ if not self.context.execute_async_v3(stream_handle):
241
+ raise RuntimeError("TensorRT YAMNet execute_async_v3 returned false")
242
+ for name, (host, device) in host_outputs.items():
243
+ self.cuda.d2h(host, device, self.stream, name)
244
+ self.cuda.synchronize(self.stream)
245
+ # The converted Google model exposes output_0 with AudioSet scores.
246
+ scores = next((value[0] for name, value in host_outputs.items() if value[0].shape[-1] == 521), None)
247
+ if scores is None:
248
+ raise RuntimeError("TensorRT YAMNet engine did not produce a 521-class score tensor")
249
+ mean_scores = np.mean(scores, axis=0)
250
+ indices = np.argsort(mean_scores)[-top_k:][::-1]
251
+ return [
252
+ {"label": self.classes[int(index)], "confidence": float(mean_scores[int(index)])}
253
+ for index in indices
254
+ ], (time.perf_counter() - started) * 1000
255
+ finally:
256
+ for allocation in allocations:
257
+ try:
258
+ self.cuda.free(allocation)
259
+ except Exception:
260
+ pass
261
+
262
+
263
+ def file_digest(path: str) -> str:
264
+ digest = hashlib.sha256()
265
+ with open(path, "rb") as handle:
266
+ while True:
267
+ chunk = handle.read(1024 * 1024)
268
+ if not chunk:
269
+ break
270
+ digest.update(chunk)
271
+ return "sha256:" + digest.hexdigest()
272
+
273
+
274
+ def main() -> int:
275
+ parser = argparse.ArgumentParser()
276
+ parser.add_argument("--engine", required=True)
277
+ parser.add_argument("--class-map", required=True)
278
+ parser.add_argument("--model-digest", required=True)
279
+ args = parser.parse_args()
280
+ # CUDA_VISIBLE_DEVICES must be a single approved accelerator before any
281
+ # CUDA/TensorRT import. Jetson's integrated GPU is logical device 0.
282
+ visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip()
283
+ if visible != "0":
284
+ fail("Audio TensorRT worker requires exactly CUDA_VISIBLE_DEVICES=0 on Jetson")
285
+ worker = YAMNetTensorRT(args.engine, args.class_map)
286
+ worker.warm()
287
+ emit({
288
+ "type": "ready",
289
+ "pid": os.getpid(),
290
+ "backend": "tensorrt-fp16",
291
+ "device": worker.device_name,
292
+ "compute_capability": worker.compute_capability,
293
+ "cuda_visible_devices": visible,
294
+ "model": "yamnet",
295
+ "model_digest": args.model_digest,
296
+ "taxonomy": "AudioSet-521",
297
+ "model_load_ms": round(worker.model_load_ms, 3),
298
+ "warmed": True,
299
+ })
300
+ for line in sys.stdin:
301
+ try:
302
+ request = json.loads(line)
303
+ if request.get("type") == "shutdown":
304
+ emit({"type": "stopped"})
305
+ return 0
306
+ if request.get("type") != "classify":
307
+ raise RuntimeError("unknown request type")
308
+ request_id = str(request.get("id") or "")
309
+ file_path = str(request.get("file") or "")
310
+ if not request_id or not file_path:
311
+ raise RuntimeError("classify requires id and file")
312
+ waveform, sample_rate, decode_ms = read_wav(file_path, worker.np)
313
+ classifications, inference_ms = worker.infer(waveform, max(1, min(int(request.get("top_k", 5)), 25)))
314
+ emit({
315
+ "type": "result",
316
+ "id": request_id,
317
+ "success": True,
318
+ "sample_rate_hz": sample_rate,
319
+ "duration_seconds": round(float(waveform.size) / sample_rate, 6),
320
+ "classifications": classifications,
321
+ "timings_ms": {"decode": round(decode_ms, 3), "inference": round(inference_ms, 3)},
322
+ })
323
+ except Exception as exc: # keep the persistent worker alive per request
324
+ emit({
325
+ "type": "result",
326
+ "id": str(locals().get("request", {}).get("id", "")),
327
+ "success": False,
328
+ "error": str(exc),
329
+ })
330
+ print(traceback.format_exc(), file=sys.stderr, flush=True)
331
+ return 0
332
+
333
+
334
+ if __name__ == "__main__":
335
+ raise SystemExit(main())
@@ -52,6 +52,13 @@ var init_model_broker = __esm({
52
52
  }
53
53
  });
54
54
 
55
+ // packages/execution/dist/process-async.js
56
+ var init_process_async = __esm({
57
+ "packages/execution/dist/process-async.js"() {
58
+ "use strict";
59
+ }
60
+ });
61
+
55
62
  // packages/execution/dist/venv-paths.js
56
63
  var isWin;
57
64
  var init_venv_paths = __esm({
@@ -61,6 +68,16 @@ var init_venv_paths = __esm({
61
68
  }
62
69
  });
63
70
 
71
+ // packages/execution/dist/tools/model-store.js
72
+ import { homedir, platform } from "node:os";
73
+ var platformId;
74
+ var init_model_store = __esm({
75
+ "packages/execution/dist/tools/model-store.js"() {
76
+ "use strict";
77
+ platformId = platform();
78
+ }
79
+ });
80
+
64
81
  // packages/execution/dist/process-kill.js
65
82
  import { execSync } from "node:child_process";
66
83
  function killProcessTree(pid, signal = "SIGKILL") {
@@ -104,7 +121,7 @@ var init_process_kill = __esm({
104
121
  // packages/execution/dist/process-lifecycle.js
105
122
  import { randomUUID, createHash } from "node:crypto";
106
123
  import { appendFileSync, existsSync, mkdirSync, readFileSync, readlinkSync, renameSync, rmSync, statSync, writeFileSync } from "node:fs";
107
- import { homedir } from "node:os";
124
+ import { homedir as homedir2 } from "node:os";
108
125
  import { dirname, isAbsolute, join, relative, resolve } from "node:path";
109
126
  function nowFrom(system) {
110
127
  return system?.now?.() ?? Date.now();
@@ -113,7 +130,7 @@ function realSleep(ms) {
113
130
  return new Promise((resolveSleep) => setTimeout(resolveSleep, ms));
114
131
  }
115
132
  function registryDir(options) {
116
- return options?.storeDir || process.env["OMNIUS_PROCESS_REGISTRY_DIR"] || join(process.env["OMNIUS_HOME"] || join(homedir(), ".omnius"), "processes");
133
+ return options?.storeDir || process.env["OMNIUS_PROCESS_REGISTRY_DIR"] || join(process.env["OMNIUS_HOME"] || join(homedir2(), ".omnius"), "processes");
117
134
  }
118
135
  function ensureStoreDir(options) {
119
136
  const dir = registryDir(options);
@@ -499,13 +516,6 @@ var init_system_deps = __esm({
499
516
  }
500
517
  });
501
518
 
502
- // packages/execution/dist/process-async.js
503
- var init_process_async = __esm({
504
- "packages/execution/dist/process-async.js"() {
505
- "use strict";
506
- }
507
- });
508
-
509
519
  // packages/execution/dist/tools/cuda-device-filter.js
510
520
  var init_cuda_device_filter = __esm({
511
521
  "packages/execution/dist/tools/cuda-device-filter.js"() {
@@ -514,16 +524,6 @@ var init_cuda_device_filter = __esm({
514
524
  }
515
525
  });
516
526
 
517
- // packages/execution/dist/tools/model-store.js
518
- import { homedir as homedir2, platform } from "node:os";
519
- var platformId;
520
- var init_model_store = __esm({
521
- "packages/execution/dist/tools/model-store.js"() {
522
- "use strict";
523
- platformId = platform();
524
- }
525
- });
526
-
527
527
  // packages/execution/dist/tools/camera-capture.js
528
528
  var DEFAULT_CAMERA_PROFILE, CAMERA_PROFILES;
529
529
  var init_camera_capture = __esm({
@@ -245069,10 +245069,21 @@ import { dirname as dirname8 } from "node:path";
245069
245069
  init_model_broker();
245070
245070
  init_jetson_monitor();
245071
245071
 
245072
+ // packages/execution/dist/audio-classifier-runtime.js
245073
+ init_process_async();
245074
+
245072
245075
  // packages/execution/dist/python-cuda-runtime.js
245073
245076
  init_jetson_monitor();
245074
245077
  init_venv_paths();
245075
245078
 
245079
+ // packages/execution/dist/audio-classifier-runtime.js
245080
+ init_jetson_monitor();
245081
+ init_model_store();
245082
+ var MODEL_REVISION = "ac2ca3bd45d12ec1f19f1144205ea529b4e9dedf";
245083
+ var MODEL_URL = `https://huggingface.co/zeropointnine/yamnet-onnx/resolve/${MODEL_REVISION}/yamnet.onnx`;
245084
+ var CLASS_MAP_URL = `https://huggingface.co/zeropointnine/yamnet-onnx/resolve/${MODEL_REVISION}/yamnet_class_map.csv`;
245085
+ var SETUP_TIMEOUT_MS = 20 * 6e4;
245086
+
245076
245087
  // packages/execution/dist/broker-mediated-backend.js
245077
245088
  init_model_broker();
245078
245089
 
@@ -3832,7 +3832,7 @@
3832
3832
  "tags": [
3833
3833
  "ASR"
3834
3834
  ],
3835
- "description": "Body: {modelId?, device?}. Installs the managed runtime and pulls the requested model. Whisper, Nemotron, and Voxtral load the requested weights to validate exact CUDA placement; VibeVoice stores its pinned model and tokenizer snapshot under Omnius' ASR runtime directories.",
3835
+ "description": "Body: {modelId?, device?}. Installs the managed runtime and pulls the requested model. Whisper, Nemotron, and Voxtral load the requested weights to validate exact CUDA placement; VibeVoice stores its pinned microsoft/VibeVoice-ASR model and tokenizer snapshot under Omnius' ASR runtime directories.",
3836
3836
  "parameters": [
3837
3837
  {
3838
3838
  "name": "engineId",
@@ -3936,7 +3936,7 @@
3936
3936
  "tags": [
3937
3937
  "ASR"
3938
3938
  ],
3939
- "description": "Body: {engineId, modelId?, setup?, device?}. Selection is validated against the registry. setup=true pulls the exact managed model and verifies CUDA placement before activation. Hardware evidence uses nvidia-smi on discrete Linux and NVIDIA's documented tegrastats plus a CUDA Torch property probe on Jetson/L4T.",
3939
+ "description": "Body: {engineId, modelId?, setup?, device?}. Selection is validated against the registry and fails fail-closed: no unavailable engine, missing weights, or unverified CUDA placement can silently replace the active selection. setup=true pulls the exact managed model and verifies CUDA placement before activation. Hardware evidence uses nvidia-smi on discrete Linux and NVIDIA's documented tegrastats plus a CUDA Torch property probe on Jetson/L4T.",
3940
3940
  "responses": {
3941
3941
  "200": {
3942
3942
  "description": "Exact selection activated"
@@ -4184,6 +4184,236 @@
4184
4184
  "packages/cli/src/api/openapi.ts"
4185
4185
  ]
4186
4186
  },
4187
+ {
4188
+ "id": "api.v1-audio-classify",
4189
+ "kind": "api",
4190
+ "title": "/v1/audio/classify",
4191
+ "summary": "Classify a supplied WAV with persistent CUDA/TensorRT YAMNet",
4192
+ "aliases": [
4193
+ "/v1/audio/classify"
4194
+ ],
4195
+ "keywords": [
4196
+ "rest",
4197
+ "openapi",
4198
+ "POST",
4199
+ "Audio",
4200
+ "v1",
4201
+ "audio",
4202
+ "classify"
4203
+ ],
4204
+ "maturity": "stable",
4205
+ "layer": "interface",
4206
+ "audiences": [
4207
+ "integrator",
4208
+ "service-agent",
4209
+ "coding-agent"
4210
+ ],
4211
+ "interfaces": [
4212
+ {
4213
+ "type": "rest",
4214
+ "target": "POST /v1/audio/classify"
4215
+ },
4216
+ {
4217
+ "type": "openapi",
4218
+ "target": "/openapi.json"
4219
+ }
4220
+ ],
4221
+ "references": [
4222
+ {
4223
+ "type": "source",
4224
+ "target": "packages/cli/src/api/openapi.ts",
4225
+ "relation": "openapi-source"
4226
+ },
4227
+ {
4228
+ "type": "documentation",
4229
+ "target": "docs/reference/rest-api.md",
4230
+ "relation": "endpoint-inventory"
4231
+ }
4232
+ ],
4233
+ "methods": [
4234
+ "POST"
4235
+ ],
4236
+ "tags": [
4237
+ "Audio"
4238
+ ],
4239
+ "operations": {
4240
+ "post": {
4241
+ "summary": "Classify a supplied WAV with persistent CUDA/TensorRT YAMNet",
4242
+ "tags": [
4243
+ "Audio"
4244
+ ],
4245
+ "description": "Alias of POST /v1/tools/audio_analyze/call. Body: {args:{action:'classify',file:'/absolute/path.wav',top_k:5},timeout_ms:90000}. Input must be caller-supplied mono PCM16/16 kHz WAV; no microphone capture, dependency install, or model download occurs on this path. Returns result.data.classifications.",
4246
+ "responses": {
4247
+ "200": {
4248
+ "description": "Structured AudioSet-521 classification"
4249
+ },
4250
+ "409": {
4251
+ "description": "One request runs and one waits; additional work is busy"
4252
+ },
4253
+ "503": {
4254
+ "description": "Classifier not ready"
4255
+ }
4256
+ }
4257
+ }
4258
+ },
4259
+ "source_of_truth": [
4260
+ "GET /openapi.json",
4261
+ "packages/cli/src/api/openapi.ts"
4262
+ ]
4263
+ },
4264
+ {
4265
+ "id": "api.v1-audio-classify-health",
4266
+ "kind": "api",
4267
+ "title": "/v1/audio/classify/health",
4268
+ "summary": "Jetson CUDA/TensorRT YAMNet readiness",
4269
+ "aliases": [
4270
+ "/v1/audio/classify/health"
4271
+ ],
4272
+ "keywords": [
4273
+ "rest",
4274
+ "openapi",
4275
+ "GET",
4276
+ "Audio",
4277
+ "v1",
4278
+ "audio",
4279
+ "classify",
4280
+ "health"
4281
+ ],
4282
+ "maturity": "stable",
4283
+ "layer": "interface",
4284
+ "audiences": [
4285
+ "integrator",
4286
+ "service-agent",
4287
+ "coding-agent"
4288
+ ],
4289
+ "interfaces": [
4290
+ {
4291
+ "type": "rest",
4292
+ "target": "GET /v1/audio/classify/health"
4293
+ },
4294
+ {
4295
+ "type": "openapi",
4296
+ "target": "/openapi.json"
4297
+ }
4298
+ ],
4299
+ "references": [
4300
+ {
4301
+ "type": "source",
4302
+ "target": "packages/cli/src/api/openapi.ts",
4303
+ "relation": "openapi-source"
4304
+ },
4305
+ {
4306
+ "type": "documentation",
4307
+ "target": "docs/reference/rest-api.md",
4308
+ "relation": "endpoint-inventory"
4309
+ }
4310
+ ],
4311
+ "methods": [
4312
+ "GET"
4313
+ ],
4314
+ "tags": [
4315
+ "Audio"
4316
+ ],
4317
+ "operations": {
4318
+ "get": {
4319
+ "summary": "Jetson CUDA/TensorRT YAMNet readiness",
4320
+ "tags": [
4321
+ "Audio"
4322
+ ],
4323
+ "description": "Reports real pinned-model, TensorRT worker, CUDA placement, warmup, queue, and error state. General /health is not an audio readiness signal.",
4324
+ "responses": {
4325
+ "200": {
4326
+ "description": "Warm audio classifier readiness"
4327
+ },
4328
+ "503": {
4329
+ "description": "Audio classifier is installing, stopped, or unavailable"
4330
+ }
4331
+ }
4332
+ }
4333
+ },
4334
+ "source_of_truth": [
4335
+ "GET /openapi.json",
4336
+ "packages/cli/src/api/openapi.ts"
4337
+ ]
4338
+ },
4339
+ {
4340
+ "id": "api.v1-audio-classify-setup",
4341
+ "kind": "api",
4342
+ "title": "/v1/audio/classify/setup",
4343
+ "summary": "Provision and warm Jetson CUDA/TensorRT YAMNet",
4344
+ "aliases": [
4345
+ "/v1/audio/classify/setup"
4346
+ ],
4347
+ "keywords": [
4348
+ "rest",
4349
+ "openapi",
4350
+ "POST",
4351
+ "Audio",
4352
+ "v1",
4353
+ "audio",
4354
+ "classify",
4355
+ "setup"
4356
+ ],
4357
+ "maturity": "stable",
4358
+ "layer": "interface",
4359
+ "audiences": [
4360
+ "integrator",
4361
+ "service-agent",
4362
+ "coding-agent"
4363
+ ],
4364
+ "interfaces": [
4365
+ {
4366
+ "type": "rest",
4367
+ "target": "POST /v1/audio/classify/setup"
4368
+ },
4369
+ {
4370
+ "type": "openapi",
4371
+ "target": "/openapi.json"
4372
+ }
4373
+ ],
4374
+ "references": [
4375
+ {
4376
+ "type": "source",
4377
+ "target": "packages/cli/src/api/openapi.ts",
4378
+ "relation": "openapi-source"
4379
+ },
4380
+ {
4381
+ "type": "documentation",
4382
+ "target": "docs/reference/rest-api.md",
4383
+ "relation": "endpoint-inventory"
4384
+ }
4385
+ ],
4386
+ "methods": [
4387
+ "POST"
4388
+ ],
4389
+ "tags": [
4390
+ "Audio"
4391
+ ],
4392
+ "operations": {
4393
+ "post": {
4394
+ "summary": "Provision and warm Jetson CUDA/TensorRT YAMNet",
4395
+ "tags": [
4396
+ "Audio"
4397
+ ],
4398
+ "description": "Admin-only. Downloads checksummed ONNX/class-map artifacts once, builds a JetPack-specific FP16 TensorRT engine, verifies CUDA placement, and warms the persistent worker. It never installs a generic CUDA/Torch wheel.",
4399
+ "responses": {
4400
+ "200": {
4401
+ "description": "Audio classifier installed and warm"
4402
+ },
4403
+ "403": {
4404
+ "description": "Admin scope required"
4405
+ },
4406
+ "500": {
4407
+ "description": "JetPack/TensorRT/CUDA preflight or setup failed"
4408
+ }
4409
+ }
4410
+ }
4411
+ },
4412
+ "source_of_truth": [
4413
+ "GET /openapi.json",
4414
+ "packages/cli/src/api/openapi.ts"
4415
+ ]
4416
+ },
4187
4417
  {
4188
4418
  "id": "api.v1-audio-embed",
4189
4419
  "kind": "api",