omnius 1.0.643 → 1.0.646
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/py-embed.js +294275 -0
- package/dist/index.js +1779 -775
- package/dist/postinstall-daemon.cjs +4 -1
- package/dist/scripts/audio-clap-semantic-worker.py +12 -4
- package/dist/update-worker.js +4 -0
- package/docs/DISCOVERY.json +4 -4
- package/docs/reference/configuration.md +4 -1
- package/docs/rest/endpoints/voice-vision.md +20 -9
- package/npm-shrinkwrap.json +340 -150
- package/package.json +1 -1
|
@@ -657,7 +657,10 @@ function installSystemd(nodeBin, omniusScript, user) {
|
|
|
657
657
|
"Environment=OMNIUS_DAEMON=1",
|
|
658
658
|
"Environment=OMNIUS_PORT=" + PORT,
|
|
659
659
|
"Environment=NODE_ENV=production",
|
|
660
|
-
|
|
660
|
+
// Keep this user-relative so a unit written during a root/global install
|
|
661
|
+
// still loads the owning user's overrides after the existing unit is
|
|
662
|
+
// rewritten on update.
|
|
663
|
+
"EnvironmentFile=%h/.config/omnius/daemon.env",
|
|
661
664
|
"ExecStart=" + nodeBin + " " + omniusScript + " serve --daemon --quiet",
|
|
662
665
|
// Restart=always (was on-failure) — also relaunch on clean exit.
|
|
663
666
|
// Some upgrade flows trigger process.exit(0) (e.g. /update reload,
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
"""Persistent, offline CLAP semantic-audio embedding worker.
|
|
3
3
|
|
|
4
|
-
This process deliberately has no setup behavior.
|
|
5
|
-
created
|
|
6
|
-
verified the immutable model artifact
|
|
4
|
+
This process deliberately has no setup behavior. The Node runtime has already
|
|
5
|
+
created an isolated venv with an explicit vendor-Torch link, installed its
|
|
6
|
+
pinned non-Torch requirements, and verified the immutable model artifact
|
|
7
|
+
before this process is started.
|
|
7
8
|
"""
|
|
8
9
|
|
|
9
10
|
import argparse
|
|
@@ -22,7 +23,14 @@ if os.environ.get("PYTHONNOUSERSITE") != "1":
|
|
|
22
23
|
import numpy as np
|
|
23
24
|
import torch
|
|
24
25
|
import torch.nn.functional as F
|
|
25
|
-
from
|
|
26
|
+
from PIL import Image
|
|
27
|
+
from transformers import (
|
|
28
|
+
ClapAudioModelWithProjection,
|
|
29
|
+
ClapFeatureExtractor,
|
|
30
|
+
ClapModel,
|
|
31
|
+
ClapProcessor,
|
|
32
|
+
ClapTextModelWithProjection,
|
|
33
|
+
)
|
|
26
34
|
|
|
27
35
|
MODEL_ID = "laion/clap-htsat-unfused"
|
|
28
36
|
# The worker receives exactly Egg's caller-conditioned audio window. It does
|
package/dist/update-worker.js
CHANGED
|
@@ -245098,6 +245098,7 @@ init_venv_paths();
|
|
|
245098
245098
|
var SETUP_TIMEOUT_MS2 = 30 * 6e4;
|
|
245099
245099
|
var CLAP_RUNTIME_WHEELS = [
|
|
245100
245100
|
{ distribution: "numpy", version: "1.26.4", importName: "numpy", filename: "numpy-1.26.4-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", source: "https://files.pythonhosted.org/packages/fc/a5/4beee6488160798683eed5bdb7eead455892c3b4e1f78d79d8d3f3b084ac/numpy-1.26.4-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", sha256: "d209d8969599b27ad20994c8e41936ee0964e6da07478d6c35016bc386b66ad4" },
|
|
245101
|
+
{ distribution: "Pillow", version: "10.4.0", importName: "PIL", filename: "pillow-10.4.0-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", source: "https://files.pythonhosted.org/packages/8a/25/1fc45761955f9359b1169aa75e241551e74ac01a09f487adaaf4c3472d11/pillow-10.4.0-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", sha256: "7928ecbf1ece13956b95d9cbcfc77137652b02763ba384d9ab508099a2eca856" },
|
|
245101
245102
|
{ distribution: "transformers", version: "4.57.3", importName: "transformers", filename: "transformers-4.57.3-py3-none-any.whl", source: "https://files.pythonhosted.org/packages/6a/6b/2f416568b3c4c91c96e5a365d164f8a4a4a88030aa8ab4644181fdadce97/transformers-4.57.3-py3-none-any.whl", sha256: "c77d353a4851b1880191603d36acb313411d3577f6e2897814f333841f7003f4" },
|
|
245102
245103
|
{ distribution: "huggingface-hub", version: "0.36.0", importName: "huggingface_hub", filename: "huggingface_hub-0.36.0-py3-none-any.whl", source: "https://files.pythonhosted.org/packages/cb/bd/1a875e0d592d447cbc02805fd3fe0f497714d6a2583f59d14fa9ebad96eb/huggingface_hub-0.36.0-py3-none-any.whl", sha256: "7bcc9ad17d5b3f07b57c78e79d527102d08313caa278a641993acddcb894548d" },
|
|
245103
245104
|
{ distribution: "hf-xet", version: "1.1.5", importName: "hf_xet", filename: "hf_xet-1.1.5-cp37-abi3-manylinux_2_28_aarch64.whl", source: "https://files.pythonhosted.org/packages/d0/54/0fcf2b619720a26fbb6cc941e89f2472a522cd963a776c089b189559447f/hf_xet-1.1.5-cp37-abi3-manylinux_2_28_aarch64.whl", sha256: "dbba1660e5d810bd0ea77c511a99e9242d920790d0e63c0e4673ed36c4022d18" },
|
|
@@ -245118,6 +245119,9 @@ var CLAP_RUNTIME_WHEELS = [
|
|
|
245118
245119
|
];
|
|
245119
245120
|
var CLAP_PYTHON_REQUIREMENTS = CLAP_RUNTIME_WHEELS.map((wheel) => `${wheel.distribution}==${wheel.version}`);
|
|
245120
245121
|
|
|
245122
|
+
// packages/execution/dist/openclip-memory-admission.js
|
|
245123
|
+
init_jetson_monitor();
|
|
245124
|
+
|
|
245121
245125
|
// packages/execution/dist/speaker-embedding-runtime.js
|
|
245122
245126
|
init_jetson_monitor();
|
|
245123
245127
|
init_process_async();
|
package/docs/DISCOVERY.json
CHANGED
|
@@ -4631,7 +4631,7 @@
|
|
|
4631
4631
|
"tags": [
|
|
4632
4632
|
"Audio"
|
|
4633
4633
|
],
|
|
4634
|
-
"description": "Non-mutating readiness for NVIDIA diar_streaming_sortformer_4spk-v2. It never downloads, installs, creates environments, or loads a model. Managed JetPack setup pins NVIDIA's Q8 GGUF plus NeMo-Speech.cpp source revision and keeps one persistent worker/controller serialized; operator-provided .nemo/Python snapshots remain supported. 200 requires the verified runtime worker to be active. Session-local labels are never durable identities.",
|
|
4634
|
+
"description": "Non-mutating readiness for NVIDIA diar_streaming_sortformer_4spk-v2. It never downloads, installs, creates environments, or loads a model. Managed JetPack setup pins NVIDIA's Q8 GGUF plus NeMo-Speech.cpp source revision and keeps one persistent worker/controller serialized; operator-provided .nemo/Python snapshots remain supported. Readiness persists setup.stage, setup.failed_stage, complete setup stderr, and the terminal error across daemon restart. 200 requires the verified runtime worker to be active. Session-local labels are never durable identities.",
|
|
4635
4635
|
"responses": {
|
|
4636
4636
|
"200": {
|
|
4637
4637
|
"description": "Verified local snapshot and warm managed live worker."
|
|
@@ -4707,7 +4707,7 @@
|
|
|
4707
4707
|
"tags": [
|
|
4708
4708
|
"Audio"
|
|
4709
4709
|
],
|
|
4710
|
-
"description": "Admin-only and asynchronous. An empty JSON object provisions the pinned public Q8 Sortformer artifact and builds an immutable-revision CUDA NeMo-Speech.cpp runtime under ~/.omnius without modifying JetPack Torch. Poll readiness after HTTP 202. Advanced operators may instead provide snapshot_path/manifest_path plus python_path for a checksum-pinned .nemo runtime. Setup may download/build; inference never does.",
|
|
4710
|
+
"description": "Admin-only and asynchronous. An empty JSON object provisions the pinned public Q8 Sortformer artifact and builds an immutable-revision CUDA NeMo-Speech.cpp runtime under ~/.omnius without modifying JetPack Torch. nvcc discovery checks OMNIUS_DIAR_NVCC, /usr/local/cuda-12.2/bin/nvcc, then /usr/local/cuda/bin/nvcc and reports exact absence before downloading a model. Poll readiness after HTTP 202. Advanced operators may instead provide snapshot_path/manifest_path plus python_path for a checksum-pinned .nemo runtime. Setup may download/build; inference never does.",
|
|
4711
4711
|
"requestBody": {
|
|
4712
4712
|
"required": true,
|
|
4713
4713
|
"content": {
|
|
@@ -5401,7 +5401,7 @@
|
|
|
5401
5401
|
"tags": [
|
|
5402
5402
|
"Audio"
|
|
5403
5403
|
],
|
|
5404
|
-
"description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses pinned JetPack YAMNet/TensorRT. speaker creates a private CPU-only WeSpeaker CAM++ venv with checksum-pinned NumPy/ONNX Runtime wheels and no Torch/Torchaudio dependency. semantic installs a checksum-locked CPython 3.10/aarch64 dependency closure
|
|
5404
|
+
"description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses pinned JetPack YAMNet/TensorRT. speaker creates a private CPU-only WeSpeaker CAM++ venv with checksum-pinned NumPy/ONNX Runtime wheels and no Torch/Torchaudio dependency. semantic creates a private include-system-site-packages=false venv, explicitly links only the validated vendor CUDA Torch provider, and installs a checksum-locked CPython 3.10/aarch64 dependency closure (including Pillow and probed Transformers CLAP imports) plus the immutable CLAP revision. Omnius never replaces or resolves a generic Torch package over JetPack Torch. JetPack daemon bootstrap provisions all three roles by default; OMNIUS_AUDIO_AUTO_SETUP=0 disables all and OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP=0 disables CLAP. Semantic provisioning may finish under memory pressure, while activation/inference still requires the 8 GiB admission threshold and never evicts another workload. CLAP readiness persists setup stage, failed_stage, complete subprocess stderr, and error. This is the only REST operation allowed to provision, and no role is substituted for another.",
|
|
5405
5405
|
"parameters": [
|
|
5406
5406
|
{
|
|
5407
5407
|
"name": "kind",
|
|
@@ -16896,7 +16896,7 @@
|
|
|
16896
16896
|
"tags": [
|
|
16897
16897
|
"Vision"
|
|
16898
16898
|
],
|
|
16899
|
-
"description": "Admin-only single-flight setup, also daemon-bootstrapped by default on JetPack (OMNIUS_VISION_AUTO_SETUP=0 disables). It creates
|
|
16899
|
+
"description": "Admin-only single-flight setup, also daemon-bootstrapped by default on JetPack (OMNIUS_VISION_AUTO_SETUP=0 disables). It creates an include-system-site-packages=false venv, explicitly links only the validated vendor CUDA Torch provider, installs a fully pinned non-Torch wheel closure including wcwidth with --no-deps, and then validates Torch, vendor torchvision, and OpenCLIP with user-site packages disabled. It downloads the OpenCLIP checkpoint from immutable revision 1a25a446712ba5ee05982a381eed697ef9b435cf with a fixed SHA-256. It never resolves Torch/torchvision from generic PyPI. A checksum-pinned local torchvision wheel override remains supported. Provisioning completes despite transient memory pressure; only model loading/inference applies the OpenCLIP-specific OMNIUS_VISION_MIN_AVAILABLE_MB admission gate.",
|
|
16900
16900
|
"responses": {
|
|
16901
16901
|
"200": {
|
|
16902
16902
|
"description": "Provisioned; readiness may still report memory-blocked until model loading is admitted"
|
|
@@ -27,7 +27,9 @@ Configuration is layered from environment, global user settings, project setting
|
|
|
27
27
|
| `OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP` | Set to `0` to disable CLAP setup specifically; JetPack default is enabled |
|
|
28
28
|
| `OMNIUS_VISION_AUTO_SETUP` | Set to `0` to disable managed OpenCLIP provisioning on JetPack |
|
|
29
29
|
| `OMNIUS_VISION_PYTHON` | Explicit vendor CUDA Python for the isolated OpenCLIP runtime |
|
|
30
|
+
| `OMNIUS_VISION_MIN_AVAILABLE_MB` | OpenCLIP-specific Jetson unified-memory admission threshold (default `8192`) |
|
|
30
31
|
| `OMNIUS_NEMO_SPEECH_BIN` | Optional existing CUDA-enabled NeMo-Speech.cpp binary; otherwise live diarization builds the pinned source revision |
|
|
32
|
+
| `OMNIUS_DIAR_NVCC` | Optional absolute JetPack CUDA compiler; discovery otherwise checks `/usr/local/cuda-12.2/bin/nvcc` then `/usr/local/cuda/bin/nvcc` |
|
|
31
33
|
| `OMNIUS_DIARIZATION_AUTO_SETUP` | Set to `0` to disable managed diarizer provisioning; live Sortformer defaults on for JetPack |
|
|
32
34
|
| `OMNIUS_HF_TOKEN` | Setup-only Hugging Face token for gated Community-1 download; never persisted or passed to inference |
|
|
33
35
|
| `OMNIUS_PYANNOTE_TERMS_ACCEPTED` | Set to `1` only after accepting Community-1 terms; permits gated daemon bootstrap when a token is present |
|
|
@@ -52,7 +54,8 @@ upstream provider credentials.
|
|
|
52
54
|
|
|
53
55
|
Linux installations load optional daemon-only overrides from
|
|
54
56
|
`~/.config/omnius/daemon.env`. The postinstall creates this file with mode
|
|
55
|
-
`0600` and the systemd unit
|
|
57
|
+
`0600` and creates or rewrites the systemd unit with
|
|
58
|
+
`EnvironmentFile=%h/.config/omnius/daemon.env`.
|
|
56
59
|
This is the appropriate place for `OMNIUS_AUDIO_PYTHON` or a setup-only gated
|
|
57
60
|
model token; credentials do not need to be embedded in the unit or sent in a
|
|
58
61
|
REST request.
|
|
@@ -81,9 +81,10 @@ descriptive claims or invoke an LLM.
|
|
|
81
81
|
First poll `GET /v1/vision/embed/readiness`. It is non-mutating and returns
|
|
82
82
|
HTTP 503 until the isolated runtime and checksum-manifested weights are ready.
|
|
83
83
|
Use the admin-scoped `POST /v1/vision/embed/setup` to create the
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
84
|
+
private runtime under `~/.omnius/runtimes/vision/open-clip` with
|
|
85
|
+
`include-system-site-packages = false`, link only the already-validated vendor
|
|
86
|
+
Torch provider, install the pinned non-Torch closure (including `wcwidth`), and
|
|
87
|
+
fetch/verify the checkpoint from immutable revision
|
|
87
88
|
`1a25a446712ba5ee05982a381eed697ef9b435cf`. Setup is single-flight and
|
|
88
89
|
daemon-bootstrapped by default on JetPack; set `OMNIUS_VISION_AUTO_SETUP=0`
|
|
89
90
|
to disable it. Inference never installs packages or downloads artifacts, and
|
|
@@ -95,8 +96,10 @@ present in the compatible vendor stack or be supplied explicitly with the local
|
|
|
95
96
|
`OMNIUS_VISION_TORCHVISION_WHEEL_SHA256`; generic PyPI Torch/torchvision is
|
|
96
97
|
blocked. Bootstrap discovery may inspect the selected vendor interpreter, but
|
|
97
98
|
the managed runtime installs its own pinned support closure and proves Torch,
|
|
98
|
-
torchvision, and OpenCLIP with user-site packages disabled.
|
|
99
|
-
|
|
99
|
+
torchvision, and OpenCLIP with user-site packages disabled. OpenCLIP reports
|
|
100
|
+
its own `component: openclip` memory-admission result, governed by
|
|
101
|
+
`OMNIUS_VISION_MIN_AVAILABLE_MB`, rather than reusing CLAP telemetry. On
|
|
102
|
+
memory-constrained Jetson systems model loading and embedding fail with a typed 503 rather than
|
|
100
103
|
evicting a resident ASR/Ollama model. Dependency and
|
|
101
104
|
weight provisioning itself is allowed to finish under transient pressure;
|
|
102
105
|
readiness separately reports `installed`, `weightsReady`,
|
|
@@ -280,7 +283,10 @@ startup provisions acoustic, speaker, and semantic roles by default; set
|
|
|
280
283
|
`OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP=0` to disable CLAP specifically.
|
|
281
284
|
|
|
282
285
|
CLAP provisioning installs a checksum-locked CPython 3.10/aarch64 wheel
|
|
283
|
-
closure and the
|
|
286
|
+
closure—including Pillow and the complete probed Transformers CLAP import
|
|
287
|
+
surface—inside a private venv with `include-system-site-packages = false`.
|
|
288
|
+
Only the validated vendor CUDA Torch site is linked explicitly. The immutable
|
|
289
|
+
model revision is provisioned without loading the model. It is
|
|
284
290
|
allowed to finish while unified memory is busy. Worker activation and
|
|
285
291
|
inference retain the 8 GiB admission gate and idle eviction, so setup cannot
|
|
286
292
|
silently evict ASR, Ollama, or another CUDA workload.
|
|
@@ -344,8 +350,11 @@ exact remediation. Inference is likewise non-provisioning and returns typed
|
|
|
344
350
|
`speaker_diarization_runtime_unavailable` until admin setup succeeds.
|
|
345
351
|
|
|
346
352
|
Admin setup is asynchronous and single-flight. It returns HTTP 202 while it
|
|
347
|
-
provisions, and readiness exposes
|
|
348
|
-
|
|
353
|
+
provisions, and readiness exposes `setup.stage`, `setup.failed_stage`, the
|
|
354
|
+
complete captured setup `stderr`, and the terminal error. Those diagnostics
|
|
355
|
+
are persisted under the managed runtime so they survive daemon restart.
|
|
356
|
+
Existing ready workers return HTTP 200. Inference never performs these setup
|
|
357
|
+
actions.
|
|
349
358
|
|
|
350
359
|
On JetPack, live setup with an empty object downloads the immutable,
|
|
351
360
|
checksum-pinned Q8 Sortformer artifact and builds NVIDIA NeMo-Speech.cpp at a
|
|
@@ -360,7 +369,9 @@ POST /v1/audio/diarization/live/setup
|
|
|
360
369
|
The native build requires the JetPack CUDA compiler plus `git` and a C++17
|
|
361
370
|
compiler. Omnius installs pinned CMake/Ninja only in its private build-tools
|
|
362
371
|
venv and never invokes `sudo`; missing native prerequisites are reported by
|
|
363
|
-
readiness for operator installation outside inference.
|
|
372
|
+
readiness for operator installation outside inference. CUDA compiler discovery
|
|
373
|
+
checks `OMNIUS_DIAR_NVCC`, `/usr/local/cuda-12.2/bin/nvcc`, then
|
|
374
|
+
`/usr/local/cuda/bin/nvcc`, and reports those exact locations when none exists. Set
|
|
364
375
|
`OMNIUS_NEMO_SPEECH_BIN` to reuse a prebuilt CUDA-enabled binary.
|
|
365
376
|
|
|
366
377
|
Managed Community-1 setup creates a separate CPU-only CPython environment, so
|