omnius 1.0.643 → 1.0.646

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -657,7 +657,10 @@ function installSystemd(nodeBin, omniusScript, user) {
657
657
  "Environment=OMNIUS_DAEMON=1",
658
658
  "Environment=OMNIUS_PORT=" + PORT,
659
659
  "Environment=NODE_ENV=production",
660
- "EnvironmentFile=-" + envPath,
660
+ // Keep this user-relative so a unit written during a root/global install
661
+ // still loads the owning user's overrides after the existing unit is
662
+ // rewritten on update.
663
+ "EnvironmentFile=%h/.config/omnius/daemon.env",
661
664
  "ExecStart=" + nodeBin + " " + omniusScript + " serve --daemon --quiet",
662
665
  // Restart=always (was on-failure) — also relaunch on clean exit.
663
666
  // Some upgrade flows trigger process.exit(0) (e.g. /update reload,
@@ -1,9 +1,10 @@
1
1
  #!/usr/bin/env python3
2
2
  """Persistent, offline CLAP semantic-audio embedding worker.
3
3
 
4
- This process deliberately has no setup behavior. The Node runtime has already
5
- created the system-site venv, installed its pinned non-Torch requirements, and
6
- verified the immutable model artifact before this process is started.
4
+ This process deliberately has no setup behavior. The Node runtime has already
5
+ created an isolated venv with an explicit vendor-Torch link, installed its
6
+ pinned non-Torch requirements, and verified the immutable model artifact
7
+ before this process is started.
7
8
  """
8
9
 
9
10
  import argparse
@@ -22,7 +23,14 @@ if os.environ.get("PYTHONNOUSERSITE") != "1":
22
23
  import numpy as np
23
24
  import torch
24
25
  import torch.nn.functional as F
25
- from transformers import ClapModel, ClapProcessor
26
+ from PIL import Image
27
+ from transformers import (
28
+ ClapAudioModelWithProjection,
29
+ ClapFeatureExtractor,
30
+ ClapModel,
31
+ ClapProcessor,
32
+ ClapTextModelWithProjection,
33
+ )
26
34
 
27
35
  MODEL_ID = "laion/clap-htsat-unfused"
28
36
  # The worker receives exactly Egg's caller-conditioned audio window. It does
@@ -245098,6 +245098,7 @@ init_venv_paths();
245098
245098
  var SETUP_TIMEOUT_MS2 = 30 * 6e4;
245099
245099
  var CLAP_RUNTIME_WHEELS = [
245100
245100
  { distribution: "numpy", version: "1.26.4", importName: "numpy", filename: "numpy-1.26.4-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", source: "https://files.pythonhosted.org/packages/fc/a5/4beee6488160798683eed5bdb7eead455892c3b4e1f78d79d8d3f3b084ac/numpy-1.26.4-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", sha256: "d209d8969599b27ad20994c8e41936ee0964e6da07478d6c35016bc386b66ad4" },
245101
+ { distribution: "Pillow", version: "10.4.0", importName: "PIL", filename: "pillow-10.4.0-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", source: "https://files.pythonhosted.org/packages/8a/25/1fc45761955f9359b1169aa75e241551e74ac01a09f487adaaf4c3472d11/pillow-10.4.0-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", sha256: "7928ecbf1ece13956b95d9cbcfc77137652b02763ba384d9ab508099a2eca856" },
245101
245102
  { distribution: "transformers", version: "4.57.3", importName: "transformers", filename: "transformers-4.57.3-py3-none-any.whl", source: "https://files.pythonhosted.org/packages/6a/6b/2f416568b3c4c91c96e5a365d164f8a4a4a88030aa8ab4644181fdadce97/transformers-4.57.3-py3-none-any.whl", sha256: "c77d353a4851b1880191603d36acb313411d3577f6e2897814f333841f7003f4" },
245102
245103
  { distribution: "huggingface-hub", version: "0.36.0", importName: "huggingface_hub", filename: "huggingface_hub-0.36.0-py3-none-any.whl", source: "https://files.pythonhosted.org/packages/cb/bd/1a875e0d592d447cbc02805fd3fe0f497714d6a2583f59d14fa9ebad96eb/huggingface_hub-0.36.0-py3-none-any.whl", sha256: "7bcc9ad17d5b3f07b57c78e79d527102d08313caa278a641993acddcb894548d" },
245103
245104
  { distribution: "hf-xet", version: "1.1.5", importName: "hf_xet", filename: "hf_xet-1.1.5-cp37-abi3-manylinux_2_28_aarch64.whl", source: "https://files.pythonhosted.org/packages/d0/54/0fcf2b619720a26fbb6cc941e89f2472a522cd963a776c089b189559447f/hf_xet-1.1.5-cp37-abi3-manylinux_2_28_aarch64.whl", sha256: "dbba1660e5d810bd0ea77c511a99e9242d920790d0e63c0e4673ed36c4022d18" },
@@ -245118,6 +245119,9 @@ var CLAP_RUNTIME_WHEELS = [
245118
245119
  ];
245119
245120
  var CLAP_PYTHON_REQUIREMENTS = CLAP_RUNTIME_WHEELS.map((wheel) => `${wheel.distribution}==${wheel.version}`);
245120
245121
 
245122
+ // packages/execution/dist/openclip-memory-admission.js
245123
+ init_jetson_monitor();
245124
+
245121
245125
  // packages/execution/dist/speaker-embedding-runtime.js
245122
245126
  init_jetson_monitor();
245123
245127
  init_process_async();
@@ -4631,7 +4631,7 @@
4631
4631
  "tags": [
4632
4632
  "Audio"
4633
4633
  ],
4634
- "description": "Non-mutating readiness for NVIDIA diar_streaming_sortformer_4spk-v2. It never downloads, installs, creates environments, or loads a model. Managed JetPack setup pins NVIDIA's Q8 GGUF plus NeMo-Speech.cpp source revision and keeps one persistent worker/controller serialized; operator-provided .nemo/Python snapshots remain supported. 200 requires the verified runtime worker to be active. Session-local labels are never durable identities.",
4634
+ "description": "Non-mutating readiness for NVIDIA diar_streaming_sortformer_4spk-v2. It never downloads, installs, creates environments, or loads a model. Managed JetPack setup pins NVIDIA's Q8 GGUF plus NeMo-Speech.cpp source revision and keeps one persistent worker/controller serialized; operator-provided .nemo/Python snapshots remain supported. Readiness persists setup.stage, setup.failed_stage, complete setup stderr, and the terminal error across daemon restart. 200 requires the verified runtime worker to be active. Session-local labels are never durable identities.",
4635
4635
  "responses": {
4636
4636
  "200": {
4637
4637
  "description": "Verified local snapshot and warm managed live worker."
@@ -4707,7 +4707,7 @@
4707
4707
  "tags": [
4708
4708
  "Audio"
4709
4709
  ],
4710
- "description": "Admin-only and asynchronous. An empty JSON object provisions the pinned public Q8 Sortformer artifact and builds an immutable-revision CUDA NeMo-Speech.cpp runtime under ~/.omnius without modifying JetPack Torch. Poll readiness after HTTP 202. Advanced operators may instead provide snapshot_path/manifest_path plus python_path for a checksum-pinned .nemo runtime. Setup may download/build; inference never does.",
4710
+ "description": "Admin-only and asynchronous. An empty JSON object provisions the pinned public Q8 Sortformer artifact and builds an immutable-revision CUDA NeMo-Speech.cpp runtime under ~/.omnius without modifying JetPack Torch. nvcc discovery checks OMNIUS_DIAR_NVCC, /usr/local/cuda-12.2/bin/nvcc, then /usr/local/cuda/bin/nvcc and reports exact absence before downloading a model. Poll readiness after HTTP 202. Advanced operators may instead provide snapshot_path/manifest_path plus python_path for a checksum-pinned .nemo runtime. Setup may download/build; inference never does.",
4711
4711
  "requestBody": {
4712
4712
  "required": true,
4713
4713
  "content": {
@@ -5401,7 +5401,7 @@
5401
5401
  "tags": [
5402
5402
  "Audio"
5403
5403
  ],
5404
- "description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses pinned JetPack YAMNet/TensorRT. speaker creates a private CPU-only WeSpeaker CAM++ venv with checksum-pinned NumPy/ONNX Runtime wheels and no Torch/Torchaudio dependency. semantic installs a checksum-locked CPython 3.10/aarch64 dependency closure plus the immutable CLAP revision while leaving vendor Torch untouched. Omnius never imports, links, replaces, or resolves a generic Torch package over JetPack Torch for either isolated runtime. JetPack daemon bootstrap provisions all three roles by default; OMNIUS_AUDIO_AUTO_SETUP=0 disables all and OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP=0 disables CLAP. Semantic provisioning may finish under memory pressure, while activation/inference still requires the 8 GiB admission threshold and never evicts another workload. This is the only REST operation allowed to provision, and no role is substituted for another.",
5404
+ "description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses pinned JetPack YAMNet/TensorRT. speaker creates a private CPU-only WeSpeaker CAM++ venv with checksum-pinned NumPy/ONNX Runtime wheels and no Torch/Torchaudio dependency. semantic creates a private include-system-site-packages=false venv, explicitly links only the validated vendor CUDA Torch provider, and installs a checksum-locked CPython 3.10/aarch64 dependency closure (including Pillow and probed Transformers CLAP imports) plus the immutable CLAP revision. Omnius never replaces or resolves a generic Torch package over JetPack Torch. JetPack daemon bootstrap provisions all three roles by default; OMNIUS_AUDIO_AUTO_SETUP=0 disables all and OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP=0 disables CLAP. Semantic provisioning may finish under memory pressure, while activation/inference still requires the 8 GiB admission threshold and never evicts another workload. CLAP readiness persists setup stage, failed_stage, complete subprocess stderr, and error. This is the only REST operation allowed to provision, and no role is substituted for another.",
5405
5405
  "parameters": [
5406
5406
  {
5407
5407
  "name": "kind",
@@ -16896,7 +16896,7 @@
16896
16896
  "tags": [
16897
16897
  "Vision"
16898
16898
  ],
16899
- "description": "Admin-only single-flight setup, also daemon-bootstrapped by default on JetPack (OMNIUS_VISION_AUTO_SETUP=0 disables). It creates a --system-site-packages venv, inherits existing CUDA Torch and vendor torchvision, installs a fully pinned non-Torch wheel closure with --no-deps, and then validates the managed runtime with user-site packages disabled. It downloads the OpenCLIP checkpoint from immutable revision 1a25a446712ba5ee05982a381eed697ef9b435cf with a fixed SHA-256. It never resolves Torch/torchvision from generic PyPI. A checksum-pinned local torchvision wheel override remains supported. Provisioning completes despite transient memory pressure; only model loading/inference applies the 8 GiB admission gate.",
16899
+ "description": "Admin-only single-flight setup, also daemon-bootstrapped by default on JetPack (OMNIUS_VISION_AUTO_SETUP=0 disables). It creates an include-system-site-packages=false venv, explicitly links only the validated vendor CUDA Torch provider, installs a fully pinned non-Torch wheel closure including wcwidth with --no-deps, and then validates Torch, vendor torchvision, and OpenCLIP with user-site packages disabled. It downloads the OpenCLIP checkpoint from immutable revision 1a25a446712ba5ee05982a381eed697ef9b435cf with a fixed SHA-256. It never resolves Torch/torchvision from generic PyPI. A checksum-pinned local torchvision wheel override remains supported. Provisioning completes despite transient memory pressure; only model loading/inference applies the OpenCLIP-specific OMNIUS_VISION_MIN_AVAILABLE_MB admission gate.",
16900
16900
  "responses": {
16901
16901
  "200": {
16902
16902
  "description": "Provisioned; readiness may still report memory-blocked until model loading is admitted"
@@ -27,7 +27,9 @@ Configuration is layered from environment, global user settings, project setting
27
27
  | `OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP` | Set to `0` to disable CLAP setup specifically; JetPack default is enabled |
28
28
  | `OMNIUS_VISION_AUTO_SETUP` | Set to `0` to disable managed OpenCLIP provisioning on JetPack |
29
29
  | `OMNIUS_VISION_PYTHON` | Explicit vendor CUDA Python for the isolated OpenCLIP runtime |
30
+ | `OMNIUS_VISION_MIN_AVAILABLE_MB` | OpenCLIP-specific Jetson unified-memory admission threshold (default `8192`) |
30
31
  | `OMNIUS_NEMO_SPEECH_BIN` | Optional existing CUDA-enabled NeMo-Speech.cpp binary; otherwise live diarization builds the pinned source revision |
32
+ | `OMNIUS_DIAR_NVCC` | Optional absolute JetPack CUDA compiler; discovery otherwise checks `/usr/local/cuda-12.2/bin/nvcc` then `/usr/local/cuda/bin/nvcc` |
31
33
  | `OMNIUS_DIARIZATION_AUTO_SETUP` | Set to `0` to disable managed diarizer provisioning; live Sortformer defaults on for JetPack |
32
34
  | `OMNIUS_HF_TOKEN` | Setup-only Hugging Face token for gated Community-1 download; never persisted or passed to inference |
33
35
  | `OMNIUS_PYANNOTE_TERMS_ACCEPTED` | Set to `1` only after accepting Community-1 terms; permits gated daemon bootstrap when a token is present |
@@ -52,7 +54,8 @@ upstream provider credentials.
52
54
 
53
55
  Linux installations load optional daemon-only overrides from
54
56
  `~/.config/omnius/daemon.env`. The postinstall creates this file with mode
55
- `0600` and the systemd unit references it with an optional `EnvironmentFile`.
57
+ `0600` and creates or rewrites the systemd unit with
58
+ `EnvironmentFile=%h/.config/omnius/daemon.env`.
56
59
  This is the appropriate place for `OMNIUS_AUDIO_PYTHON` or a setup-only gated
57
60
  model token; credentials do not need to be embedded in the unit or sent in a
58
61
  REST request.
@@ -81,9 +81,10 @@ descriptive claims or invoke an LLM.
81
81
  First poll `GET /v1/vision/embed/readiness`. It is non-mutating and returns
82
82
  HTTP 503 until the isolated runtime and checksum-manifested weights are ready.
83
83
  Use the admin-scoped `POST /v1/vision/embed/setup` to create the
84
- `--system-site-packages` runtime under
85
- `~/.omnius/runtimes/vision/open-clip`, install pinned non-Torch dependencies,
86
- and fetch/verify the checkpoint from immutable revision
84
+ private runtime under `~/.omnius/runtimes/vision/open-clip` with
85
+ `include-system-site-packages = false`, link only the already-validated vendor
86
+ Torch provider, install the pinned non-Torch closure (including `wcwidth`), and
87
+ fetch/verify the checkpoint from immutable revision
87
88
  `1a25a446712ba5ee05982a381eed697ef9b435cf`. Setup is single-flight and
88
89
  daemon-bootstrapped by default on JetPack; set `OMNIUS_VISION_AUTO_SETUP=0`
89
90
  to disable it. Inference never installs packages or downloads artifacts, and
@@ -95,8 +96,10 @@ present in the compatible vendor stack or be supplied explicitly with the local
95
96
  `OMNIUS_VISION_TORCHVISION_WHEEL_SHA256`; generic PyPI Torch/torchvision is
96
97
  blocked. Bootstrap discovery may inspect the selected vendor interpreter, but
97
98
  the managed runtime installs its own pinned support closure and proves Torch,
98
- torchvision, and OpenCLIP with user-site packages disabled. On memory-constrained
99
- Jetson systems model loading and embedding fail with a typed 503 rather than
99
+ torchvision, and OpenCLIP with user-site packages disabled. OpenCLIP reports
100
+ its own `component: openclip` memory-admission result, governed by
101
+ `OMNIUS_VISION_MIN_AVAILABLE_MB`, rather than reusing CLAP telemetry. On
102
+ memory-constrained Jetson systems model loading and embedding fail with a typed 503 rather than
100
103
  evicting a resident ASR/Ollama model. Dependency and
101
104
  weight provisioning itself is allowed to finish under transient pressure;
102
105
  readiness separately reports `installed`, `weightsReady`,
@@ -280,7 +283,10 @@ startup provisions acoustic, speaker, and semantic roles by default; set
280
283
  `OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP=0` to disable CLAP specifically.
281
284
 
282
285
  CLAP provisioning installs a checksum-locked CPython 3.10/aarch64 wheel
283
- closure and the immutable model revision without loading the model. It is
286
+ closure—including Pillow and the complete probed Transformers CLAP import
287
+ surface—inside a private venv with `include-system-site-packages = false`.
288
+ Only the validated vendor CUDA Torch site is linked explicitly. The immutable
289
+ model revision is provisioned without loading the model. It is
284
290
  allowed to finish while unified memory is busy. Worker activation and
285
291
  inference retain the 8 GiB admission gate and idle eviction, so setup cannot
286
292
  silently evict ASR, Ollama, or another CUDA workload.
@@ -344,8 +350,11 @@ exact remediation. Inference is likewise non-provisioning and returns typed
344
350
  `speaker_diarization_runtime_unavailable` until admin setup succeeds.
345
351
 
346
352
  Admin setup is asynchronous and single-flight. It returns HTTP 202 while it
347
- provisions, and readiness exposes the exact phase or terminal error. Existing
348
- ready workers return HTTP 200. Inference never performs these setup actions.
353
+ provisions, and readiness exposes `setup.stage`, `setup.failed_stage`, the
354
+ complete captured setup `stderr`, and the terminal error. Those diagnostics
355
+ are persisted under the managed runtime so they survive daemon restart.
356
+ Existing ready workers return HTTP 200. Inference never performs these setup
357
+ actions.
349
358
 
350
359
  On JetPack, live setup with an empty object downloads the immutable,
351
360
  checksum-pinned Q8 Sortformer artifact and builds NVIDIA NeMo-Speech.cpp at a
@@ -360,7 +369,9 @@ POST /v1/audio/diarization/live/setup
360
369
  The native build requires the JetPack CUDA compiler plus `git` and a C++17
361
370
  compiler. Omnius installs pinned CMake/Ninja only in its private build-tools
362
371
  venv and never invokes `sudo`; missing native prerequisites are reported by
363
- readiness for operator installation outside inference. Set
372
+ readiness for operator installation outside inference. CUDA compiler discovery
373
+ checks `OMNIUS_DIAR_NVCC`, `/usr/local/cuda-12.2/bin/nvcc`, then
374
+ `/usr/local/cuda/bin/nvcc`, and reports those exact locations when none exists. Set
364
375
  `OMNIUS_NEMO_SPEECH_BIN` to reuse a prebuilt CUDA-enabled binary.
365
376
 
366
377
  Managed Community-1 setup creates a separate CPU-only CPython environment, so