omnius 1.0.638 → 1.0.640
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +1314 -857
- package/dist/scripts/audio-speaker-embedding-worker.py +115 -31
- package/dist/scripts/live-whisper.py +41 -0
- package/dist/scripts/ocr-advanced.py +273 -140
- package/dist/scripts/transcribe-file.py +61 -6
- package/dist/update-worker.js +0 -1
- package/docs/DISCOVERY.json +26 -6
- package/docs/rest/endpoints/chat.md +5 -1
- package/docs/rest/endpoints/voice-vision.md +26 -23
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
package/docs/DISCOVERY.json
CHANGED
|
@@ -4171,7 +4171,7 @@
|
|
|
4171
4171
|
],
|
|
4172
4172
|
"responses": {
|
|
4173
4173
|
"200": {
|
|
4174
|
-
"description": "Transcription with actual engine/model identity
|
|
4174
|
+
"description": "Transcription with actual engine/model identity; exact digital PCM silence returns text='' plus typed noSpeech evidence without model inference"
|
|
4175
4175
|
},
|
|
4176
4176
|
"400": {
|
|
4177
4177
|
"description": "Empty or tiny audio"
|
|
@@ -5399,7 +5399,7 @@
|
|
|
5399
5399
|
"tags": [
|
|
5400
5400
|
"Audio"
|
|
5401
5401
|
],
|
|
5402
|
-
"description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses the pinned JetPack YAMNet/TensorRT setup; speaker
|
|
5402
|
+
"description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses the pinned JetPack YAMNet/TensorRT setup; speaker creates a private CPython 3.10 CPU-only WeSpeaker CAM++ venv, installs only checksum-pinned NumPy, ONNX Runtime, and its locked CPU support wheels with --no-deps/--no-index, then downloads the pinned model. Its 80-bin Kaldi-compatible NumPy fbank is validated with a deterministic signature before readiness. The speaker path never imports, links, replaces, or otherwise depends on JetPack Torch/Torchaudio, so an incompatible generic Torchaudio wheel cannot affect CUDA-enabled Egg Torch. semantic installs the isolated JetPack CUDA CLAP dependencies and pinned model. This is the only REST operation allowed to provision. It activates and warms only the requested role worker; no role is substituted for another.",
|
|
5403
5403
|
"parameters": [
|
|
5404
5404
|
{
|
|
5405
5405
|
"name": "kind",
|
|
@@ -6070,7 +6070,7 @@
|
|
|
6070
6070
|
"$ref": "#/components/parameters/MinimumOmniusVersion"
|
|
6071
6071
|
}
|
|
6072
6072
|
],
|
|
6073
|
-
"description": "Standard OpenAI chat-completion proxy by default. Set agent_loop=true to enter a synchronous server-side loop that auto-executes daemon tools
|
|
6073
|
+
"description": "Standard OpenAI chat-completion proxy by default. Set agent_loop=true to enter a bounded synchronous server-side loop that auto-executes selected daemon tools and re-prompts the model. Ollama targets use native /api/chat tool messages. timeout_s bounds each backend call; agent_timeout_s bounds the whole loop (default 45s, max 600s). daemon_tool_names explicitly limits the offered catalog; factual-first implicitly limits it to web_search/web_fetch when no names are supplied. Identical repeated calls fail with HTTP 508 instead of looping. SSE streaming outside agent_loop honors OpenAI tool_calls deltas.",
|
|
6074
6074
|
"requestBody": {
|
|
6075
6075
|
"required": true,
|
|
6076
6076
|
"content": {
|
|
@@ -6140,7 +6140,21 @@
|
|
|
6140
6140
|
"admin"
|
|
6141
6141
|
]
|
|
6142
6142
|
},
|
|
6143
|
-
"description": "
|
|
6143
|
+
"description": "Scopes from which Omnius may offer its bounded core daemon-tool catalog. Use daemon_tool_names to request other exact tools."
|
|
6144
|
+
},
|
|
6145
|
+
"daemon_tool_names": {
|
|
6146
|
+
"type": "array",
|
|
6147
|
+
"items": {
|
|
6148
|
+
"type": "string"
|
|
6149
|
+
},
|
|
6150
|
+
"description": "Optional exact allowlist of daemon tool names. Use this to keep local-model tool prompts small."
|
|
6151
|
+
},
|
|
6152
|
+
"agent_timeout_s": {
|
|
6153
|
+
"type": "number",
|
|
6154
|
+
"minimum": 5,
|
|
6155
|
+
"maximum": 600,
|
|
6156
|
+
"default": 45,
|
|
6157
|
+
"description": "Total server-side agent-loop deadline. Distinct from per-backend timeout_s."
|
|
6144
6158
|
},
|
|
6145
6159
|
"max_turns": {
|
|
6146
6160
|
"type": "integer",
|
|
@@ -6160,7 +6174,7 @@
|
|
|
6160
6174
|
},
|
|
6161
6175
|
"responses": {
|
|
6162
6176
|
"200": {
|
|
6163
|
-
"description": "OpenAI chat.completion shape, SSE if stream=true. agent_loop responses include _agent_loop:{turns,log,done,reason}."
|
|
6177
|
+
"description": "OpenAI chat.completion shape, SSE if stream=true. agent_loop responses include _agent_loop:{turns,log,done,reason,elapsed_ms,backend_transport}."
|
|
6164
6178
|
},
|
|
6165
6179
|
"400": {
|
|
6166
6180
|
"description": "Invalid request",
|
|
@@ -6231,6 +6245,12 @@
|
|
|
6231
6245
|
}
|
|
6232
6246
|
}
|
|
6233
6247
|
}
|
|
6248
|
+
},
|
|
6249
|
+
"504": {
|
|
6250
|
+
"description": "Backend round or total agent-loop deadline expired"
|
|
6251
|
+
},
|
|
6252
|
+
"508": {
|
|
6253
|
+
"description": "The model repeated an identical daemon tool call"
|
|
6234
6254
|
}
|
|
6235
6255
|
}
|
|
6236
6256
|
}
|
|
@@ -18540,7 +18560,7 @@
|
|
|
18540
18560
|
"tags": [
|
|
18541
18561
|
"Voice"
|
|
18542
18562
|
],
|
|
18543
|
-
"description": "Returns synthesized audio and automatically enables/warms the selected runtime. Format `wav` (default) ships a complete WAV with header; `pcm` ships raw Int16 LE PCM. Set model=voxtral-4b-tts-2603 and voice=<preset> for the managed CUDA Voxtral renderer; the OpenAI-compatible alias also accepts model=mistralai/Voxtral-4B-TTS-2603. Other model transactions remain serialized and restore the persistent renderer. Headers report X-Voice-Model, X-Voice-Backend, and X-Sample-Rate.",
|
|
18563
|
+
"description": "Returns synthesized audio and automatically enables/warms the selected runtime. An explicit model has no silent fallback. Format `wav` (default) ships a complete WAV with header; `pcm` ships raw Int16 LE PCM. Set model=voxtral-4b-tts-2603 and voice=<preset> for the managed CUDA Voxtral renderer; the OpenAI-compatible alias also accepts model=mistralai/Voxtral-4B-TTS-2603. Other model transactions remain serialized and restore the persistent renderer. Headers report X-Voice-Model, X-Voice-Backend, and X-Sample-Rate.",
|
|
18544
18564
|
"responses": {
|
|
18545
18565
|
"200": {
|
|
18546
18566
|
"description": "Audio bytes (audio/wav or audio/L16). Headers: X-Voice-Model, X-Sample-Rate."
|
|
@@ -43,7 +43,9 @@ Important body fields:
|
|
|
43
43
|
| `parallel_tool_calls` | boolean | Forwarded to backend |
|
|
44
44
|
| `timeout_s` | number | Per-request timeout |
|
|
45
45
|
| `agent_loop` | boolean | Run server-side tool loop |
|
|
46
|
-
| `include_daemon_tools` | array |
|
|
46
|
+
| `include_daemon_tools` | array | Permit the bounded core daemon-tool catalog by scope: `read`, `run`, `admin` |
|
|
47
|
+
| `daemon_tool_names` | array | Exact daemon-tool allowlist; recommended for local models |
|
|
48
|
+
| `agent_timeout_s` | number | Whole-loop deadline, default 45 seconds and maximum 600 |
|
|
47
49
|
| `max_turns` | integer | Server-side agent loop turn cap |
|
|
48
50
|
| `prompt_template` | string | Optional template such as `factual-first` |
|
|
49
51
|
|
|
@@ -126,3 +128,5 @@ For ASR/TTS systems that only need the text brain, use `/realtime` or `/v1/realt
|
|
|
126
128
|
## Server-Side Agent Loop
|
|
127
129
|
|
|
128
130
|
`/v1/chat/completions` can run an internal tool loop when `agent_loop: true`. This lets clients collapse multiple model/tool round trips into one daemon request. Daemon tool calls execute inline; client-owned tool calls can still be yielded in OpenAI-compatible shape.
|
|
131
|
+
|
|
132
|
+
Ollama-backed loops use its native `/api/chat` tool protocol. `timeout_s` applies to each backend round, while `agent_timeout_s` bounds the complete loop and defaults to 45 seconds. Omnius returns a typed HTTP 504 when that budget expires and HTTP 508 when a model repeats the same daemon tool with identical arguments. Tool results are bounded before the next prompt. Without `daemon_tool_names`, Omnius offers only a compact core catalog permitted by `include_daemon_tools`, avoiding a huge local-model prompt; request any other tools by exact name. `prompt_template: "factual-first"` narrows the implicit catalog further to `web_search` and `web_fetch`.
|
|
@@ -115,7 +115,7 @@ pipeline payload in `result.data`.
|
|
|
115
115
|
```bash
|
|
116
116
|
curl -sS -X POST http://127.0.0.1:11435/v1/ocr/advanced \
|
|
117
117
|
-H 'content-type: application/json' \
|
|
118
|
-
-d '{"args":{"image":"/data/invoice.png","language":"eng","psm":6},"timeout_ms":
|
|
118
|
+
-d '{"args":{"image":"/data/invoice.png","language":"eng","psm":6},"timeout_ms":90000}'
|
|
119
119
|
```
|
|
120
120
|
|
|
121
121
|
Inference never invokes sudo, apt, pip, or venv creation. If readiness is
|
|
@@ -126,6 +126,15 @@ version results without requiring particular dpkg package names. The caller
|
|
|
126
126
|
needs `run` scope because optional
|
|
127
127
|
`output_dir`, batch, and debug modes write OCR artifacts.
|
|
128
128
|
|
|
129
|
+
The managed pipeline is bounded: it starts with high-yield preprocessing and
|
|
130
|
+
expands variants only for low-evidence or larger images. Small crops avoid the
|
|
131
|
+
former all-variant/all-PSM explosion. The REST default is 90 seconds (maximum
|
|
132
|
+
180 seconds), while the worker has an 80-second internal deadline so it can
|
|
133
|
+
return diagnostics. A timeout or cancellation returns
|
|
134
|
+
`result.data.schema=omnius.ocr-diagnostic.v1` with code `ocr_timeout` or
|
|
135
|
+
`ocr_cancelled`; cancellation terminates the Python/Tesseract process group
|
|
136
|
+
with TERM followed by KILL.
|
|
137
|
+
|
|
129
138
|
## TTS
|
|
130
139
|
|
|
131
140
|
`POST /v1/voice/tts` returns audio bytes. `format` can be `wav` or `pcm`. `X-Sample-Rate` reports the sample rate.
|
|
@@ -251,28 +260,22 @@ startup follows `OMNIUS_AUDIO_AUTO_SETUP` for acoustic and speaker; semantic
|
|
|
251
260
|
startup additionally requires `OMNIUS_SEMANTIC_AUDIO_AUTO_SETUP=1`.
|
|
252
261
|
|
|
253
262
|
On JetPack, speaker setup deliberately leaves `OMNIUS_AUDIO_PYTHON` (including
|
|
254
|
-
an Egg project `.venv`) unchanged. It creates
|
|
255
|
-
`~/.omnius/runtimes/audio/speaker/venv
|
|
256
|
-
|
|
257
|
-
and
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
can import that module only from its Python user site: speaker inference keeps
|
|
271
|
-
`PYTHONNOUSERSITE=1`, so it never borrows mutable user-site packages at runtime.
|
|
272
|
-
The bootstrap step inspects only CPython/site-package location before this
|
|
273
|
-
repair; the managed-venv Torch import is authoritative. If another dependency
|
|
274
|
-
is missing, readiness and setup preserve the exact missing module name instead
|
|
275
|
-
of incorrectly reporting that the selected JetPack Torch is invalid.
|
|
263
|
+
an Egg project `.venv`) unchanged. It creates a private, non-system-site
|
|
264
|
+
`~/.omnius/runtimes/audio/speaker/venv`, then installs only a locked,
|
|
265
|
+
checksum-verified CPython 3.10/aarch64 CPU dependency set: NumPy 1.26.4,
|
|
266
|
+
ONNX Runtime 1.17.3, and the exact CPU support wheels required by that ONNX
|
|
267
|
+
Runtime release. Each wheel is materialized and SHA-256 verified before the
|
|
268
|
+
managed venv installs it with `--no-deps --no-index`.
|
|
269
|
+
|
|
270
|
+
The worker implements the WeSpeaker CAM++ 80-bin Kaldi configuration in pure
|
|
271
|
+
NumPy: 25 ms / 10 ms Hamming frames, dither disabled, Kaldi pre-emphasis and
|
|
272
|
+
mel bank behavior, then full-clip CMN without CVN. Readiness invokes a
|
|
273
|
+
deterministic CPU preprocessing probe and requires its pinned rounded-feature
|
|
274
|
+
SHA-256 signature before it can report ready. The speaker path neither imports
|
|
275
|
+
nor links Torch or Torchaudio, so the generic `torchaudio-2.2.0` ABI mismatch
|
|
276
|
+
cannot bind against, replace, or otherwise affect JetPack's CUDA-enabled Egg
|
|
277
|
+
Torch. Inference remains install-free and network-free; failed package imports
|
|
278
|
+
or calibration leave the role unready with the exact probe error.
|
|
276
279
|
|
|
277
280
|
The dedicated route returns typed HTTP 503 whenever its requested runtime is
|
|
278
281
|
unprovisioned or unavailable: `audio_classifier_not_ready` (acoustic),
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "omnius",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.640",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "omnius",
|
|
9
|
-
"version": "1.0.
|
|
9
|
+
"version": "1.0.640",
|
|
10
10
|
"bundleDependencies": [
|
|
11
11
|
"image-to-ascii"
|
|
12
12
|
],
|
package/package.json
CHANGED