omnius 1.0.628 → 1.0.629

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10657,7 +10657,7 @@
10657
10657
  "id": "api.v1-ocr-advanced",
10658
10658
  "kind": "api",
10659
10659
  "title": "/v1/ocr/advanced",
10660
- "summary": "Advanced OCR — multi-PSM tesseract pipeline with optional vision refinement — body {imagePath, visionModel?, prompt?}",
10660
+ "summary": "Alias of POST /v1/tools/ocr_image_advanced/call — managed multi-variant Tesseract OCR",
10661
10661
  "aliases": [
10662
10662
  "/v1/ocr/advanced"
10663
10663
  ],
@@ -10665,6 +10665,7 @@
10665
10665
  "rest",
10666
10666
  "openapi",
10667
10667
  "POST",
10668
+ "Tools",
10668
10669
  "Vision",
10669
10670
  "v1",
10670
10671
  "ocr",
@@ -10703,23 +10704,94 @@
10703
10704
  "POST"
10704
10705
  ],
10705
10706
  "tags": [
10707
+ "Tools",
10706
10708
  "Vision"
10707
10709
  ],
10708
10710
  "operations": {
10709
10711
  "post": {
10710
- "summary": "Advanced OCR — multi-PSM tesseract pipeline with optional vision refinement — body {imagePath, visionModel?, prompt?}",
10712
+ "summary": "Alias of POST /v1/tools/ocr_image_advanced/call — managed multi-variant Tesseract OCR",
10711
10713
  "tags": [
10714
+ "Tools",
10712
10715
  "Vision"
10713
10716
  ],
10717
+ "description": "Uses the identical advanced OCR tool exposed to agents: auto-provisions Tesseract, requested traineddata, and the managed Python pipeline. Body is the canonical direct-tool envelope; result.data contains the parsed OCR pipeline result.",
10718
+ "requestBody": {
10719
+ "required": true,
10720
+ "content": {
10721
+ "application/json": {
10722
+ "schema": {
10723
+ "type": "object",
10724
+ "required": [
10725
+ "args"
10726
+ ],
10727
+ "properties": {
10728
+ "args": {
10729
+ "type": "object",
10730
+ "required": [
10731
+ "image"
10732
+ ],
10733
+ "properties": {
10734
+ "image": {
10735
+ "type": "string"
10736
+ },
10737
+ "language": {
10738
+ "type": "string",
10739
+ "example": "eng+fra"
10740
+ },
10741
+ "regions": {
10742
+ "type": "boolean"
10743
+ },
10744
+ "region": {
10745
+ "type": "string",
10746
+ "example": "0,0,1200,180"
10747
+ },
10748
+ "psm": {
10749
+ "type": "integer",
10750
+ "enum": [
10751
+ 4,
10752
+ 6,
10753
+ 11
10754
+ ]
10755
+ },
10756
+ "output_dir": {
10757
+ "type": "string"
10758
+ },
10759
+ "batch": {
10760
+ "type": "boolean"
10761
+ },
10762
+ "debug": {
10763
+ "type": "boolean"
10764
+ }
10765
+ }
10766
+ },
10767
+ "timeout_ms": {
10768
+ "type": "integer",
10769
+ "description": "Defaults to 300000ms; maximum 600000ms."
10770
+ },
10771
+ "max_output_chars": {
10772
+ "type": "integer"
10773
+ },
10774
+ "profile": {
10775
+ "type": "string"
10776
+ },
10777
+ "working_dir": {
10778
+ "type": "string",
10779
+ "description": "Admin scope only."
10780
+ }
10781
+ }
10782
+ }
10783
+ }
10784
+ }
10785
+ },
10714
10786
  "responses": {
10715
10787
  "200": {
10716
- "description": "OCR text + context block"
10788
+ "description": "Direct-tool result envelope; consume result.data for structured OCR output."
10717
10789
  },
10718
10790
  "400": {
10719
- "description": "Missing imagePath"
10791
+ "description": "Invalid direct-tool request or OCR arguments"
10720
10792
  },
10721
- "501": {
10722
- "description": "OCR tool unavailable"
10793
+ "403": {
10794
+ "description": "Tool scope/profile/working-directory denied"
10723
10795
  }
10724
10796
  }
10725
10797
  }
@@ -28854,7 +28926,7 @@
28854
28926
  "id": "guide.rest-endpoints-voice-vision",
28855
28927
  "kind": "guide",
28856
28928
  "title": "Voice, Audio, Vision, And Voicechat",
28857
- "summary": "POST /v1/voice/tts returns audio bytes. format can be wav or pcm. X-Sample-Rate reports the sample rate.",
28929
+ "summary": "POST /v1/ocr/advanced is a compatibility alias for the exact agent-facing ocrimageadvanced tool. Use the same direct-tool envelope; the response retains display text in result.output and returns the parsed pipeline payload in result.data.",
28858
28930
  "keywords": [
28859
28931
  "rest",
28860
28932
  "endpoints",
package/docs/DISCOVERY.md CHANGED
@@ -170,7 +170,7 @@ Daemon equivalents are `GET /v1/discovery/bootstrap`, `GET /v1/discovery?q=<inte
170
170
  | `api.v1-memory-write` | /v1/memory/write | Write a memory entry (PT-04) |
171
171
  | `api.v1-models` | /v1/models | List aggregated models |
172
172
  | `api.v1-nexus-status` | /v1/nexus/status | Nexus peer state |
173
- | `api.v1-ocr-advanced` | /v1/ocr/advanced | Advanced OCR — multi-PSM tesseract pipeline with optional vision refinement — body {imagePath, visionModel?, prompt?} |
173
+ | `api.v1-ocr-advanced` | /v1/ocr/advanced | Alias of POST /v1/tools/ocr_image_advanced/call — managed multi-variant Tesseract OCR |
174
174
  | `api.v1-ollama-pool-cleanup` | /v1/ollama/pool/cleanup | Clean up stale Ollama pool processes |
175
175
  | `api.v1-ollama-pool-processes` | /v1/ollama/pool/processes | Scan local Ollama pool processes |
176
176
  | `api.v1-profiles` | /v1/profiles | List tool profiles; Create tool profile |
@@ -461,7 +461,7 @@ Daemon equivalents are `GET /v1/discovery/bootstrap`, `GET /v1/discovery?q=<inte
461
461
  | `guide.rest-endpoints-run` | Runs, Jobs, Todos, And Scheduled Work | Submits a long-running task to the daemon and returns an accepted job record. Runs are tracked under .omnius/jobs/. |
462
462
  | `guide.rest-endpoints-skills` | Skills And Commands | The REST docs bundle is available as: |
463
463
  | `guide.rest-endpoints-tools` | Tools, MCP, Hooks, Agents, And Code Graph | GET /v1/tools and POST /v1/tools/{name}/call use the same direct-call registry, built from the @omnius/execution tool manifest plus virtual lifecycle tools such as taskcomplete. If a tool appears in /v1/tools with directcallable: true, the same name is callable at /v1/tools/{name}/call. |
464
- | `guide.rest-endpoints-voice-vision` | Voice, Audio, Vision, And Voicechat | POST /v1/voice/tts returns audio bytes. format can be wav or pcm. X-Sample-Rate reports the sample rate. |
464
+ | `guide.rest-endpoints-voice-vision` | Voice, Audio, Vision, And Voicechat | POST /v1/ocr/advanced is a compatibility alias for the exact agent-facing ocrimageadvanced tool. Use the same direct-tool envelope; the response retains display text in result.output and returns the parsed pipeline payload in result.data. |
465
465
  | `guide.rest-errors-pagination-etags` | Errors, Pagination, ETags, And Request IDs | The OpenAPI contract documents RFC 7807 Problem Details for common errors: |
466
466
  | `guide.rest-examples-curl` | Curl Examples | curl -s "$BASE/openapi.json" \| jq '.info.title, .info.version' |
467
467
  | `guide.rest-examples-openai-sdk` | OpenAI SDK Examples | Omnius exposes an OpenAI-compatible chat-completions endpoint at /v1/chat/completions. |
@@ -309,7 +309,7 @@ the store, `OMNIUS_DISABLE_CONTEXT_WINDOW_DUMPS=1` to disable it, and
309
309
  | `POST` | `/v1/vision/describe` | Vision describe placeholder |
310
310
  | `POST` | `/v1/vision/embed` | Create a vision embedding from media |
311
311
  | `POST` | `/v1/audio/embed` | Create an audio embedding from audio input |
312
- | `POST` | `/v1/ocr/advanced` | Advanced OCR — multi-PSM tesseract pipeline with optional vision refinement |
312
+ | `POST` | `/v1/ocr/advanced` | Agent-equivalent managed advanced OCR (alias of `/v1/tools/ocr_image_advanced/call`) |
313
313
 
314
314
  `POST /v1/voice/tts` and `/v1/audio/speech` automatically warm the daemon.
315
315
  An explicit model must render exactly or the request fails; Omnius does not
@@ -113,7 +113,7 @@ Scopes:
113
113
  | Nexus | `/v1/nexus/status`, `/v1/sponsors` |
114
114
  | Ollama pool | `/v1/ollama/pool/processes`, `/v1/ollama/pool/cleanup` |
115
115
  | Voice and audio | `/v1/voice/state`, `/v1/voice/models`, `/v1/voice/tts`, `/v1/audio/speech`, `/v1/asr/engines`, `/v1/asr/selection`, `/v1/asr/activate`, `/v1/asr/transcriptions`, `/v1/asr/test`, `/v1/audio/transcriptions`, `/v1/voicechat/ws`, `/v1/media/av/analyze` |
116
- | Vision/audio embeddings | `/v1/vision/describe`, `/v1/vision/embed`, `/v1/audio/embed`, `/v1/ocr/advanced` |
116
+ | Vision/audio/OCR | `/v1/vision/describe`, `/v1/vision/embed`, `/v1/audio/embed`, `/v1/ocr/advanced` (agent-equivalent advanced OCR) |
117
117
  | Projects | `/v1/projects`, `/v1/projects/current`, `/v1/projects/switch`, `/v1/projects/register`, `/v1/projects/rename`, `/v1/projects/scan`, `/v1/projects/preferences` |
118
118
  | Code graph | `/v1/codegraph/snapshot`, `/v1/codegraph/events` |
119
119
  | Scheduled jobs | `/v1/scheduled`, `/v1/scheduled/all`, `/v1/scheduled/status`, `/v1/scheduled/kill`, `/v1/scheduled/fixup`, `/v1/scheduled/reconcile` |
@@ -32,9 +32,28 @@
32
32
  | `POST` | `/v1/voice/speak` | Synthesize and broadcast to voicechat clients |
33
33
  | `WS` | `/v1/voicechat/ws` | Full-duplex voicechat WebSocket |
34
34
  | `POST` | `/v1/vision/describe` | Vision describe placeholder/deferred endpoint |
35
- | `POST` | `/v1/ocr/advanced` | Advanced OCR — multi-PSM tesseract pipeline with optional vision refinement |
35
+ | `POST` | `/v1/ocr/advanced` | Agent-equivalent managed advanced OCR (alias of `/v1/tools/ocr_image_advanced/call`) |
36
36
  | `POST` | `/v1/media/av/analyze` | Grounded AV/audio comprehension of a media file into entities/events |
37
37
 
38
+ ## Advanced OCR
39
+
40
+ `POST /v1/ocr/advanced` is a compatibility alias for the exact
41
+ agent-facing `ocr_image_advanced` tool. Use the same direct-tool envelope;
42
+ the response retains display text in `result.output` and returns the parsed
43
+ pipeline payload in `result.data`.
44
+
45
+ ```bash
46
+ curl -sS -X POST http://127.0.0.1:11435/v1/ocr/advanced \
47
+ -H 'content-type: application/json' \
48
+ -d '{"args":{"image":"/data/invoice.png","language":"eng","psm":6},"timeout_ms":300000}'
49
+ ```
50
+
51
+ On first use, Omnius verifies and provisions Tesseract, requested traineddata,
52
+ and the advanced Python OCR pipeline. Jetson uses Ubuntu/JetPack system Python
53
+ packages in an isolated system-site venv; it never installs generic CUDA,
54
+ Torch, or OpenCV replacements. The caller needs `run` scope because optional
55
+ `output_dir`, batch, and debug modes write OCR artifacts.
56
+
38
57
  ## TTS
39
58
 
40
59
  `POST /v1/voice/tts` returns audio bytes. `format` can be `wav` or `pcm`. `X-Sample-Rate` reports the sample rate.
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "omnius",
3
- "version": "1.0.628",
3
+ "version": "1.0.629",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "omnius",
9
- "version": "1.0.628",
9
+ "version": "1.0.629",
10
10
  "bundleDependencies": [
11
11
  "image-to-ascii"
12
12
  ],
@@ -5519,9 +5519,9 @@
5519
5519
  }
5520
5520
  },
5521
5521
  "node_modules/node-addon-api": {
5522
- "version": "8.9.1",
5523
- "resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.9.1.tgz",
5524
- "integrity": "sha512-4eUQWVPCUUUiBjLnHS3cXWeC6ryoPUc0U3rP7IuzapoGbzMqd/r6KKO0clr0b+snQhsrueFEhCZDdK+LK7hxKg==",
5522
+ "version": "8.9.2",
5523
+ "resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.9.2.tgz",
5524
+ "integrity": "sha512-VijLXbi3UACN69I0JVXJsX4tjACjNoQDgv2gTF6sx2wWEi8tkSg2eX8p5gSIFi8z2+DL3oHmY6OyKce38SDolg==",
5525
5525
  "license": "MIT",
5526
5526
  "optional": true,
5527
5527
  "engines": {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "omnius",
3
- "version": "1.0.628",
3
+ "version": "1.0.629",
4
4
  "description": "AI coding agent powered by open-source models (Ollama/vLLM) — interactive TUI with agentic tool-calling loop",
5
5
  "type": "module",
6
6
  "main": "./dist/library.js",