omnius 1.0.628 → 1.0.631
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/dist/index.js +5088 -4568
- package/dist/postinstall-daemon.cjs +71 -6
- package/dist/scripts/audio-yamnet-tensorrt-worker.py +18 -1
- package/dist/scripts/ocr-advanced.py +22 -12
- package/dist/update-worker.js +142 -135
- package/docs/DISCOVERY.json +1034 -1284
- package/docs/DISCOVERY.md +130 -128
- package/docs/reference/rest-api.md +3 -1
- package/docs/rest/INDEX.md +1 -1
- package/docs/rest/endpoints/voice-vision.md +43 -4
- package/npm-shrinkwrap.json +5 -5
- package/package.json +2 -2
|
@@ -38,7 +38,7 @@ var SERVICE_LABEL = "omnius-daemon";
|
|
|
38
38
|
var LAUNCHD_LABEL = "ai.omnius.daemon";
|
|
39
39
|
var WIN_TASK_NAME = "OmniusDaemon";
|
|
40
40
|
var REQUIRED_AIWG_VERSION = "2026.5.11";
|
|
41
|
-
var INSTALL_STEPS =
|
|
41
|
+
var INSTALL_STEPS = 8;
|
|
42
42
|
var DISCOVERY_NOTICE_PRINTED = false;
|
|
43
43
|
|
|
44
44
|
// ─── Helpers ────────────────────────────────────────────────────────────────
|
|
@@ -99,6 +99,68 @@ function hasCmd(name) {
|
|
|
99
99
|
return runQuiet("which " + name);
|
|
100
100
|
}
|
|
101
101
|
|
|
102
|
+
// Advanced OCR on JetPack uses Ubuntu's ABI-matched OpenCV/NumPy/Pillow stack.
|
|
103
|
+
// Native packages belong to package installation, never to an inference or
|
|
104
|
+
// readiness request. Jammy has no python3-pytesseract candidate, so that one
|
|
105
|
+
// pure-Python package is intentionally handled later by /v1/ocr/setup.
|
|
106
|
+
function ensureJetsonOcrNativeDependencies() {
|
|
107
|
+
if (!IS_LINUX) return;
|
|
108
|
+
var jetson = fs.existsSync("/etc/nv_tegra_release") || fs.existsSync("/sys/module/tegra_fuse");
|
|
109
|
+
if (!jetson) {
|
|
110
|
+
try {
|
|
111
|
+
var model = fs.readFileSync("/proc/device-tree/model", "utf8").replace(/\0/g, " ");
|
|
112
|
+
jetson = /\b(nvidia|jetson|orin|tegra)\b/i.test(model);
|
|
113
|
+
} catch (e) {}
|
|
114
|
+
}
|
|
115
|
+
if (!jetson || !hasCmd("apt-get") || !hasCmd("dpkg-query")) return;
|
|
116
|
+
|
|
117
|
+
var packages = [
|
|
118
|
+
"tesseract-ocr",
|
|
119
|
+
"python3-venv",
|
|
120
|
+
"python3-opencv",
|
|
121
|
+
"python3-numpy",
|
|
122
|
+
"python3-pil",
|
|
123
|
+
"python3-reportlab",
|
|
124
|
+
];
|
|
125
|
+
var missing = packages.filter(function (pkg) {
|
|
126
|
+
return !runQuiet("dpkg-query -W -f='${Status}' " + shellQuote(pkg) + " 2>/dev/null | grep -q 'install ok installed'");
|
|
127
|
+
});
|
|
128
|
+
if (missing.length === 0) {
|
|
129
|
+
log("Jetson advanced-OCR native dependencies are installed.");
|
|
130
|
+
return;
|
|
131
|
+
}
|
|
132
|
+
var install = "apt-get install -y " + missing.map(shellQuote).join(" ");
|
|
133
|
+
if (process.getuid && process.getuid() === 0) {
|
|
134
|
+
log("Provisioning Jetson advanced-OCR native dependencies: " + missing.join(", "));
|
|
135
|
+
if (!runQuiet(install, { timeout: 300000, env: Object.assign({}, process.env, { DEBIAN_FRONTEND: "noninteractive" }) })) {
|
|
136
|
+
warn("Jetson OCR native dependency install failed. Run: sudo " + install);
|
|
137
|
+
}
|
|
138
|
+
return;
|
|
139
|
+
}
|
|
140
|
+
if (hasCmd("sudo") && runQuiet("sudo -n true")) {
|
|
141
|
+
log("Provisioning Jetson advanced-OCR native dependencies with the install-time sudo authorization: " + missing.join(", "));
|
|
142
|
+
if (!runQuiet("sudo -n " + install, { timeout: 300000, env: Object.assign({}, process.env, { DEBIAN_FRONTEND: "noninteractive" }) })) {
|
|
143
|
+
warn("Jetson OCR native dependency install failed. Run: sudo " + install);
|
|
144
|
+
}
|
|
145
|
+
return;
|
|
146
|
+
}
|
|
147
|
+
if (hasCmd("sudo") && process.stdin.isTTY && process.stdout.isTTY) {
|
|
148
|
+
log("Jetson advanced OCR needs native Ubuntu packages; requesting elevation during package installation.");
|
|
149
|
+
try {
|
|
150
|
+
cp.execSync("sudo " + install, {
|
|
151
|
+
stdio: "inherit",
|
|
152
|
+
timeout: 300000,
|
|
153
|
+
env: Object.assign({}, process.env, { DEBIAN_FRONTEND: "noninteractive" }),
|
|
154
|
+
});
|
|
155
|
+
return;
|
|
156
|
+
} catch (e) {
|
|
157
|
+
warn("Jetson OCR native dependency install was not authorized. Run: sudo " + install);
|
|
158
|
+
return;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
warn("Jetson OCR native dependencies are missing; install outside inference: sudo " + install);
|
|
162
|
+
}
|
|
163
|
+
|
|
102
164
|
function shellQuote(value) {
|
|
103
165
|
if (IS_WIN) return '"' + String(value).replace(/"/g, '\\"') + '"';
|
|
104
166
|
return "'" + String(value).replace(/'/g, "'\\''") + "'";
|
|
@@ -924,10 +986,13 @@ function main() {
|
|
|
924
986
|
warn("wrapper auto-repair crashed (non-fatal): " + (e && e.message));
|
|
925
987
|
}
|
|
926
988
|
|
|
927
|
-
progress(3, INSTALL_STEPS, "
|
|
989
|
+
progress(3, INSTALL_STEPS, "Provisioning Jetson OCR native dependencies");
|
|
990
|
+
ensureJetsonOcrNativeDependencies();
|
|
991
|
+
|
|
992
|
+
progress(4, INSTALL_STEPS, "Ensuring AIWG CLI");
|
|
928
993
|
ensureGlobalAiwgInstall();
|
|
929
994
|
|
|
930
|
-
progress(
|
|
995
|
+
progress(5, INSTALL_STEPS, "Ensuring local REST key");
|
|
931
996
|
ensureBootstrapApiKeyForUser(effectiveUser());
|
|
932
997
|
|
|
933
998
|
if (process.env.OMNIUS_UPDATE_COORDINATED === "1") {
|
|
@@ -946,7 +1011,7 @@ function main() {
|
|
|
946
1011
|
// restart can't reach it, so we have to clean up explicitly. Without this,
|
|
947
1012
|
// the new service-managed daemon fails to bind port 11435 and the user
|
|
948
1013
|
// ends up running stale in-memory code from the previous version.
|
|
949
|
-
progress(
|
|
1014
|
+
progress(6, INSTALL_STEPS, "Clearing daemon port");
|
|
950
1015
|
stopAttestedDaemonPortHolder(PORT, function (killedCount) {
|
|
951
1016
|
if (killedCount > 0) {
|
|
952
1017
|
log("Killed " + killedCount + " stale daemon process(es) holding port " + PORT + ".");
|
|
@@ -977,7 +1042,7 @@ function runMainAfterKill() {
|
|
|
977
1042
|
// idempotent and ensures the unit file matches the current bundle.
|
|
978
1043
|
}
|
|
979
1044
|
|
|
980
|
-
progress(
|
|
1045
|
+
progress(7, INSTALL_STEPS, "Installing daemon service");
|
|
981
1046
|
log("Installing Omnius API daemon service for port " + PORT + " ...");
|
|
982
1047
|
log(" node: " + nodeBin);
|
|
983
1048
|
log(" omnius script: " + omniusScript);
|
|
@@ -1008,7 +1073,7 @@ function runMainAfterKill() {
|
|
|
1008
1073
|
}
|
|
1009
1074
|
|
|
1010
1075
|
// Wait up to 15s for /health to come up, but don't fail npm install.
|
|
1011
|
-
progress(
|
|
1076
|
+
progress(8, INSTALL_STEPS, "Verifying daemon health");
|
|
1012
1077
|
waitForHealth(15000, function (healthy) {
|
|
1013
1078
|
if (healthy) {
|
|
1014
1079
|
log("Omnius API daemon is live: http://127.0.0.1:" + PORT + "/health");
|
|
@@ -150,6 +150,10 @@ def read_wav(path: str, np):
|
|
|
150
150
|
if sample_rate != 16000:
|
|
151
151
|
raise RuntimeError(f"Expected 16 kHz WAV from the caller, received {sample_rate} Hz")
|
|
152
152
|
waveform = np.frombuffer(raw, dtype="<i2").astype(np.float32) / 32768.0
|
|
153
|
+
if waveform.size > 96000:
|
|
154
|
+
raise RuntimeError(
|
|
155
|
+
f"Expected at most 6 seconds of caller-conditioned audio, received {waveform.size / sample_rate:.3f} seconds"
|
|
156
|
+
)
|
|
153
157
|
return waveform, sample_rate, (time.perf_counter() - started) * 1000
|
|
154
158
|
|
|
155
159
|
|
|
@@ -310,7 +314,18 @@ def main() -> int:
|
|
|
310
314
|
if not request_id or not file_path:
|
|
311
315
|
raise RuntimeError("classify requires id and file")
|
|
312
316
|
waveform, sample_rate, decode_ms = read_wav(file_path, worker.np)
|
|
313
|
-
|
|
317
|
+
rms = float(worker.np.sqrt(worker.np.mean(worker.np.square(waveform)))) if waveform.size else 0.0
|
|
318
|
+
peak = float(worker.np.max(worker.np.abs(waveform))) if waveform.size else 0.0
|
|
319
|
+
# Do not force YAMNet to invent an AudioSet label for digital or
|
|
320
|
+
# near-digital silence. This remains a valid, grounded result.
|
|
321
|
+
low_information = rms < 0.0005 and peak < 0.002
|
|
322
|
+
if low_information:
|
|
323
|
+
classifications, inference_ms = [], 0.0
|
|
324
|
+
else:
|
|
325
|
+
classifications, inference_ms = worker.infer(
|
|
326
|
+
waveform,
|
|
327
|
+
max(1, min(int(request.get("top_k", 5)), 25)),
|
|
328
|
+
)
|
|
314
329
|
emit({
|
|
315
330
|
"type": "result",
|
|
316
331
|
"id": request_id,
|
|
@@ -318,6 +333,8 @@ def main() -> int:
|
|
|
318
333
|
"sample_rate_hz": sample_rate,
|
|
319
334
|
"duration_seconds": round(float(waveform.size) / sample_rate, 6),
|
|
320
335
|
"classifications": classifications,
|
|
336
|
+
"low_information": low_information,
|
|
337
|
+
"acoustic": {"rms": round(rms, 8), "peak": round(peak, 8)},
|
|
321
338
|
"timings_ms": {"decode": round(decode_ms, 3), "inference": round(inference_ms, 3)},
|
|
322
339
|
})
|
|
323
340
|
except Exception as exc: # keep the persistent worker alive per request
|
|
@@ -45,7 +45,7 @@ def check_deps():
|
|
|
45
45
|
try:
|
|
46
46
|
import cv2
|
|
47
47
|
except ImportError:
|
|
48
|
-
missing.append("
|
|
48
|
+
missing.append("cv2")
|
|
49
49
|
try:
|
|
50
50
|
import numpy
|
|
51
51
|
except ImportError:
|
|
@@ -57,12 +57,12 @@ def check_deps():
|
|
|
57
57
|
try:
|
|
58
58
|
from PIL import Image
|
|
59
59
|
except ImportError:
|
|
60
|
-
missing.append("
|
|
60
|
+
missing.append("PIL")
|
|
61
61
|
|
|
62
62
|
if missing:
|
|
63
63
|
print(json.dumps({
|
|
64
|
-
"error": f"
|
|
65
|
-
|
|
64
|
+
"error": f"OCR runtime imports became unavailable: {', '.join(missing)}. "
|
|
65
|
+
"Run POST /v1/ocr/setup and poll GET /v1/ocr/readiness; inference never installs packages.",
|
|
66
66
|
"missing": missing,
|
|
67
67
|
}))
|
|
68
68
|
sys.exit(1)
|
|
@@ -179,10 +179,9 @@ def run_tesseract(binary_img, language="eng", psm=6):
|
|
|
179
179
|
pil_img = Image.fromarray(binary_img)
|
|
180
180
|
config = f"--psm {psm}"
|
|
181
181
|
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
return "", 0.0, 0
|
|
182
|
+
# Engine failures are not blank documents. Let callers distinguish a real
|
|
183
|
+
# low-information image from a missing language pack or broken Tesseract.
|
|
184
|
+
text = pytesseract.image_to_string(pil_img, lang=language, config=config).strip()
|
|
186
185
|
|
|
187
186
|
line_count = len([l for l in text.split("\n") if l.strip()])
|
|
188
187
|
|
|
@@ -322,6 +321,7 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
|
|
|
322
321
|
|
|
323
322
|
# Generate all variants and run OCR
|
|
324
323
|
all_results = {}
|
|
324
|
+
ocr_errors = []
|
|
325
325
|
best_key = None
|
|
326
326
|
best_score = -1
|
|
327
327
|
|
|
@@ -338,7 +338,11 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
|
|
|
338
338
|
|
|
339
339
|
for psm in psm_modes:
|
|
340
340
|
key = f"{vname}_psm{psm}"
|
|
341
|
-
|
|
341
|
+
try:
|
|
342
|
+
text, confidence, line_count = run_tesseract(binary, language, psm)
|
|
343
|
+
except Exception as error:
|
|
344
|
+
ocr_errors.append(f"{key}: {error}")
|
|
345
|
+
continue
|
|
342
346
|
char_count = len(text)
|
|
343
347
|
score = compute_score(text, confidence, line_count)
|
|
344
348
|
|
|
@@ -355,7 +359,8 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
|
|
|
355
359
|
best_key = key
|
|
356
360
|
|
|
357
361
|
if not best_key:
|
|
358
|
-
|
|
362
|
+
detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
|
|
363
|
+
return {"error": f"Tesseract failed for every OCR variant: {detail}"}
|
|
359
364
|
|
|
360
365
|
best = all_results[best_key]
|
|
361
366
|
result = {
|
|
@@ -400,7 +405,10 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
|
|
|
400
405
|
if debug_dir:
|
|
401
406
|
cv2.imwrite(os.path.join(debug_dir, f"region_{rname}_{vname}.png"), binary)
|
|
402
407
|
|
|
403
|
-
|
|
408
|
+
try:
|
|
409
|
+
text, conf, lc = run_tesseract(binary, language, 6)
|
|
410
|
+
except Exception:
|
|
411
|
+
continue
|
|
404
412
|
score = compute_score(text, conf, lc)
|
|
405
413
|
if score > region_best_score:
|
|
406
414
|
region_best_score = score
|
|
@@ -531,7 +539,7 @@ def main():
|
|
|
531
539
|
print(f"Processed {result['images_processed']} images → {result['output_dir']}")
|
|
532
540
|
else:
|
|
533
541
|
print(json.dumps(result, indent=2))
|
|
534
|
-
sys.exit(0)
|
|
542
|
+
sys.exit(1 if "error" in result else 0)
|
|
535
543
|
|
|
536
544
|
# Single image mode
|
|
537
545
|
if not os.path.isfile(args.image):
|
|
@@ -565,6 +573,8 @@ def main():
|
|
|
565
573
|
print(result["text"])
|
|
566
574
|
else:
|
|
567
575
|
print(json.dumps(result, indent=2))
|
|
576
|
+
if "error" in result:
|
|
577
|
+
sys.exit(1)
|
|
568
578
|
|
|
569
579
|
|
|
570
580
|
if __name__ == "__main__":
|