omnius 1.0.628 → 1.0.631

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -38,7 +38,7 @@ var SERVICE_LABEL = "omnius-daemon";
38
38
  var LAUNCHD_LABEL = "ai.omnius.daemon";
39
39
  var WIN_TASK_NAME = "OmniusDaemon";
40
40
  var REQUIRED_AIWG_VERSION = "2026.5.11";
41
- var INSTALL_STEPS = 7;
41
+ var INSTALL_STEPS = 8;
42
42
  var DISCOVERY_NOTICE_PRINTED = false;
43
43
 
44
44
  // ─── Helpers ────────────────────────────────────────────────────────────────
@@ -99,6 +99,68 @@ function hasCmd(name) {
99
99
  return runQuiet("which " + name);
100
100
  }
101
101
 
102
+ // Advanced OCR on JetPack uses Ubuntu's ABI-matched OpenCV/NumPy/Pillow stack.
103
+ // Native packages belong to package installation, never to an inference or
104
+ // readiness request. Jammy has no python3-pytesseract candidate, so that one
105
+ // pure-Python package is intentionally handled later by /v1/ocr/setup.
106
+ function ensureJetsonOcrNativeDependencies() {
107
+ if (!IS_LINUX) return;
108
+ var jetson = fs.existsSync("/etc/nv_tegra_release") || fs.existsSync("/sys/module/tegra_fuse");
109
+ if (!jetson) {
110
+ try {
111
+ var model = fs.readFileSync("/proc/device-tree/model", "utf8").replace(/\0/g, " ");
112
+ jetson = /\b(nvidia|jetson|orin|tegra)\b/i.test(model);
113
+ } catch (e) {}
114
+ }
115
+ if (!jetson || !hasCmd("apt-get") || !hasCmd("dpkg-query")) return;
116
+
117
+ var packages = [
118
+ "tesseract-ocr",
119
+ "python3-venv",
120
+ "python3-opencv",
121
+ "python3-numpy",
122
+ "python3-pil",
123
+ "python3-reportlab",
124
+ ];
125
+ var missing = packages.filter(function (pkg) {
126
+ return !runQuiet("dpkg-query -W -f='${Status}' " + shellQuote(pkg) + " 2>/dev/null | grep -q 'install ok installed'");
127
+ });
128
+ if (missing.length === 0) {
129
+ log("Jetson advanced-OCR native dependencies are installed.");
130
+ return;
131
+ }
132
+ var install = "apt-get install -y " + missing.map(shellQuote).join(" ");
133
+ if (process.getuid && process.getuid() === 0) {
134
+ log("Provisioning Jetson advanced-OCR native dependencies: " + missing.join(", "));
135
+ if (!runQuiet(install, { timeout: 300000, env: Object.assign({}, process.env, { DEBIAN_FRONTEND: "noninteractive" }) })) {
136
+ warn("Jetson OCR native dependency install failed. Run: sudo " + install);
137
+ }
138
+ return;
139
+ }
140
+ if (hasCmd("sudo") && runQuiet("sudo -n true")) {
141
+ log("Provisioning Jetson advanced-OCR native dependencies with the install-time sudo authorization: " + missing.join(", "));
142
+ if (!runQuiet("sudo -n " + install, { timeout: 300000, env: Object.assign({}, process.env, { DEBIAN_FRONTEND: "noninteractive" }) })) {
143
+ warn("Jetson OCR native dependency install failed. Run: sudo " + install);
144
+ }
145
+ return;
146
+ }
147
+ if (hasCmd("sudo") && process.stdin.isTTY && process.stdout.isTTY) {
148
+ log("Jetson advanced OCR needs native Ubuntu packages; requesting elevation during package installation.");
149
+ try {
150
+ cp.execSync("sudo " + install, {
151
+ stdio: "inherit",
152
+ timeout: 300000,
153
+ env: Object.assign({}, process.env, { DEBIAN_FRONTEND: "noninteractive" }),
154
+ });
155
+ return;
156
+ } catch (e) {
157
+ warn("Jetson OCR native dependency install was not authorized. Run: sudo " + install);
158
+ return;
159
+ }
160
+ }
161
+ warn("Jetson OCR native dependencies are missing; install outside inference: sudo " + install);
162
+ }
163
+
102
164
  function shellQuote(value) {
103
165
  if (IS_WIN) return '"' + String(value).replace(/"/g, '\\"') + '"';
104
166
  return "'" + String(value).replace(/'/g, "'\\''") + "'";
@@ -924,10 +986,13 @@ function main() {
924
986
  warn("wrapper auto-repair crashed (non-fatal): " + (e && e.message));
925
987
  }
926
988
 
927
- progress(3, INSTALL_STEPS, "Ensuring AIWG CLI");
989
+ progress(3, INSTALL_STEPS, "Provisioning Jetson OCR native dependencies");
990
+ ensureJetsonOcrNativeDependencies();
991
+
992
+ progress(4, INSTALL_STEPS, "Ensuring AIWG CLI");
928
993
  ensureGlobalAiwgInstall();
929
994
 
930
- progress(4, INSTALL_STEPS, "Ensuring local REST key");
995
+ progress(5, INSTALL_STEPS, "Ensuring local REST key");
931
996
  ensureBootstrapApiKeyForUser(effectiveUser());
932
997
 
933
998
  if (process.env.OMNIUS_UPDATE_COORDINATED === "1") {
@@ -946,7 +1011,7 @@ function main() {
946
1011
  // restart can't reach it, so we have to clean up explicitly. Without this,
947
1012
  // the new service-managed daemon fails to bind port 11435 and the user
948
1013
  // ends up running stale in-memory code from the previous version.
949
- progress(5, INSTALL_STEPS, "Clearing daemon port");
1014
+ progress(6, INSTALL_STEPS, "Clearing daemon port");
950
1015
  stopAttestedDaemonPortHolder(PORT, function (killedCount) {
951
1016
  if (killedCount > 0) {
952
1017
  log("Killed " + killedCount + " stale daemon process(es) holding port " + PORT + ".");
@@ -977,7 +1042,7 @@ function runMainAfterKill() {
977
1042
  // idempotent and ensures the unit file matches the current bundle.
978
1043
  }
979
1044
 
980
- progress(6, INSTALL_STEPS, "Installing daemon service");
1045
+ progress(7, INSTALL_STEPS, "Installing daemon service");
981
1046
  log("Installing Omnius API daemon service for port " + PORT + " ...");
982
1047
  log(" node: " + nodeBin);
983
1048
  log(" omnius script: " + omniusScript);
@@ -1008,7 +1073,7 @@ function runMainAfterKill() {
1008
1073
  }
1009
1074
 
1010
1075
  // Wait up to 15s for /health to come up, but don't fail npm install.
1011
- progress(7, INSTALL_STEPS, "Verifying daemon health");
1076
+ progress(8, INSTALL_STEPS, "Verifying daemon health");
1012
1077
  waitForHealth(15000, function (healthy) {
1013
1078
  if (healthy) {
1014
1079
  log("Omnius API daemon is live: http://127.0.0.1:" + PORT + "/health");
@@ -150,6 +150,10 @@ def read_wav(path: str, np):
150
150
  if sample_rate != 16000:
151
151
  raise RuntimeError(f"Expected 16 kHz WAV from the caller, received {sample_rate} Hz")
152
152
  waveform = np.frombuffer(raw, dtype="<i2").astype(np.float32) / 32768.0
153
+ if waveform.size > 96000:
154
+ raise RuntimeError(
155
+ f"Expected at most 6 seconds of caller-conditioned audio, received {waveform.size / sample_rate:.3f} seconds"
156
+ )
153
157
  return waveform, sample_rate, (time.perf_counter() - started) * 1000
154
158
 
155
159
 
@@ -310,7 +314,18 @@ def main() -> int:
310
314
  if not request_id or not file_path:
311
315
  raise RuntimeError("classify requires id and file")
312
316
  waveform, sample_rate, decode_ms = read_wav(file_path, worker.np)
313
- classifications, inference_ms = worker.infer(waveform, max(1, min(int(request.get("top_k", 5)), 25)))
317
+ rms = float(worker.np.sqrt(worker.np.mean(worker.np.square(waveform)))) if waveform.size else 0.0
318
+ peak = float(worker.np.max(worker.np.abs(waveform))) if waveform.size else 0.0
319
+ # Do not force YAMNet to invent an AudioSet label for digital or
320
+ # near-digital silence. This remains a valid, grounded result.
321
+ low_information = rms < 0.0005 and peak < 0.002
322
+ if low_information:
323
+ classifications, inference_ms = [], 0.0
324
+ else:
325
+ classifications, inference_ms = worker.infer(
326
+ waveform,
327
+ max(1, min(int(request.get("top_k", 5)), 25)),
328
+ )
314
329
  emit({
315
330
  "type": "result",
316
331
  "id": request_id,
@@ -318,6 +333,8 @@ def main() -> int:
318
333
  "sample_rate_hz": sample_rate,
319
334
  "duration_seconds": round(float(waveform.size) / sample_rate, 6),
320
335
  "classifications": classifications,
336
+ "low_information": low_information,
337
+ "acoustic": {"rms": round(rms, 8), "peak": round(peak, 8)},
321
338
  "timings_ms": {"decode": round(decode_ms, 3), "inference": round(inference_ms, 3)},
322
339
  })
323
340
  except Exception as exc: # keep the persistent worker alive per request
@@ -45,7 +45,7 @@ def check_deps():
45
45
  try:
46
46
  import cv2
47
47
  except ImportError:
48
- missing.append("opencv-python-headless")
48
+ missing.append("cv2")
49
49
  try:
50
50
  import numpy
51
51
  except ImportError:
@@ -57,12 +57,12 @@ def check_deps():
57
57
  try:
58
58
  from PIL import Image
59
59
  except ImportError:
60
- missing.append("Pillow")
60
+ missing.append("PIL")
61
61
 
62
62
  if missing:
63
63
  print(json.dumps({
64
- "error": f"Missing Python packages: {', '.join(missing)}. "
65
- f"Install with: pip install {' '.join(missing)}",
64
+ "error": f"OCR runtime imports became unavailable: {', '.join(missing)}. "
65
+ "Run POST /v1/ocr/setup and poll GET /v1/ocr/readiness; inference never installs packages.",
66
66
  "missing": missing,
67
67
  }))
68
68
  sys.exit(1)
@@ -179,10 +179,9 @@ def run_tesseract(binary_img, language="eng", psm=6):
179
179
  pil_img = Image.fromarray(binary_img)
180
180
  config = f"--psm {psm}"
181
181
 
182
- try:
183
- text = pytesseract.image_to_string(pil_img, lang=language, config=config).strip()
184
- except Exception:
185
- return "", 0.0, 0
182
+ # Engine failures are not blank documents. Let callers distinguish a real
183
+ # low-information image from a missing language pack or broken Tesseract.
184
+ text = pytesseract.image_to_string(pil_img, lang=language, config=config).strip()
186
185
 
187
186
  line_count = len([l for l in text.split("\n") if l.strip()])
188
187
 
@@ -322,6 +321,7 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
322
321
 
323
322
  # Generate all variants and run OCR
324
323
  all_results = {}
324
+ ocr_errors = []
325
325
  best_key = None
326
326
  best_score = -1
327
327
 
@@ -338,7 +338,11 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
338
338
 
339
339
  for psm in psm_modes:
340
340
  key = f"{vname}_psm{psm}"
341
- text, confidence, line_count = run_tesseract(binary, language, psm)
341
+ try:
342
+ text, confidence, line_count = run_tesseract(binary, language, psm)
343
+ except Exception as error:
344
+ ocr_errors.append(f"{key}: {error}")
345
+ continue
342
346
  char_count = len(text)
343
347
  score = compute_score(text, confidence, line_count)
344
348
 
@@ -355,7 +359,8 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
355
359
  best_key = key
356
360
 
357
361
  if not best_key:
358
- return {"error": "All OCR variants failed to produce output"}
362
+ detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
363
+ return {"error": f"Tesseract failed for every OCR variant: {detail}"}
359
364
 
360
365
  best = all_results[best_key]
361
366
  result = {
@@ -400,7 +405,10 @@ def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
400
405
  if debug_dir:
401
406
  cv2.imwrite(os.path.join(debug_dir, f"region_{rname}_{vname}.png"), binary)
402
407
 
403
- text, conf, lc = run_tesseract(binary, language, 6)
408
+ try:
409
+ text, conf, lc = run_tesseract(binary, language, 6)
410
+ except Exception:
411
+ continue
404
412
  score = compute_score(text, conf, lc)
405
413
  if score > region_best_score:
406
414
  region_best_score = score
@@ -531,7 +539,7 @@ def main():
531
539
  print(f"Processed {result['images_processed']} images → {result['output_dir']}")
532
540
  else:
533
541
  print(json.dumps(result, indent=2))
534
- sys.exit(0)
542
+ sys.exit(1 if "error" in result else 0)
535
543
 
536
544
  # Single image mode
537
545
  if not os.path.isfile(args.image):
@@ -565,6 +573,8 @@ def main():
565
573
  print(result["text"])
566
574
  else:
567
575
  print(json.dumps(result, indent=2))
576
+ if "error" in result:
577
+ sys.exit(1)
568
578
 
569
579
 
570
580
  if __name__ == "__main__":