omnius 1.0.640 → 1.0.642

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -37,10 +37,11 @@ MIN_RMS = 0.003
37
37
  MIN_PEAK = 0.01
38
38
  MAX_CLIPPED_FRACTION = 0.01
39
39
  FBANK_IMPLEMENTATION = "kaldi-fbank-numpy-v1"
40
- # SHA-256 over the rounded deterministic 2-second calibration fbank after
41
- # utterance CMN. This protects the source implementation from silent feature
42
- # drift while avoiding irrelevant last-bit FFT differences across CPU builds.
43
- FBANK_VALIDATION_SIGNATURE = "sha256:19bb5552b8068d5cd2ccdbb07ac7a7ed1137400d3bdc3c54fb9c21d4f06e16fa"
40
+ # A CPU FFT may differ in harmless last bits across x86/aarch64 and NumPy
41
+ # builds. Validate the public Kaldi configuration with numeric invariants,
42
+ # not a byte hash of implementation-dependent FFT output.
43
+ FBANK_VALIDATION = "kaldi-fbank-invariants-v2"
44
+ CALIBRATION_BAND_MEANS = (11.1405, 17.5231, 12.8845, 9.2557, 7.7213, 6.7842, 6.6405)
44
45
 
45
46
 
46
47
  def emit(payload: dict) -> None:
@@ -163,25 +164,46 @@ def kaldi_fbank80_numpy(waveform, np):
163
164
 
164
165
 
165
166
  def validate_kaldi_fbank_numpy(np):
166
- """Validate the no-Torch preprocessing path before readiness is true."""
167
+ """Validate the no-Torch preprocessing path before readiness is true.
168
+
169
+ The calibration intentionally asserts shape, finite/CMN invariants, the
170
+ expected 220/440 Hz Kaldi mel-band profile and energy range with tolerances
171
+ that are stable across supported CPU FFT implementations. It must reject a
172
+ changed window, frame layout, PCM scale, mel bank, pre-emphasis or CMN, but
173
+ must not reject Jetson due to last-bit numerical differences.
174
+ """
167
175
  t = np.arange(SAMPLE_RATE * 2, dtype=np.float32) / np.float32(SAMPLE_RATE)
168
176
  samples = (
169
177
  np.float32(0.075) * np.sin(np.float32(2.0 * np.pi * 220.0) * t)
170
178
  + np.float32(0.025) * np.sin(np.float32(2.0 * np.pi * 440.0) * t)
171
179
  ).astype(np.float32)
172
- features = kaldi_fbank80_numpy(samples * np.float32(1 << 15), np)
173
- features = (features - np.mean(features, axis=0, dtype=np.float32, keepdims=True)).astype(np.float32)
180
+ raw_features = kaldi_fbank80_numpy(samples * np.float32(1 << 15), np)
181
+ features = (
182
+ raw_features - np.mean(raw_features, axis=0, dtype=np.float32, keepdims=True)
183
+ ).astype(np.float32)
174
184
  if features.shape != (198, 80) or not np.all(np.isfinite(features)):
175
185
  raise RuntimeError("NumPy Kaldi fbank calibration produced an invalid feature matrix")
176
- if not np.all(np.abs(np.mean(features, axis=0)) < np.float32(5e-5)):
186
+ cmn_abs_mean_max = float(np.max(np.abs(np.mean(features, axis=0))))
187
+ if cmn_abs_mean_max > 1e-3:
177
188
  raise RuntimeError("NumPy Kaldi fbank calibration failed utterance CMN")
178
- rounded = np.rint(features * np.float32(10_000)).astype("<i4", copy=False)
179
- signature = "sha256:" + hashlib.sha256(rounded.tobytes()).hexdigest()
180
- if signature != FBANK_VALIDATION_SIGNATURE:
189
+ raw_min = float(np.min(raw_features))
190
+ raw_max = float(np.max(raw_features))
191
+ cmn_rms = float(np.sqrt(np.mean(np.square(features))))
192
+ band_means = raw_features.mean(axis=0)[[0, 5, 10, 20, 40, 60, 79]]
193
+ if not (4.5 < raw_min < 6.0 and 19.0 < raw_max < 21.5 and 0.55 < cmn_rms < 0.75):
181
194
  raise RuntimeError(
182
- "NumPy Kaldi fbank calibration signature mismatch: " + signature
195
+ "NumPy Kaldi fbank calibration energy/range invariant failed"
183
196
  )
184
- return signature
197
+ if not np.allclose(band_means, np.asarray(CALIBRATION_BAND_MEANS, dtype=np.float32), rtol=0.0, atol=0.25):
198
+ raise RuntimeError("NumPy Kaldi fbank calibration mel-band invariant failed")
199
+ return {
200
+ "validation": FBANK_VALIDATION,
201
+ "frames": int(features.shape[0]),
202
+ "bins": int(features.shape[1]),
203
+ "cmn_abs_mean_max": round(cmn_abs_mean_max, 8),
204
+ "cmn_rms": round(cmn_rms, 6),
205
+ "raw_feature_range": [round(raw_min, 6), round(raw_max, 6)],
206
+ }
185
207
 
186
208
 
187
209
  class WeSpeakerCamPlus:
@@ -191,7 +213,7 @@ class WeSpeakerCamPlus:
191
213
  import onnxruntime as ort
192
214
 
193
215
  self.np = np
194
- self.preprocessing_signature = validate_kaldi_fbank_numpy(np)
216
+ self.preprocessing_validation = validate_kaldi_fbank_numpy(np)
195
217
  options = ort.SessionOptions()
196
218
  options.inter_op_num_threads = 1
197
219
  options.intra_op_num_threads = 1
@@ -265,7 +287,7 @@ def main() -> int:
265
287
  {
266
288
  "type": "preprocessing_probe",
267
289
  "implementation": FBANK_IMPLEMENTATION,
268
- "validation_signature": validate_kaldi_fbank_numpy(np),
290
+ "validation": validate_kaldi_fbank_numpy(np),
269
291
  }
270
292
  )
271
293
  return 0
@@ -285,7 +307,7 @@ def main() -> int:
285
307
  "model_load_ms": round(worker.model_load_ms, 3),
286
308
  "warmed": True,
287
309
  "preprocessing": FBANK_IMPLEMENTATION,
288
- "preprocessing_signature": worker.preprocessing_signature,
310
+ "preprocessing_validation": worker.preprocessing_validation,
289
311
  }
290
312
  )
291
313
  for line in sys.stdin:
@@ -179,7 +179,17 @@ SMALL_CROP_AREA_PX = 512_000
179
179
  MEDIUM_IMAGE_AREA_PX = 2_000_000
180
180
  MAX_PIPELINE_DEADLINE_MS = 80_000
181
181
  MAX_TESSERACT_ATTEMPT_SECONDS = 12.0
182
+ # Small conditioned crops normally complete in well under a second. A
183
+ # six-second cap still permits a busy Tesseract child, while keeping a failed
184
+ # recovery pair from consuming most of the REST deadline.
185
+ SMALL_CROP_TESSERACT_ATTEMPT_SECONDS = 6.0
182
186
  MIN_TESSERACT_ATTEMPT_SECONDS = 0.25
187
+ MIN_ACCEPTED_CONFIDENCE = 50.0
188
+ MIN_SUBSTANTIVE_TEXT_CHARS = 12
189
+ HIGH_VOLUME_GARBAGE_CHARS = 64
190
+ HIGH_VOLUME_GARBAGE_CONFIDENCE = 35.0
191
+ TERMINAL_SYMBOL_GARBAGE_CHARS = 24
192
+ TERMINAL_SYMBOL_GARBAGE_ALNUM_RATIO = 0.20
183
193
  ACTIVE_DEADLINE = None
184
194
 
185
195
 
@@ -207,11 +217,11 @@ class OcrDeadline:
207
217
  if self.remaining_seconds() <= 0:
208
218
  raise OcrPipelineTimeout(f"OCR deadline exceeded during {stage}")
209
219
 
210
- def tesseract_timeout_seconds(self):
220
+ def tesseract_timeout_seconds(self, attempt_cap_seconds=MAX_TESSERACT_ATTEMPT_SECONDS):
211
221
  self.check("Tesseract scheduling")
212
222
  return max(
213
223
  MIN_TESSERACT_ATTEMPT_SECONDS,
214
- min(MAX_TESSERACT_ATTEMPT_SECONDS, self.remaining_seconds()),
224
+ min(attempt_cap_seconds, self.remaining_seconds()),
215
225
  )
216
226
 
217
227
 
@@ -247,7 +257,8 @@ def text_from_tesseract_data(data):
247
257
  return "\n".join(" ".join(words) for words in lines.values()).strip()
248
258
 
249
259
 
250
- def run_tesseract(binary_img, deadline, language="eng", psm=6):
260
+ def run_tesseract(binary_img, deadline, language="eng", psm=6,
261
+ attempt_cap_seconds=MAX_TESSERACT_ATTEMPT_SECONDS):
251
262
  """Run one bounded Tesseract TSV pass and derive text plus confidence."""
252
263
  deadline.check("Tesseract")
253
264
  pil_img = Image.fromarray(binary_img)
@@ -257,7 +268,7 @@ def run_tesseract(binary_img, deadline, language="eng", psm=6):
257
268
  lang=language,
258
269
  config=config,
259
270
  output_type=pytesseract.Output.DICT,
260
- timeout=deadline.tesseract_timeout_seconds(),
271
+ timeout=deadline.tesseract_timeout_seconds(attempt_cap_seconds),
261
272
  )
262
273
  text = text_from_tesseract_data(data)
263
274
  confs = []
@@ -273,14 +284,87 @@ def run_tesseract(binary_img, deadline, language="eng", psm=6):
273
284
  return text, avg_conf, line_count
274
285
 
275
286
 
276
- def compute_score(text, confidence, line_count):
287
+ def assess_ocr_evidence(text, confidence, line_count):
288
+ """Classify OCR output before it can become agent-visible evidence."""
289
+ normalized = str(text or "").strip()
290
+ chars = len(normalized)
291
+ if chars == 0:
292
+ return {
293
+ "state": "low_information",
294
+ "accepted": False,
295
+ "reason": "no_readable_text",
296
+ "chars": 0,
297
+ "confidence": round(float(confidence), 1),
298
+ "lines": int(line_count),
299
+ }
300
+ alnum_ratio = sum(character.isalnum() for character in normalized) / max(1, chars)
301
+ if chars < MIN_SUBSTANTIVE_TEXT_CHARS and confidence < MIN_ACCEPTED_CONFIDENCE:
302
+ return {
303
+ "state": "low_information",
304
+ "accepted": False,
305
+ "reason": "insufficient_low_confidence_text",
306
+ "chars": chars,
307
+ "confidence": round(float(confidence), 1),
308
+ "lines": int(line_count),
309
+ "alnum_ratio": round(alnum_ratio, 3),
310
+ }
311
+ if (
312
+ (chars >= HIGH_VOLUME_GARBAGE_CHARS and confidence < HIGH_VOLUME_GARBAGE_CONFIDENCE)
313
+ or confidence < MIN_ACCEPTED_CONFIDENCE
314
+ or alnum_ratio < 0.45
315
+ ):
316
+ reason = (
317
+ "high_volume_low_confidence_text"
318
+ if chars >= HIGH_VOLUME_GARBAGE_CHARS and confidence < HIGH_VOLUME_GARBAGE_CONFIDENCE
319
+ else "low_confidence_or_symbol_heavy_text"
320
+ )
321
+ return {
322
+ "state": "rejected",
323
+ "accepted": False,
324
+ "reason": reason,
325
+ "chars": chars,
326
+ "confidence": round(float(confidence), 1),
327
+ "lines": int(line_count),
328
+ "alnum_ratio": round(alnum_ratio, 3),
329
+ }
330
+ return {
331
+ "state": "accepted",
332
+ "accepted": True,
333
+ "reason": "confidence_and_text_quality_met",
334
+ "chars": chars,
335
+ "confidence": round(float(confidence), 1),
336
+ "lines": int(line_count),
337
+ "alnum_ratio": round(alnum_ratio, 3),
338
+ }
339
+
340
+
341
+ def is_terminal_small_crop_rejection(evidence):
342
+ """Whether a first small-crop pass proves another recovery pass is futile.
343
+
344
+ Blanks and plausible alphanumeric low-confidence text remain recoverable
345
+ through the second variant. A large very-low-confidence transcript or a
346
+ distinctly symbol-heavy stream cannot become safe evidence through another
347
+ PSM6 pass, and should not make callers wait for one.
348
+ """
349
+ if evidence.get("state") != "rejected":
350
+ return False
351
+ if evidence.get("reason") == "high_volume_low_confidence_text":
352
+ return True
353
+ return (
354
+ evidence.get("chars", 0) >= TERMINAL_SYMBOL_GARBAGE_CHARS
355
+ and evidence.get("alnum_ratio", 1.0) < TERMINAL_SYMBOL_GARBAGE_ALNUM_RATIO
356
+ )
357
+
358
+
359
+ def compute_score(text, confidence, line_count, evidence=None):
277
360
  """Combined scoring heuristic:
278
361
  - confidence * sqrt(char_count) — rewards quality and coverage
279
362
  - + line_count * 10 — bonus for structured output (more lines = better parse)
280
363
  The agent discovered that line-count is a strong proxy for successful parsing
281
364
  on structured documents like invoices and forms."""
365
+ quality = evidence or assess_ocr_evidence(text, confidence, line_count)
282
366
  char_count = len(text)
283
- if char_count == 0:
367
+ if not quality["accepted"] or char_count == 0:
284
368
  return 0
285
369
  return confidence * (char_count ** 0.5) + line_count * 10
286
370
 
@@ -319,7 +403,8 @@ def build_ocr_plan(image_area_px, single_psm=None):
319
403
 
320
404
 
321
405
  def has_sufficient_evidence(text, confidence, line_count):
322
- return len(text.strip()) >= 8 and confidence >= 70.0 and line_count >= 1
406
+ quality = assess_ocr_evidence(text, confidence, line_count)
407
+ return quality["accepted"] and confidence >= 70.0 and line_count >= 1
323
408
 
324
409
 
325
410
  # ---------------------------------------------------------------------------
@@ -391,15 +476,18 @@ def write_all_outputs(text, base_name, output_dir):
391
476
  # Main pipeline
392
477
  # ---------------------------------------------------------------------------
393
478
 
394
- def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefix="full"):
479
+ def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefix="full",
480
+ attempt_cap_seconds=MAX_TESSERACT_ATTEMPT_SECONDS,
481
+ stop_on_terminal_rejection=False):
395
482
  """Run a bounded plan, stopping early once a legible result is proven."""
396
483
  all_results = {}
397
484
  ocr_errors = []
398
485
  best_key = None
399
486
  best_score = -1
400
487
  early_exit = False
488
+ terminal_rejection = False
401
489
 
402
- for vname, psm in plan:
490
+ for attempt_index, (vname, psm) in enumerate(plan):
403
491
  deadline.check("preprocessing")
404
492
  key = f"{vname}_psm{psm}"
405
493
  try:
@@ -411,7 +499,9 @@ def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefi
411
499
  os.makedirs(debug_dir, exist_ok=True)
412
500
  cv2.imwrite(os.path.join(debug_dir, f"{debug_prefix}_{vname}.png"), binary)
413
501
  try:
414
- text, confidence, line_count = run_tesseract(binary, deadline, language, psm)
502
+ text, confidence, line_count = run_tesseract(
503
+ binary, deadline, language, psm, attempt_cap_seconds
504
+ )
415
505
  except OcrPipelineTimeout:
416
506
  raise
417
507
  except OcrPipelineCancelled:
@@ -420,22 +510,31 @@ def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefi
420
510
  ocr_errors.append(f"{key}: {error}")
421
511
  continue
422
512
  char_count = len(text)
423
- score = compute_score(text, confidence, line_count)
513
+ evidence = assess_ocr_evidence(text, confidence, line_count)
514
+ score = compute_score(text, confidence, line_count, evidence)
424
515
  all_results[key] = {
425
516
  "text": text,
426
517
  "chars": char_count,
427
518
  "lines": line_count,
428
519
  "confidence": round(confidence, 1),
429
520
  "score": round(score, 1),
521
+ "evidence": evidence,
430
522
  }
431
- if score > best_score:
523
+ if evidence["accepted"] and score > best_score:
432
524
  best_score = score
433
525
  best_key = key
434
526
  if has_sufficient_evidence(text, confidence, line_count):
435
527
  early_exit = True
436
528
  break
529
+ if (
530
+ stop_on_terminal_rejection
531
+ and attempt_index == 0
532
+ and is_terminal_small_crop_rejection(evidence)
533
+ ):
534
+ terminal_rejection = True
535
+ break
437
536
 
438
- return all_results, ocr_errors, best_key, early_exit
537
+ return all_results, ocr_errors, best_key, early_exit, terminal_rejection
439
538
 
440
539
 
441
540
  def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_dir=None,
@@ -460,17 +559,89 @@ def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_d
460
559
  deadline.check("image conditioning")
461
560
  gray_2x = upscale_2x(gray)
462
561
  plan = build_ocr_plan(effective_area, single_psm)
562
+ is_small_crop = effective_area <= SMALL_CROP_AREA_PX
563
+ attempt_cap_seconds = (
564
+ SMALL_CROP_TESSERACT_ATTEMPT_SECONDS
565
+ if is_small_crop
566
+ else MAX_TESSERACT_ATTEMPT_SECONDS
567
+ )
463
568
  attempts_completed = 0
464
569
  try:
465
- all_results, ocr_errors, best_key, early_exit = run_variant_plan(
466
- gray_2x, plan, deadline, language, debug_dir
570
+ all_results, ocr_errors, best_key, early_exit, terminal_rejection = run_variant_plan(
571
+ gray_2x,
572
+ plan,
573
+ deadline,
574
+ language,
575
+ debug_dir,
576
+ attempt_cap_seconds=attempt_cap_seconds,
577
+ stop_on_terminal_rejection=is_small_crop,
467
578
  )
468
579
  attempts_completed = len(all_results)
469
580
  if not best_key:
470
- detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
471
- return {"error": f"Tesseract failed for every bounded OCR attempt: {detail}"}
581
+ if not all_results:
582
+ detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
583
+ return {"error": f"Tesseract failed for every bounded OCR attempt: {detail}"}
584
+ rejected = [item for item in all_results.values() if item["evidence"]["state"] == "rejected"]
585
+ if rejected:
586
+ worst = max(rejected, key=lambda item: (item["chars"], -item["confidence"]))
587
+ evidence = worst["evidence"]
588
+ message = (
589
+ "OCR text was suppressed because it did not meet evidence-quality requirements "
590
+ f"({evidence['reason']}; {evidence['chars']} chars at {evidence['confidence']}% confidence)."
591
+ )
592
+ return {
593
+ "error": message,
594
+ "diagnostic": {
595
+ **diagnostic("ocr_evidence_rejected", message, deadline, "evidence_quality", attempts_completed, len(plan)),
596
+ "evidence": evidence,
597
+ "terminal_small_crop_rejection": terminal_rejection,
598
+ "attempt_cap_seconds": attempt_cap_seconds,
599
+ },
600
+ }
601
+ # Empty or tiny non-substantive detections are valid observations:
602
+ # do not invent text and do not report them as a pipeline error.
603
+ result = {
604
+ "text": "",
605
+ "confidence": 0.0,
606
+ "variant": "none",
607
+ "chars": 0,
608
+ "lines": 0,
609
+ "score": 0.0,
610
+ "image_size": f"{w_orig}x{h_orig}",
611
+ "variants_tested": len(all_results),
612
+ "all_variants": {},
613
+ "quality": {
614
+ "schema": "omnius.ocr-evidence.v1",
615
+ "state": "low_information",
616
+ "accepted": False,
617
+ "reason": "no_accepted_readable_text",
618
+ "low_information_variants": len(all_results),
619
+ },
620
+ "diagnostic": diagnostic(
621
+ "ocr_low_information",
622
+ "OCR produced no accepted readable text; the result is low-information rather than evidence.",
623
+ deadline,
624
+ "evidence_quality",
625
+ attempts_completed,
626
+ len(plan),
627
+ ),
628
+ "strategy": {
629
+ "effective_area_px": effective_area,
630
+ "attempts_planned": len(plan),
631
+ "attempts_completed": attempts_completed,
632
+ "early_exit": False,
633
+ "terminal_small_crop_rejection": terminal_rejection,
634
+ "attempt_cap_seconds": attempt_cap_seconds,
635
+ "deadline_ms": deadline.deadline_ms,
636
+ },
637
+ }
638
+ return result
472
639
 
473
640
  best = all_results[best_key]
641
+ accepted_results = {
642
+ key: value for key, value in all_results.items()
643
+ if value["evidence"]["state"] == "accepted"
644
+ }
474
645
  result = {
475
646
  "text": best["text"],
476
647
  "confidence": best["confidence"],
@@ -480,12 +651,25 @@ def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_d
480
651
  "score": best["score"],
481
652
  "image_size": f"{w_orig}x{h_orig}",
482
653
  "variants_tested": len(all_results),
483
- "all_variants": all_results,
654
+ # Never leak rejected raw OCR as alternate evidence to agents.
655
+ "all_variants": accepted_results,
656
+ "quality": {
657
+ "schema": "omnius.ocr-evidence.v1",
658
+ **best["evidence"],
659
+ "rejected_variants": sum(
660
+ 1 for item in all_results.values() if item["evidence"]["state"] == "rejected"
661
+ ),
662
+ "low_information_variants": sum(
663
+ 1 for item in all_results.values() if item["evidence"]["state"] == "low_information"
664
+ ),
665
+ },
484
666
  "strategy": {
485
667
  "effective_area_px": effective_area,
486
668
  "attempts_planned": len(plan),
487
669
  "attempts_completed": attempts_completed,
488
670
  "early_exit": early_exit,
671
+ "terminal_small_crop_rejection": terminal_rejection,
672
+ "attempt_cap_seconds": attempt_cap_seconds,
489
673
  "deadline_ms": deadline.deadline_ms,
490
674
  },
491
675
  }
@@ -498,9 +682,22 @@ def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_d
498
682
  region_gray = extract_region(gray_2x, y_start, y_end)
499
683
  # Region requests use at most two high-yield attempts. The
500
684
  # main result already provides full-frame coverage.
501
- region_plan = build_ocr_plan(region_gray.shape[0] * region_gray.shape[1], single_psm)[:2]
502
- region_results, _errors, region_best_key, _early_exit = run_variant_plan(
503
- region_gray, region_plan, deadline, language, debug_dir, f"region_{rname}"
685
+ region_area = region_gray.shape[0] * region_gray.shape[1]
686
+ region_plan = build_ocr_plan(region_area, single_psm)[:2]
687
+ region_is_small = region_area <= SMALL_CROP_AREA_PX
688
+ region_results, _errors, region_best_key, _early_exit, _terminal_rejection = run_variant_plan(
689
+ region_gray,
690
+ region_plan,
691
+ deadline,
692
+ language,
693
+ debug_dir,
694
+ f"region_{rname}",
695
+ attempt_cap_seconds=(
696
+ SMALL_CROP_TESSERACT_ATTEMPT_SECONDS
697
+ if region_is_small
698
+ else MAX_TESSERACT_ATTEMPT_SECONDS
699
+ ),
700
+ stop_on_terminal_rejection=region_is_small,
504
701
  )
505
702
  regions[rname] = region_results[region_best_key]["text"] if region_best_key else ""
506
703
  result["regions"] = regions
@@ -5399,7 +5399,7 @@
5399
5399
  "tags": [
5400
5400
  "Audio"
5401
5401
  ],
5402
- "description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses the pinned JetPack YAMNet/TensorRT setup; speaker creates a private CPython 3.10 CPU-only WeSpeaker CAM++ venv, installs only checksum-pinned NumPy, ONNX Runtime, and its locked CPU support wheels with --no-deps/--no-index, then downloads the pinned model. Its 80-bin Kaldi-compatible NumPy fbank is validated with a deterministic signature before readiness. The speaker path never imports, links, replaces, or otherwise depends on JetPack Torch/Torchaudio, so an incompatible generic Torchaudio wheel cannot affect CUDA-enabled Egg Torch. semantic installs the isolated JetPack CUDA CLAP dependencies and pinned model. This is the only REST operation allowed to provision. It activates and warms only the requested role worker; no role is substituted for another.",
5402
+ "description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses the pinned JetPack YAMNet/TensorRT setup; speaker creates a private CPython 3.10 CPU-only WeSpeaker CAM++ venv, installs only checksum-pinned NumPy, ONNX Runtime, and its locked CPU support wheels with --no-deps/--no-index, then downloads the pinned model. Its 80-bin Kaldi-compatible NumPy fbank is validated with a versioned cross-platform numeric-invariant probe before readiness. The speaker path never imports, links, replaces, or otherwise depends on JetPack Torch/Torchaudio, so an incompatible generic Torchaudio wheel cannot affect CUDA-enabled Egg Torch. semantic installs the isolated JetPack CUDA CLAP dependencies and pinned model. This is the only REST operation allowed to provision. It activates and warms only the requested role worker; no role is substituted for another.",
5403
5403
  "parameters": [
5404
5404
  {
5405
5405
  "name": "kind",
@@ -6156,6 +6156,18 @@
6156
6156
  "default": 45,
6157
6157
  "description": "Total server-side agent-loop deadline. Distinct from per-backend timeout_s."
6158
6158
  },
6159
+ "agent_max_tool_rounds": {
6160
+ "type": "integer",
6161
+ "minimum": 1,
6162
+ "maximum": 8,
6163
+ "default": 1,
6164
+ "description": "Maximum daemon-tool planning rounds. The default executes one tool round, then removes daemon schemas for lower-latency final synthesis."
6165
+ },
6166
+ "agent_prefetch_web_search": {
6167
+ "type": "boolean",
6168
+ "default": false,
6169
+ "description": "Explicitly execute authorized web_search with the latest user text before one backend synthesis. factual-first enables this automatically."
6170
+ },
6159
6171
  "max_turns": {
6160
6172
  "type": "integer",
6161
6173
  "description": "Q2 — agent_loop max iterations (default 8, max 64)."
@@ -6165,7 +6177,7 @@
6165
6177
  "enum": [
6166
6178
  "factual-first"
6167
6179
  ],
6168
- "description": "Q8 — prepended system policy template. 'factual-first' instructs model to call web_search FIRST for any factual question."
6180
+ "description": "Factual-first prefetches authorized web_search from the latest user turn and performs one grounded synthesis generation."
6169
6181
  }
6170
6182
  }
6171
6183
  }
@@ -46,6 +46,8 @@ Important body fields:
46
46
  | `include_daemon_tools` | array | Permit the bounded core daemon-tool catalog by scope: `read`, `run`, `admin` |
47
47
  | `daemon_tool_names` | array | Exact daemon-tool allowlist; recommended for local models |
48
48
  | `agent_timeout_s` | number | Whole-loop deadline, default 45 seconds and maximum 600 |
49
+ | `agent_max_tool_rounds` | integer | Daemon tool rounds before forced final synthesis; default 1 |
50
+ | `agent_prefetch_web_search` | boolean | Explicit one-generation web-search prefetch; factual-first enables it automatically |
49
51
  | `max_turns` | integer | Server-side agent loop turn cap |
50
52
  | `prompt_template` | string | Optional template such as `factual-first` |
51
53
 
@@ -129,4 +131,4 @@ For ASR/TTS systems that only need the text brain, use `/realtime` or `/v1/realt
129
131
 
130
132
  `/v1/chat/completions` can run an internal tool loop when `agent_loop: true`. This lets clients collapse multiple model/tool round trips into one daemon request. Daemon tool calls execute inline; client-owned tool calls can still be yielded in OpenAI-compatible shape.
131
133
 
132
- Ollama-backed loops use its native `/api/chat` tool protocol. `timeout_s` applies to each backend round, while `agent_timeout_s` bounds the complete loop and defaults to 45 seconds. Omnius returns a typed HTTP 504 when that budget expires and HTTP 508 when a model repeats the same daemon tool with identical arguments. Tool results are bounded before the next prompt. Without `daemon_tool_names`, Omnius offers only a compact core catalog permitted by `include_daemon_tools`, avoiding a huge local-model prompt; request any other tools by exact name. `prompt_template: "factual-first"` narrows the implicit catalog further to `web_search` and `web_fetch`.
134
+ Ollama-backed loops use its native `/api/chat` tool protocol. `timeout_s` applies to each backend round, while `agent_timeout_s` bounds the complete loop and defaults to 45 seconds. For a normal loop, the planning turn is capped at 96 output tokens; one daemon-tool round is the default, after which Omnius removes daemon schemas for final synthesis. `prompt_template: "factual-first"` skips that model-planning round entirely: Omnius executes the already-mandated, authorized `web_search` using the latest user turn, inserts the protocol-correct tool evidence, and performs one grounded synthesis generation. `agent_prefetch_web_search: true` opts into the same path directly. Generic/deeper tool workflows remain available through `agent_max_tool_rounds`. Omnius returns typed HTTP 504/508 failures, caps tool evidence at 6,000 characters, and limits the implicit catalog unless `daemon_tool_names` requests exact additions.
@@ -128,13 +128,23 @@ needs `run` scope because optional
128
128
 
129
129
  The managed pipeline is bounded: it starts with high-yield preprocessing and
130
130
  expands variants only for low-evidence or larger images. Small crops avoid the
131
- former all-variant/all-PSM explosion. The REST default is 90 seconds (maximum
132
- 180 seconds), while the worker has an 80-second internal deadline so it can
133
- return diagnostics. A timeout or cancellation returns
131
+ former all-variant/all-PSM explosion and use a six-second cap per Tesseract
132
+ attempt. If the first small-crop pass proves high-volume very-low-confidence or
133
+ distinctly symbol-heavy garbage, recovery stops immediately; blanks and
134
+ plausibly recoverable text still receive the second high-yield attempt. The
135
+ REST default is 90 seconds (maximum 180 seconds), while the worker has an
136
+ 80-second internal deadline so it can return diagnostics. A timeout or cancellation returns
134
137
  `result.data.schema=omnius.ocr-diagnostic.v1` with code `ocr_timeout` or
135
138
  `ocr_cancelled`; cancellation terminates the Python/Tesseract process group
136
139
  with TERM followed by KILL.
137
140
 
141
+ OCR text is evidence-gated before it is returned. Strong text with adequate
142
+ confidence is accepted. An empty or tiny non-substantive result is a successful
143
+ `omnius.ocr-evidence.v1` `low_information` observation with diagnostic code
144
+ `ocr_low_information`, not a fabricated transcript. High-volume low-confidence
145
+ or symbol-heavy output is suppressed and returned as
146
+ `ocr_evidence_rejected`; its raw text is not exposed as alternate OCR evidence.
147
+
138
148
  ## TTS
139
149
 
140
150
  `POST /v1/voice/tts` returns audio bytes. `format` can be `wav` or `pcm`. `X-Sample-Rate` reports the sample rate.
@@ -270,8 +280,9 @@ managed venv installs it with `--no-deps --no-index`.
270
280
  The worker implements the WeSpeaker CAM++ 80-bin Kaldi configuration in pure
271
281
  NumPy: 25 ms / 10 ms Hamming frames, dither disabled, Kaldi pre-emphasis and
272
282
  mel bank behavior, then full-clip CMN without CVN. Readiness invokes a
273
- deterministic CPU preprocessing probe and requires its pinned rounded-feature
274
- SHA-256 signature before it can report ready. The speaker path neither imports
283
+ versioned CPU preprocessing probe. It checks the fixed Kaldi configuration,
284
+ feature shape, CMN, energy range, and expected mel-band profile with explicit
285
+ cross-platform numeric tolerances before it can report ready. The speaker path neither imports
275
286
  nor links Torch or Torchaudio, so the generic `torchaudio-2.2.0` ABI mismatch
276
287
  cannot bind against, replace, or otherwise affect JetPack's CUDA-enabled Egg
277
288
  Torch. Inference remains install-free and network-free; failed package imports
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "omnius",
3
- "version": "1.0.640",
3
+ "version": "1.0.642",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "omnius",
9
- "version": "1.0.640",
9
+ "version": "1.0.642",
10
10
  "bundleDependencies": [
11
11
  "image-to-ascii"
12
12
  ],
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "omnius",
3
- "version": "1.0.640",
3
+ "version": "1.0.642",
4
4
  "description": "AI coding agent powered by open-source models (Ollama/vLLM) — interactive TUI with agentic tool-calling loop",
5
5
  "type": "module",
6
6
  "main": "./dist/library.js",