omnius 1.0.640 → 1.0.642
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +370 -54
- package/dist/scripts/audio-speaker-embedding-worker.py +38 -16
- package/dist/scripts/ocr-advanced.py +218 -21
- package/docs/DISCOVERY.json +14 -2
- package/docs/rest/endpoints/chat.md +3 -1
- package/docs/rest/endpoints/voice-vision.md +16 -5
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
|
@@ -37,10 +37,11 @@ MIN_RMS = 0.003
|
|
|
37
37
|
MIN_PEAK = 0.01
|
|
38
38
|
MAX_CLIPPED_FRACTION = 0.01
|
|
39
39
|
FBANK_IMPLEMENTATION = "kaldi-fbank-numpy-v1"
|
|
40
|
-
#
|
|
41
|
-
#
|
|
42
|
-
#
|
|
43
|
-
|
|
40
|
+
# A CPU FFT may differ in harmless last bits across x86/aarch64 and NumPy
|
|
41
|
+
# builds. Validate the public Kaldi configuration with numeric invariants,
|
|
42
|
+
# not a byte hash of implementation-dependent FFT output.
|
|
43
|
+
FBANK_VALIDATION = "kaldi-fbank-invariants-v2"
|
|
44
|
+
CALIBRATION_BAND_MEANS = (11.1405, 17.5231, 12.8845, 9.2557, 7.7213, 6.7842, 6.6405)
|
|
44
45
|
|
|
45
46
|
|
|
46
47
|
def emit(payload: dict) -> None:
|
|
@@ -163,25 +164,46 @@ def kaldi_fbank80_numpy(waveform, np):
|
|
|
163
164
|
|
|
164
165
|
|
|
165
166
|
def validate_kaldi_fbank_numpy(np):
|
|
166
|
-
"""Validate the no-Torch preprocessing path before readiness is true.
|
|
167
|
+
"""Validate the no-Torch preprocessing path before readiness is true.
|
|
168
|
+
|
|
169
|
+
The calibration intentionally asserts shape, finite/CMN invariants, the
|
|
170
|
+
expected 220/440 Hz Kaldi mel-band profile and energy range with tolerances
|
|
171
|
+
that are stable across supported CPU FFT implementations. It must reject a
|
|
172
|
+
changed window, frame layout, PCM scale, mel bank, pre-emphasis or CMN, but
|
|
173
|
+
must not reject Jetson due to last-bit numerical differences.
|
|
174
|
+
"""
|
|
167
175
|
t = np.arange(SAMPLE_RATE * 2, dtype=np.float32) / np.float32(SAMPLE_RATE)
|
|
168
176
|
samples = (
|
|
169
177
|
np.float32(0.075) * np.sin(np.float32(2.0 * np.pi * 220.0) * t)
|
|
170
178
|
+ np.float32(0.025) * np.sin(np.float32(2.0 * np.pi * 440.0) * t)
|
|
171
179
|
).astype(np.float32)
|
|
172
|
-
|
|
173
|
-
features = (
|
|
180
|
+
raw_features = kaldi_fbank80_numpy(samples * np.float32(1 << 15), np)
|
|
181
|
+
features = (
|
|
182
|
+
raw_features - np.mean(raw_features, axis=0, dtype=np.float32, keepdims=True)
|
|
183
|
+
).astype(np.float32)
|
|
174
184
|
if features.shape != (198, 80) or not np.all(np.isfinite(features)):
|
|
175
185
|
raise RuntimeError("NumPy Kaldi fbank calibration produced an invalid feature matrix")
|
|
176
|
-
|
|
186
|
+
cmn_abs_mean_max = float(np.max(np.abs(np.mean(features, axis=0))))
|
|
187
|
+
if cmn_abs_mean_max > 1e-3:
|
|
177
188
|
raise RuntimeError("NumPy Kaldi fbank calibration failed utterance CMN")
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
189
|
+
raw_min = float(np.min(raw_features))
|
|
190
|
+
raw_max = float(np.max(raw_features))
|
|
191
|
+
cmn_rms = float(np.sqrt(np.mean(np.square(features))))
|
|
192
|
+
band_means = raw_features.mean(axis=0)[[0, 5, 10, 20, 40, 60, 79]]
|
|
193
|
+
if not (4.5 < raw_min < 6.0 and 19.0 < raw_max < 21.5 and 0.55 < cmn_rms < 0.75):
|
|
181
194
|
raise RuntimeError(
|
|
182
|
-
"NumPy Kaldi fbank calibration
|
|
195
|
+
"NumPy Kaldi fbank calibration energy/range invariant failed"
|
|
183
196
|
)
|
|
184
|
-
|
|
197
|
+
if not np.allclose(band_means, np.asarray(CALIBRATION_BAND_MEANS, dtype=np.float32), rtol=0.0, atol=0.25):
|
|
198
|
+
raise RuntimeError("NumPy Kaldi fbank calibration mel-band invariant failed")
|
|
199
|
+
return {
|
|
200
|
+
"validation": FBANK_VALIDATION,
|
|
201
|
+
"frames": int(features.shape[0]),
|
|
202
|
+
"bins": int(features.shape[1]),
|
|
203
|
+
"cmn_abs_mean_max": round(cmn_abs_mean_max, 8),
|
|
204
|
+
"cmn_rms": round(cmn_rms, 6),
|
|
205
|
+
"raw_feature_range": [round(raw_min, 6), round(raw_max, 6)],
|
|
206
|
+
}
|
|
185
207
|
|
|
186
208
|
|
|
187
209
|
class WeSpeakerCamPlus:
|
|
@@ -191,7 +213,7 @@ class WeSpeakerCamPlus:
|
|
|
191
213
|
import onnxruntime as ort
|
|
192
214
|
|
|
193
215
|
self.np = np
|
|
194
|
-
self.
|
|
216
|
+
self.preprocessing_validation = validate_kaldi_fbank_numpy(np)
|
|
195
217
|
options = ort.SessionOptions()
|
|
196
218
|
options.inter_op_num_threads = 1
|
|
197
219
|
options.intra_op_num_threads = 1
|
|
@@ -265,7 +287,7 @@ def main() -> int:
|
|
|
265
287
|
{
|
|
266
288
|
"type": "preprocessing_probe",
|
|
267
289
|
"implementation": FBANK_IMPLEMENTATION,
|
|
268
|
-
"
|
|
290
|
+
"validation": validate_kaldi_fbank_numpy(np),
|
|
269
291
|
}
|
|
270
292
|
)
|
|
271
293
|
return 0
|
|
@@ -285,7 +307,7 @@ def main() -> int:
|
|
|
285
307
|
"model_load_ms": round(worker.model_load_ms, 3),
|
|
286
308
|
"warmed": True,
|
|
287
309
|
"preprocessing": FBANK_IMPLEMENTATION,
|
|
288
|
-
"
|
|
310
|
+
"preprocessing_validation": worker.preprocessing_validation,
|
|
289
311
|
}
|
|
290
312
|
)
|
|
291
313
|
for line in sys.stdin:
|
|
@@ -179,7 +179,17 @@ SMALL_CROP_AREA_PX = 512_000
|
|
|
179
179
|
MEDIUM_IMAGE_AREA_PX = 2_000_000
|
|
180
180
|
MAX_PIPELINE_DEADLINE_MS = 80_000
|
|
181
181
|
MAX_TESSERACT_ATTEMPT_SECONDS = 12.0
|
|
182
|
+
# Small conditioned crops normally complete in well under a second. A
|
|
183
|
+
# six-second cap still permits a busy Tesseract child, while keeping a failed
|
|
184
|
+
# recovery pair from consuming most of the REST deadline.
|
|
185
|
+
SMALL_CROP_TESSERACT_ATTEMPT_SECONDS = 6.0
|
|
182
186
|
MIN_TESSERACT_ATTEMPT_SECONDS = 0.25
|
|
187
|
+
MIN_ACCEPTED_CONFIDENCE = 50.0
|
|
188
|
+
MIN_SUBSTANTIVE_TEXT_CHARS = 12
|
|
189
|
+
HIGH_VOLUME_GARBAGE_CHARS = 64
|
|
190
|
+
HIGH_VOLUME_GARBAGE_CONFIDENCE = 35.0
|
|
191
|
+
TERMINAL_SYMBOL_GARBAGE_CHARS = 24
|
|
192
|
+
TERMINAL_SYMBOL_GARBAGE_ALNUM_RATIO = 0.20
|
|
183
193
|
ACTIVE_DEADLINE = None
|
|
184
194
|
|
|
185
195
|
|
|
@@ -207,11 +217,11 @@ class OcrDeadline:
|
|
|
207
217
|
if self.remaining_seconds() <= 0:
|
|
208
218
|
raise OcrPipelineTimeout(f"OCR deadline exceeded during {stage}")
|
|
209
219
|
|
|
210
|
-
def tesseract_timeout_seconds(self):
|
|
220
|
+
def tesseract_timeout_seconds(self, attempt_cap_seconds=MAX_TESSERACT_ATTEMPT_SECONDS):
|
|
211
221
|
self.check("Tesseract scheduling")
|
|
212
222
|
return max(
|
|
213
223
|
MIN_TESSERACT_ATTEMPT_SECONDS,
|
|
214
|
-
min(
|
|
224
|
+
min(attempt_cap_seconds, self.remaining_seconds()),
|
|
215
225
|
)
|
|
216
226
|
|
|
217
227
|
|
|
@@ -247,7 +257,8 @@ def text_from_tesseract_data(data):
|
|
|
247
257
|
return "\n".join(" ".join(words) for words in lines.values()).strip()
|
|
248
258
|
|
|
249
259
|
|
|
250
|
-
def run_tesseract(binary_img, deadline, language="eng", psm=6
|
|
260
|
+
def run_tesseract(binary_img, deadline, language="eng", psm=6,
|
|
261
|
+
attempt_cap_seconds=MAX_TESSERACT_ATTEMPT_SECONDS):
|
|
251
262
|
"""Run one bounded Tesseract TSV pass and derive text plus confidence."""
|
|
252
263
|
deadline.check("Tesseract")
|
|
253
264
|
pil_img = Image.fromarray(binary_img)
|
|
@@ -257,7 +268,7 @@ def run_tesseract(binary_img, deadline, language="eng", psm=6):
|
|
|
257
268
|
lang=language,
|
|
258
269
|
config=config,
|
|
259
270
|
output_type=pytesseract.Output.DICT,
|
|
260
|
-
timeout=deadline.tesseract_timeout_seconds(),
|
|
271
|
+
timeout=deadline.tesseract_timeout_seconds(attempt_cap_seconds),
|
|
261
272
|
)
|
|
262
273
|
text = text_from_tesseract_data(data)
|
|
263
274
|
confs = []
|
|
@@ -273,14 +284,87 @@ def run_tesseract(binary_img, deadline, language="eng", psm=6):
|
|
|
273
284
|
return text, avg_conf, line_count
|
|
274
285
|
|
|
275
286
|
|
|
276
|
-
def
|
|
287
|
+
def assess_ocr_evidence(text, confidence, line_count):
|
|
288
|
+
"""Classify OCR output before it can become agent-visible evidence."""
|
|
289
|
+
normalized = str(text or "").strip()
|
|
290
|
+
chars = len(normalized)
|
|
291
|
+
if chars == 0:
|
|
292
|
+
return {
|
|
293
|
+
"state": "low_information",
|
|
294
|
+
"accepted": False,
|
|
295
|
+
"reason": "no_readable_text",
|
|
296
|
+
"chars": 0,
|
|
297
|
+
"confidence": round(float(confidence), 1),
|
|
298
|
+
"lines": int(line_count),
|
|
299
|
+
}
|
|
300
|
+
alnum_ratio = sum(character.isalnum() for character in normalized) / max(1, chars)
|
|
301
|
+
if chars < MIN_SUBSTANTIVE_TEXT_CHARS and confidence < MIN_ACCEPTED_CONFIDENCE:
|
|
302
|
+
return {
|
|
303
|
+
"state": "low_information",
|
|
304
|
+
"accepted": False,
|
|
305
|
+
"reason": "insufficient_low_confidence_text",
|
|
306
|
+
"chars": chars,
|
|
307
|
+
"confidence": round(float(confidence), 1),
|
|
308
|
+
"lines": int(line_count),
|
|
309
|
+
"alnum_ratio": round(alnum_ratio, 3),
|
|
310
|
+
}
|
|
311
|
+
if (
|
|
312
|
+
(chars >= HIGH_VOLUME_GARBAGE_CHARS and confidence < HIGH_VOLUME_GARBAGE_CONFIDENCE)
|
|
313
|
+
or confidence < MIN_ACCEPTED_CONFIDENCE
|
|
314
|
+
or alnum_ratio < 0.45
|
|
315
|
+
):
|
|
316
|
+
reason = (
|
|
317
|
+
"high_volume_low_confidence_text"
|
|
318
|
+
if chars >= HIGH_VOLUME_GARBAGE_CHARS and confidence < HIGH_VOLUME_GARBAGE_CONFIDENCE
|
|
319
|
+
else "low_confidence_or_symbol_heavy_text"
|
|
320
|
+
)
|
|
321
|
+
return {
|
|
322
|
+
"state": "rejected",
|
|
323
|
+
"accepted": False,
|
|
324
|
+
"reason": reason,
|
|
325
|
+
"chars": chars,
|
|
326
|
+
"confidence": round(float(confidence), 1),
|
|
327
|
+
"lines": int(line_count),
|
|
328
|
+
"alnum_ratio": round(alnum_ratio, 3),
|
|
329
|
+
}
|
|
330
|
+
return {
|
|
331
|
+
"state": "accepted",
|
|
332
|
+
"accepted": True,
|
|
333
|
+
"reason": "confidence_and_text_quality_met",
|
|
334
|
+
"chars": chars,
|
|
335
|
+
"confidence": round(float(confidence), 1),
|
|
336
|
+
"lines": int(line_count),
|
|
337
|
+
"alnum_ratio": round(alnum_ratio, 3),
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def is_terminal_small_crop_rejection(evidence):
|
|
342
|
+
"""Whether a first small-crop pass proves another recovery pass is futile.
|
|
343
|
+
|
|
344
|
+
Blanks and plausible alphanumeric low-confidence text remain recoverable
|
|
345
|
+
through the second variant. A large very-low-confidence transcript or a
|
|
346
|
+
distinctly symbol-heavy stream cannot become safe evidence through another
|
|
347
|
+
PSM6 pass, and should not make callers wait for one.
|
|
348
|
+
"""
|
|
349
|
+
if evidence.get("state") != "rejected":
|
|
350
|
+
return False
|
|
351
|
+
if evidence.get("reason") == "high_volume_low_confidence_text":
|
|
352
|
+
return True
|
|
353
|
+
return (
|
|
354
|
+
evidence.get("chars", 0) >= TERMINAL_SYMBOL_GARBAGE_CHARS
|
|
355
|
+
and evidence.get("alnum_ratio", 1.0) < TERMINAL_SYMBOL_GARBAGE_ALNUM_RATIO
|
|
356
|
+
)
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def compute_score(text, confidence, line_count, evidence=None):
|
|
277
360
|
"""Combined scoring heuristic:
|
|
278
361
|
- confidence * sqrt(char_count) — rewards quality and coverage
|
|
279
362
|
- + line_count * 10 — bonus for structured output (more lines = better parse)
|
|
280
363
|
The agent discovered that line-count is a strong proxy for successful parsing
|
|
281
364
|
on structured documents like invoices and forms."""
|
|
365
|
+
quality = evidence or assess_ocr_evidence(text, confidence, line_count)
|
|
282
366
|
char_count = len(text)
|
|
283
|
-
if char_count == 0:
|
|
367
|
+
if not quality["accepted"] or char_count == 0:
|
|
284
368
|
return 0
|
|
285
369
|
return confidence * (char_count ** 0.5) + line_count * 10
|
|
286
370
|
|
|
@@ -319,7 +403,8 @@ def build_ocr_plan(image_area_px, single_psm=None):
|
|
|
319
403
|
|
|
320
404
|
|
|
321
405
|
def has_sufficient_evidence(text, confidence, line_count):
|
|
322
|
-
|
|
406
|
+
quality = assess_ocr_evidence(text, confidence, line_count)
|
|
407
|
+
return quality["accepted"] and confidence >= 70.0 and line_count >= 1
|
|
323
408
|
|
|
324
409
|
|
|
325
410
|
# ---------------------------------------------------------------------------
|
|
@@ -391,15 +476,18 @@ def write_all_outputs(text, base_name, output_dir):
|
|
|
391
476
|
# Main pipeline
|
|
392
477
|
# ---------------------------------------------------------------------------
|
|
393
478
|
|
|
394
|
-
def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefix="full"
|
|
479
|
+
def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefix="full",
|
|
480
|
+
attempt_cap_seconds=MAX_TESSERACT_ATTEMPT_SECONDS,
|
|
481
|
+
stop_on_terminal_rejection=False):
|
|
395
482
|
"""Run a bounded plan, stopping early once a legible result is proven."""
|
|
396
483
|
all_results = {}
|
|
397
484
|
ocr_errors = []
|
|
398
485
|
best_key = None
|
|
399
486
|
best_score = -1
|
|
400
487
|
early_exit = False
|
|
488
|
+
terminal_rejection = False
|
|
401
489
|
|
|
402
|
-
for vname, psm in plan:
|
|
490
|
+
for attempt_index, (vname, psm) in enumerate(plan):
|
|
403
491
|
deadline.check("preprocessing")
|
|
404
492
|
key = f"{vname}_psm{psm}"
|
|
405
493
|
try:
|
|
@@ -411,7 +499,9 @@ def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefi
|
|
|
411
499
|
os.makedirs(debug_dir, exist_ok=True)
|
|
412
500
|
cv2.imwrite(os.path.join(debug_dir, f"{debug_prefix}_{vname}.png"), binary)
|
|
413
501
|
try:
|
|
414
|
-
text, confidence, line_count = run_tesseract(
|
|
502
|
+
text, confidence, line_count = run_tesseract(
|
|
503
|
+
binary, deadline, language, psm, attempt_cap_seconds
|
|
504
|
+
)
|
|
415
505
|
except OcrPipelineTimeout:
|
|
416
506
|
raise
|
|
417
507
|
except OcrPipelineCancelled:
|
|
@@ -420,22 +510,31 @@ def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefi
|
|
|
420
510
|
ocr_errors.append(f"{key}: {error}")
|
|
421
511
|
continue
|
|
422
512
|
char_count = len(text)
|
|
423
|
-
|
|
513
|
+
evidence = assess_ocr_evidence(text, confidence, line_count)
|
|
514
|
+
score = compute_score(text, confidence, line_count, evidence)
|
|
424
515
|
all_results[key] = {
|
|
425
516
|
"text": text,
|
|
426
517
|
"chars": char_count,
|
|
427
518
|
"lines": line_count,
|
|
428
519
|
"confidence": round(confidence, 1),
|
|
429
520
|
"score": round(score, 1),
|
|
521
|
+
"evidence": evidence,
|
|
430
522
|
}
|
|
431
|
-
if score > best_score:
|
|
523
|
+
if evidence["accepted"] and score > best_score:
|
|
432
524
|
best_score = score
|
|
433
525
|
best_key = key
|
|
434
526
|
if has_sufficient_evidence(text, confidence, line_count):
|
|
435
527
|
early_exit = True
|
|
436
528
|
break
|
|
529
|
+
if (
|
|
530
|
+
stop_on_terminal_rejection
|
|
531
|
+
and attempt_index == 0
|
|
532
|
+
and is_terminal_small_crop_rejection(evidence)
|
|
533
|
+
):
|
|
534
|
+
terminal_rejection = True
|
|
535
|
+
break
|
|
437
536
|
|
|
438
|
-
return all_results, ocr_errors, best_key, early_exit
|
|
537
|
+
return all_results, ocr_errors, best_key, early_exit, terminal_rejection
|
|
439
538
|
|
|
440
539
|
|
|
441
540
|
def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_dir=None,
|
|
@@ -460,17 +559,89 @@ def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_d
|
|
|
460
559
|
deadline.check("image conditioning")
|
|
461
560
|
gray_2x = upscale_2x(gray)
|
|
462
561
|
plan = build_ocr_plan(effective_area, single_psm)
|
|
562
|
+
is_small_crop = effective_area <= SMALL_CROP_AREA_PX
|
|
563
|
+
attempt_cap_seconds = (
|
|
564
|
+
SMALL_CROP_TESSERACT_ATTEMPT_SECONDS
|
|
565
|
+
if is_small_crop
|
|
566
|
+
else MAX_TESSERACT_ATTEMPT_SECONDS
|
|
567
|
+
)
|
|
463
568
|
attempts_completed = 0
|
|
464
569
|
try:
|
|
465
|
-
all_results, ocr_errors, best_key, early_exit = run_variant_plan(
|
|
466
|
-
gray_2x,
|
|
570
|
+
all_results, ocr_errors, best_key, early_exit, terminal_rejection = run_variant_plan(
|
|
571
|
+
gray_2x,
|
|
572
|
+
plan,
|
|
573
|
+
deadline,
|
|
574
|
+
language,
|
|
575
|
+
debug_dir,
|
|
576
|
+
attempt_cap_seconds=attempt_cap_seconds,
|
|
577
|
+
stop_on_terminal_rejection=is_small_crop,
|
|
467
578
|
)
|
|
468
579
|
attempts_completed = len(all_results)
|
|
469
580
|
if not best_key:
|
|
470
|
-
|
|
471
|
-
|
|
581
|
+
if not all_results:
|
|
582
|
+
detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
|
|
583
|
+
return {"error": f"Tesseract failed for every bounded OCR attempt: {detail}"}
|
|
584
|
+
rejected = [item for item in all_results.values() if item["evidence"]["state"] == "rejected"]
|
|
585
|
+
if rejected:
|
|
586
|
+
worst = max(rejected, key=lambda item: (item["chars"], -item["confidence"]))
|
|
587
|
+
evidence = worst["evidence"]
|
|
588
|
+
message = (
|
|
589
|
+
"OCR text was suppressed because it did not meet evidence-quality requirements "
|
|
590
|
+
f"({evidence['reason']}; {evidence['chars']} chars at {evidence['confidence']}% confidence)."
|
|
591
|
+
)
|
|
592
|
+
return {
|
|
593
|
+
"error": message,
|
|
594
|
+
"diagnostic": {
|
|
595
|
+
**diagnostic("ocr_evidence_rejected", message, deadline, "evidence_quality", attempts_completed, len(plan)),
|
|
596
|
+
"evidence": evidence,
|
|
597
|
+
"terminal_small_crop_rejection": terminal_rejection,
|
|
598
|
+
"attempt_cap_seconds": attempt_cap_seconds,
|
|
599
|
+
},
|
|
600
|
+
}
|
|
601
|
+
# Empty or tiny non-substantive detections are valid observations:
|
|
602
|
+
# do not invent text and do not report them as a pipeline error.
|
|
603
|
+
result = {
|
|
604
|
+
"text": "",
|
|
605
|
+
"confidence": 0.0,
|
|
606
|
+
"variant": "none",
|
|
607
|
+
"chars": 0,
|
|
608
|
+
"lines": 0,
|
|
609
|
+
"score": 0.0,
|
|
610
|
+
"image_size": f"{w_orig}x{h_orig}",
|
|
611
|
+
"variants_tested": len(all_results),
|
|
612
|
+
"all_variants": {},
|
|
613
|
+
"quality": {
|
|
614
|
+
"schema": "omnius.ocr-evidence.v1",
|
|
615
|
+
"state": "low_information",
|
|
616
|
+
"accepted": False,
|
|
617
|
+
"reason": "no_accepted_readable_text",
|
|
618
|
+
"low_information_variants": len(all_results),
|
|
619
|
+
},
|
|
620
|
+
"diagnostic": diagnostic(
|
|
621
|
+
"ocr_low_information",
|
|
622
|
+
"OCR produced no accepted readable text; the result is low-information rather than evidence.",
|
|
623
|
+
deadline,
|
|
624
|
+
"evidence_quality",
|
|
625
|
+
attempts_completed,
|
|
626
|
+
len(plan),
|
|
627
|
+
),
|
|
628
|
+
"strategy": {
|
|
629
|
+
"effective_area_px": effective_area,
|
|
630
|
+
"attempts_planned": len(plan),
|
|
631
|
+
"attempts_completed": attempts_completed,
|
|
632
|
+
"early_exit": False,
|
|
633
|
+
"terminal_small_crop_rejection": terminal_rejection,
|
|
634
|
+
"attempt_cap_seconds": attempt_cap_seconds,
|
|
635
|
+
"deadline_ms": deadline.deadline_ms,
|
|
636
|
+
},
|
|
637
|
+
}
|
|
638
|
+
return result
|
|
472
639
|
|
|
473
640
|
best = all_results[best_key]
|
|
641
|
+
accepted_results = {
|
|
642
|
+
key: value for key, value in all_results.items()
|
|
643
|
+
if value["evidence"]["state"] == "accepted"
|
|
644
|
+
}
|
|
474
645
|
result = {
|
|
475
646
|
"text": best["text"],
|
|
476
647
|
"confidence": best["confidence"],
|
|
@@ -480,12 +651,25 @@ def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_d
|
|
|
480
651
|
"score": best["score"],
|
|
481
652
|
"image_size": f"{w_orig}x{h_orig}",
|
|
482
653
|
"variants_tested": len(all_results),
|
|
483
|
-
|
|
654
|
+
# Never leak rejected raw OCR as alternate evidence to agents.
|
|
655
|
+
"all_variants": accepted_results,
|
|
656
|
+
"quality": {
|
|
657
|
+
"schema": "omnius.ocr-evidence.v1",
|
|
658
|
+
**best["evidence"],
|
|
659
|
+
"rejected_variants": sum(
|
|
660
|
+
1 for item in all_results.values() if item["evidence"]["state"] == "rejected"
|
|
661
|
+
),
|
|
662
|
+
"low_information_variants": sum(
|
|
663
|
+
1 for item in all_results.values() if item["evidence"]["state"] == "low_information"
|
|
664
|
+
),
|
|
665
|
+
},
|
|
484
666
|
"strategy": {
|
|
485
667
|
"effective_area_px": effective_area,
|
|
486
668
|
"attempts_planned": len(plan),
|
|
487
669
|
"attempts_completed": attempts_completed,
|
|
488
670
|
"early_exit": early_exit,
|
|
671
|
+
"terminal_small_crop_rejection": terminal_rejection,
|
|
672
|
+
"attempt_cap_seconds": attempt_cap_seconds,
|
|
489
673
|
"deadline_ms": deadline.deadline_ms,
|
|
490
674
|
},
|
|
491
675
|
}
|
|
@@ -498,9 +682,22 @@ def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_d
|
|
|
498
682
|
region_gray = extract_region(gray_2x, y_start, y_end)
|
|
499
683
|
# Region requests use at most two high-yield attempts. The
|
|
500
684
|
# main result already provides full-frame coverage.
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
685
|
+
region_area = region_gray.shape[0] * region_gray.shape[1]
|
|
686
|
+
region_plan = build_ocr_plan(region_area, single_psm)[:2]
|
|
687
|
+
region_is_small = region_area <= SMALL_CROP_AREA_PX
|
|
688
|
+
region_results, _errors, region_best_key, _early_exit, _terminal_rejection = run_variant_plan(
|
|
689
|
+
region_gray,
|
|
690
|
+
region_plan,
|
|
691
|
+
deadline,
|
|
692
|
+
language,
|
|
693
|
+
debug_dir,
|
|
694
|
+
f"region_{rname}",
|
|
695
|
+
attempt_cap_seconds=(
|
|
696
|
+
SMALL_CROP_TESSERACT_ATTEMPT_SECONDS
|
|
697
|
+
if region_is_small
|
|
698
|
+
else MAX_TESSERACT_ATTEMPT_SECONDS
|
|
699
|
+
),
|
|
700
|
+
stop_on_terminal_rejection=region_is_small,
|
|
504
701
|
)
|
|
505
702
|
regions[rname] = region_results[region_best_key]["text"] if region_best_key else ""
|
|
506
703
|
result["regions"] = regions
|
package/docs/DISCOVERY.json
CHANGED
|
@@ -5399,7 +5399,7 @@
|
|
|
5399
5399
|
"tags": [
|
|
5400
5400
|
"Audio"
|
|
5401
5401
|
],
|
|
5402
|
-
"description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses the pinned JetPack YAMNet/TensorRT setup; speaker creates a private CPython 3.10 CPU-only WeSpeaker CAM++ venv, installs only checksum-pinned NumPy, ONNX Runtime, and its locked CPU support wheels with --no-deps/--no-index, then downloads the pinned model. Its 80-bin Kaldi-compatible NumPy fbank is validated with a
|
|
5402
|
+
"description": "Admin-only. Requires kind=acoustic|speaker|semantic in the query (canonical) or JSON body. acoustic reuses the pinned JetPack YAMNet/TensorRT setup; speaker creates a private CPython 3.10 CPU-only WeSpeaker CAM++ venv, installs only checksum-pinned NumPy, ONNX Runtime, and its locked CPU support wheels with --no-deps/--no-index, then downloads the pinned model. Its 80-bin Kaldi-compatible NumPy fbank is validated with a versioned cross-platform numeric-invariant probe before readiness. The speaker path never imports, links, replaces, or otherwise depends on JetPack Torch/Torchaudio, so an incompatible generic Torchaudio wheel cannot affect CUDA-enabled Egg Torch. semantic installs the isolated JetPack CUDA CLAP dependencies and pinned model. This is the only REST operation allowed to provision. It activates and warms only the requested role worker; no role is substituted for another.",
|
|
5403
5403
|
"parameters": [
|
|
5404
5404
|
{
|
|
5405
5405
|
"name": "kind",
|
|
@@ -6156,6 +6156,18 @@
|
|
|
6156
6156
|
"default": 45,
|
|
6157
6157
|
"description": "Total server-side agent-loop deadline. Distinct from per-backend timeout_s."
|
|
6158
6158
|
},
|
|
6159
|
+
"agent_max_tool_rounds": {
|
|
6160
|
+
"type": "integer",
|
|
6161
|
+
"minimum": 1,
|
|
6162
|
+
"maximum": 8,
|
|
6163
|
+
"default": 1,
|
|
6164
|
+
"description": "Maximum daemon-tool planning rounds. The default executes one tool round, then removes daemon schemas for lower-latency final synthesis."
|
|
6165
|
+
},
|
|
6166
|
+
"agent_prefetch_web_search": {
|
|
6167
|
+
"type": "boolean",
|
|
6168
|
+
"default": false,
|
|
6169
|
+
"description": "Explicitly execute authorized web_search with the latest user text before one backend synthesis. factual-first enables this automatically."
|
|
6170
|
+
},
|
|
6159
6171
|
"max_turns": {
|
|
6160
6172
|
"type": "integer",
|
|
6161
6173
|
"description": "Q2 — agent_loop max iterations (default 8, max 64)."
|
|
@@ -6165,7 +6177,7 @@
|
|
|
6165
6177
|
"enum": [
|
|
6166
6178
|
"factual-first"
|
|
6167
6179
|
],
|
|
6168
|
-
"description": "
|
|
6180
|
+
"description": "Factual-first prefetches authorized web_search from the latest user turn and performs one grounded synthesis generation."
|
|
6169
6181
|
}
|
|
6170
6182
|
}
|
|
6171
6183
|
}
|
|
@@ -46,6 +46,8 @@ Important body fields:
|
|
|
46
46
|
| `include_daemon_tools` | array | Permit the bounded core daemon-tool catalog by scope: `read`, `run`, `admin` |
|
|
47
47
|
| `daemon_tool_names` | array | Exact daemon-tool allowlist; recommended for local models |
|
|
48
48
|
| `agent_timeout_s` | number | Whole-loop deadline, default 45 seconds and maximum 600 |
|
|
49
|
+
| `agent_max_tool_rounds` | integer | Daemon tool rounds before forced final synthesis; default 1 |
|
|
50
|
+
| `agent_prefetch_web_search` | boolean | Explicit one-generation web-search prefetch; factual-first enables it automatically |
|
|
49
51
|
| `max_turns` | integer | Server-side agent loop turn cap |
|
|
50
52
|
| `prompt_template` | string | Optional template such as `factual-first` |
|
|
51
53
|
|
|
@@ -129,4 +131,4 @@ For ASR/TTS systems that only need the text brain, use `/realtime` or `/v1/realt
|
|
|
129
131
|
|
|
130
132
|
`/v1/chat/completions` can run an internal tool loop when `agent_loop: true`. This lets clients collapse multiple model/tool round trips into one daemon request. Daemon tool calls execute inline; client-owned tool calls can still be yielded in OpenAI-compatible shape.
|
|
131
133
|
|
|
132
|
-
Ollama-backed loops use its native `/api/chat` tool protocol. `timeout_s` applies to each backend round, while `agent_timeout_s` bounds the complete loop and defaults to 45 seconds.
|
|
134
|
+
Ollama-backed loops use its native `/api/chat` tool protocol. `timeout_s` applies to each backend round, while `agent_timeout_s` bounds the complete loop and defaults to 45 seconds. For a normal loop, the planning turn is capped at 96 output tokens; one daemon-tool round is the default, after which Omnius removes daemon schemas for final synthesis. `prompt_template: "factual-first"` skips that model-planning round entirely: Omnius executes the already-mandated, authorized `web_search` using the latest user turn, inserts the protocol-correct tool evidence, and performs one grounded synthesis generation. `agent_prefetch_web_search: true` opts into the same path directly. Generic/deeper tool workflows remain available through `agent_max_tool_rounds`. Omnius returns typed HTTP 504/508 failures, caps tool evidence at 6,000 characters, and limits the implicit catalog unless `daemon_tool_names` requests exact additions.
|
|
@@ -128,13 +128,23 @@ needs `run` scope because optional
|
|
|
128
128
|
|
|
129
129
|
The managed pipeline is bounded: it starts with high-yield preprocessing and
|
|
130
130
|
expands variants only for low-evidence or larger images. Small crops avoid the
|
|
131
|
-
former all-variant/all-PSM explosion
|
|
132
|
-
|
|
133
|
-
|
|
131
|
+
former all-variant/all-PSM explosion and use a six-second cap per Tesseract
|
|
132
|
+
attempt. If the first small-crop pass proves high-volume very-low-confidence or
|
|
133
|
+
distinctly symbol-heavy garbage, recovery stops immediately; blanks and
|
|
134
|
+
plausibly recoverable text still receive the second high-yield attempt. The
|
|
135
|
+
REST default is 90 seconds (maximum 180 seconds), while the worker has an
|
|
136
|
+
80-second internal deadline so it can return diagnostics. A timeout or cancellation returns
|
|
134
137
|
`result.data.schema=omnius.ocr-diagnostic.v1` with code `ocr_timeout` or
|
|
135
138
|
`ocr_cancelled`; cancellation terminates the Python/Tesseract process group
|
|
136
139
|
with TERM followed by KILL.
|
|
137
140
|
|
|
141
|
+
OCR text is evidence-gated before it is returned. Strong text with adequate
|
|
142
|
+
confidence is accepted. An empty or tiny non-substantive result is a successful
|
|
143
|
+
`omnius.ocr-evidence.v1` `low_information` observation with diagnostic code
|
|
144
|
+
`ocr_low_information`, not a fabricated transcript. High-volume low-confidence
|
|
145
|
+
or symbol-heavy output is suppressed and returned as
|
|
146
|
+
`ocr_evidence_rejected`; its raw text is not exposed as alternate OCR evidence.
|
|
147
|
+
|
|
138
148
|
## TTS
|
|
139
149
|
|
|
140
150
|
`POST /v1/voice/tts` returns audio bytes. `format` can be `wav` or `pcm`. `X-Sample-Rate` reports the sample rate.
|
|
@@ -270,8 +280,9 @@ managed venv installs it with `--no-deps --no-index`.
|
|
|
270
280
|
The worker implements the WeSpeaker CAM++ 80-bin Kaldi configuration in pure
|
|
271
281
|
NumPy: 25 ms / 10 ms Hamming frames, dither disabled, Kaldi pre-emphasis and
|
|
272
282
|
mel bank behavior, then full-clip CMN without CVN. Readiness invokes a
|
|
273
|
-
|
|
274
|
-
|
|
283
|
+
versioned CPU preprocessing probe. It checks the fixed Kaldi configuration,
|
|
284
|
+
feature shape, CMN, energy range, and expected mel-band profile with explicit
|
|
285
|
+
cross-platform numeric tolerances before it can report ready. The speaker path neither imports
|
|
275
286
|
nor links Torch or Torchaudio, so the generic `torchaudio-2.2.0` ABI mismatch
|
|
276
287
|
cannot bind against, replace, or otherwise affect JetPack's CUDA-enabled Egg
|
|
277
288
|
Torch. Inference remains install-free and network-free; failed package imports
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "omnius",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.642",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "omnius",
|
|
9
|
-
"version": "1.0.
|
|
9
|
+
"version": "1.0.642",
|
|
10
10
|
"bundleDependencies": [
|
|
11
11
|
"image-to-ascii"
|
|
12
12
|
],
|
package/package.json
CHANGED