omnius 1.0.638 → 1.0.640

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -37,6 +37,9 @@ import os
37
37
  import json
38
38
  import csv
39
39
  import argparse
40
+ import signal
41
+ import time
42
+ from collections import OrderedDict
40
43
  from pathlib import Path
41
44
 
42
45
  def check_deps():
@@ -168,34 +171,105 @@ PSM_MODES = {
168
171
  11: "sparse",
169
172
  }
170
173
 
174
+ # Advanced OCR is a bounded recovery pipeline, not a blind Cartesian product.
175
+ # A 327x333 crop previously paid for 24 variants/PSMs and two Tesseract child
176
+ # processes per attempt. Small inputs now begin with two high-yield attempts;
177
+ # larger/low-evidence images expand only while budget remains.
178
+ SMALL_CROP_AREA_PX = 512_000
179
+ MEDIUM_IMAGE_AREA_PX = 2_000_000
180
+ MAX_PIPELINE_DEADLINE_MS = 80_000
181
+ MAX_TESSERACT_ATTEMPT_SECONDS = 12.0
182
+ MIN_TESSERACT_ATTEMPT_SECONDS = 0.25
183
+ ACTIVE_DEADLINE = None
171
184
 
172
- # ---------------------------------------------------------------------------
173
- # OCR execution
174
- # ---------------------------------------------------------------------------
175
185
 
176
- def run_tesseract(binary_img, language="eng", psm=6):
177
- """Run Tesseract on a preprocessed binary image.
178
- Returns (text, confidence, line_count)."""
179
- pil_img = Image.fromarray(binary_img)
180
- config = f"--psm {psm}"
186
+ class OcrPipelineTimeout(RuntimeError):
187
+ pass
181
188
 
182
- # Engine failures are not blank documents. Let callers distinguish a real
183
- # low-information image from a missing language pack or broken Tesseract.
184
- text = pytesseract.image_to_string(pil_img, lang=language, config=config).strip()
185
189
 
186
- line_count = len([l for l in text.split("\n") if l.strip()])
190
+ class OcrPipelineCancelled(RuntimeError):
191
+ pass
187
192
 
188
- # Get confidence via image_to_data
189
- try:
190
- data = pytesseract.image_to_data(
191
- pil_img, lang=language, config=config,
192
- output_type=pytesseract.Output.DICT,
193
+
194
+ def cancellation_signal_handler(signum, _frame):
195
+ raise OcrPipelineCancelled(f"received signal {signum}")
196
+
197
+
198
+ class OcrDeadline:
199
+ def __init__(self, deadline_ms):
200
+ self.deadline_ms = max(1_000, min(int(deadline_ms), MAX_PIPELINE_DEADLINE_MS))
201
+ self.started = time.monotonic()
202
+
203
+ def remaining_seconds(self):
204
+ return self.deadline_ms / 1000.0 - (time.monotonic() - self.started)
205
+
206
+ def check(self, stage):
207
+ if self.remaining_seconds() <= 0:
208
+ raise OcrPipelineTimeout(f"OCR deadline exceeded during {stage}")
209
+
210
+ def tesseract_timeout_seconds(self):
211
+ self.check("Tesseract scheduling")
212
+ return max(
213
+ MIN_TESSERACT_ATTEMPT_SECONDS,
214
+ min(MAX_TESSERACT_ATTEMPT_SECONDS, self.remaining_seconds()),
215
+ )
216
+
217
+
218
+ def diagnostic(code, message, deadline, stage, attempts_completed=0, attempts_planned=0):
219
+ return {
220
+ "schema": "omnius.ocr-diagnostic.v1",
221
+ "code": code,
222
+ "message": message,
223
+ "stage": stage,
224
+ "deadline_ms": deadline.deadline_ms,
225
+ "attempts_completed": attempts_completed,
226
+ "attempts_planned": attempts_planned,
227
+ }
228
+
229
+
230
+ # ---------------------------------------------------------------------------
231
+ # OCR execution
232
+ # ---------------------------------------------------------------------------
233
+
234
+ def text_from_tesseract_data(data):
235
+ """Reconstruct line breaks from one TSV pass; do not spawn Tesseract twice."""
236
+ lines = OrderedDict()
237
+ texts = data.get("text", [])
238
+ for index, raw_text in enumerate(texts):
239
+ text = str(raw_text or "").strip()
240
+ if not text:
241
+ continue
242
+ key = tuple(
243
+ int(data.get(field, [0] * len(texts))[index] or 0)
244
+ for field in ("block_num", "par_num", "line_num")
193
245
  )
194
- confs = [int(c) for c in data["conf"] if int(c) >= 0]
195
- avg_conf = sum(confs) / len(confs) if confs else 0.0
196
- except Exception:
197
- avg_conf = 0.0
246
+ lines.setdefault(key, []).append(text)
247
+ return "\n".join(" ".join(words) for words in lines.values()).strip()
248
+
198
249
 
250
+ def run_tesseract(binary_img, deadline, language="eng", psm=6):
251
+ """Run one bounded Tesseract TSV pass and derive text plus confidence."""
252
+ deadline.check("Tesseract")
253
+ pil_img = Image.fromarray(binary_img)
254
+ config = f"--psm {psm}"
255
+ data = pytesseract.image_to_data(
256
+ pil_img,
257
+ lang=language,
258
+ config=config,
259
+ output_type=pytesseract.Output.DICT,
260
+ timeout=deadline.tesseract_timeout_seconds(),
261
+ )
262
+ text = text_from_tesseract_data(data)
263
+ confs = []
264
+ for value in data.get("conf", []):
265
+ try:
266
+ confidence = float(value)
267
+ except (TypeError, ValueError):
268
+ continue
269
+ if confidence >= 0:
270
+ confs.append(confidence)
271
+ avg_conf = sum(confs) / len(confs) if confs else 0.0
272
+ line_count = len([line for line in text.split("\n") if line.strip()])
199
273
  return text, avg_conf, line_count
200
274
 
201
275
 
@@ -226,6 +300,28 @@ def extract_pixel_region(gray, x, y, w, h):
226
300
  return gray[y:y+h, x:x+w]
227
301
 
228
302
 
303
+ def build_ocr_plan(image_area_px, single_psm=None):
304
+ """Return the smallest credible variant/PSM plan for the effective image."""
305
+ psm_modes = [single_psm] if single_psm else (
306
+ [6] if image_area_px <= SMALL_CROP_AREA_PX
307
+ else [6, 11] if image_area_px <= MEDIUM_IMAGE_AREA_PX
308
+ else [6, 11, 4]
309
+ )
310
+ variants = (
311
+ ["otsu", "adaptive_fine"] if image_area_px <= SMALL_CROP_AREA_PX
312
+ else ["otsu", "adaptive_fine", "denoise", "sharpen_unsharp"]
313
+ if image_area_px <= MEDIUM_IMAGE_AREA_PX
314
+ else list(ALL_VARIANTS.keys())
315
+ )
316
+ # OTSU/PSM6 is deliberately first: a legible result ends the search
317
+ # instead of paying for redundant variants that cannot improve the answer.
318
+ return [(variant, psm) for psm in psm_modes for variant in variants]
319
+
320
+
321
+ def has_sufficient_evidence(text, confidence, line_count):
322
+ return len(text.strip()) >= 8 and confidence >= 70.0 and line_count >= 1
323
+
324
+
229
325
  # ---------------------------------------------------------------------------
230
326
  # Output writers
231
327
  # ---------------------------------------------------------------------------
@@ -295,142 +391,139 @@ def write_all_outputs(text, base_name, output_dir):
295
391
  # Main pipeline
296
392
  # ---------------------------------------------------------------------------
297
393
 
298
- def run_pipeline(image_path, language="eng", do_regions=False, debug_dir=None,
299
- single_psm=None, pixel_region=None, output_dir=None):
300
- """Run the full multi-variant, multi-PSM OCR pipeline."""
301
-
302
- # Load image
303
- img = cv2.imread(image_path)
304
- if img is None:
305
- return {"error": f"Could not load image: {image_path}"}
306
-
307
- h_orig, w_orig = img.shape[:2]
308
- gray = to_grayscale(img)
309
-
310
- # Upscale 2x
311
- gray_2x = upscale_2x(gray)
312
-
313
- # If a pixel region is specified, crop before processing
314
- if pixel_region:
315
- rx, ry, rw, rh = pixel_region
316
- # Scale region coords to match 2x upscale
317
- gray_2x = extract_pixel_region(gray_2x, rx * 2, ry * 2, rw * 2, rh * 2)
318
-
319
- # Determine PSM modes to test
320
- psm_modes = [single_psm] if single_psm else [4, 6, 11]
321
-
322
- # Generate all variants and run OCR
394
+ def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefix="full"):
395
+ """Run a bounded plan, stopping early once a legible result is proven."""
323
396
  all_results = {}
324
397
  ocr_errors = []
325
398
  best_key = None
326
399
  best_score = -1
400
+ early_exit = False
327
401
 
328
- for vname, vfunc in ALL_VARIANTS.items():
402
+ for vname, psm in plan:
403
+ deadline.check("preprocessing")
404
+ key = f"{vname}_psm{psm}"
329
405
  try:
330
- binary = vfunc(gray_2x)
331
- except Exception:
406
+ binary = ALL_VARIANTS[vname](gray)
407
+ except Exception as error:
408
+ ocr_errors.append(f"{key}: preprocessing failed: {error}")
332
409
  continue
333
-
334
- # Save debug images
335
410
  if debug_dir:
336
411
  os.makedirs(debug_dir, exist_ok=True)
337
- cv2.imwrite(os.path.join(debug_dir, f"full_{vname}.png"), binary)
338
-
339
- for psm in psm_modes:
340
- key = f"{vname}_psm{psm}"
341
- try:
342
- text, confidence, line_count = run_tesseract(binary, language, psm)
343
- except Exception as error:
344
- ocr_errors.append(f"{key}: {error}")
345
- continue
346
- char_count = len(text)
347
- score = compute_score(text, confidence, line_count)
348
-
349
- all_results[key] = {
350
- "text": text,
351
- "chars": char_count,
352
- "lines": line_count,
353
- "confidence": round(confidence, 1),
354
- "score": round(score, 1),
355
- }
356
-
357
- if score > best_score:
358
- best_score = score
359
- best_key = key
360
-
361
- if not best_key:
362
- detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
363
- return {"error": f"Tesseract failed for every OCR variant: {detail}"}
364
-
365
- best = all_results[best_key]
366
- result = {
367
- "text": best["text"],
368
- "confidence": best["confidence"],
369
- "variant": best_key,
370
- "chars": best["chars"],
371
- "lines": best["lines"],
372
- "score": best["score"],
373
- "image_size": f"{w_orig}x{h_orig}",
374
- "variants_tested": len(all_results),
375
- "all_variants": all_results,
376
- }
377
-
378
- # Region-based OCR
379
- if do_regions:
380
- regions = {}
381
- region_defs = {
382
- "header": (0, 35),
383
- "body": (30, 80),
384
- "footer": (75, 100),
412
+ cv2.imwrite(os.path.join(debug_dir, f"{debug_prefix}_{vname}.png"), binary)
413
+ try:
414
+ text, confidence, line_count = run_tesseract(binary, deadline, language, psm)
415
+ except OcrPipelineTimeout:
416
+ raise
417
+ except OcrPipelineCancelled:
418
+ raise
419
+ except Exception as error:
420
+ ocr_errors.append(f"{key}: {error}")
421
+ continue
422
+ char_count = len(text)
423
+ score = compute_score(text, confidence, line_count)
424
+ all_results[key] = {
425
+ "text": text,
426
+ "chars": char_count,
427
+ "lines": line_count,
428
+ "confidence": round(confidence, 1),
429
+ "score": round(score, 1),
385
430
  }
431
+ if score > best_score:
432
+ best_score = score
433
+ best_key = key
434
+ if has_sufficient_evidence(text, confidence, line_count):
435
+ early_exit = True
436
+ break
386
437
 
387
- for rname, (y_start, y_end) in region_defs.items():
388
- region_gray = extract_region(gray_2x, y_start, y_end)
389
-
390
- if debug_dir:
391
- cv2.imwrite(os.path.join(debug_dir, f"region_{rname}.png"), region_gray)
392
-
393
- # Test all variants on each region for best accuracy
394
- region_best = ""
395
- region_best_score = -1
396
-
397
- for vname in ["otsu", "denoise", "adaptive_fine", "sharpen_unsharp"]:
398
- if vname not in ALL_VARIANTS:
399
- continue
400
- try:
401
- binary = ALL_VARIANTS[vname](region_gray)
402
- except Exception:
403
- continue
438
+ return all_results, ocr_errors, best_key, early_exit
404
439
 
405
- if debug_dir:
406
- cv2.imwrite(os.path.join(debug_dir, f"region_{rname}_{vname}.png"), binary)
407
440
 
408
- try:
409
- text, conf, lc = run_tesseract(binary, language, 6)
410
- except Exception:
411
- continue
412
- score = compute_score(text, conf, lc)
413
- if score > region_best_score:
414
- region_best_score = score
415
- region_best = text
416
-
417
- regions[rname] = region_best
418
-
419
- result["regions"] = regions
441
+ def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_dir=None,
442
+ single_psm=None, pixel_region=None, output_dir=None):
443
+ """Run bounded OCR. The effective crop controls plan size, not the full frame."""
444
+ deadline.check("image load")
445
+ img = cv2.imread(image_path)
446
+ if img is None:
447
+ return {"error": f"Could not load image: {image_path}"}
420
448
 
421
- if debug_dir:
422
- result["debug_dir"] = debug_dir
449
+ h_orig, w_orig = img.shape[:2]
450
+ gray = to_grayscale(img)
451
+ if pixel_region:
452
+ rx, ry, rw, rh = pixel_region
453
+ if rx >= w_orig or ry >= h_orig:
454
+ return {"error": "Requested OCR region lies outside the image"}
455
+ gray = extract_pixel_region(gray, rx, ry, rw, rh)
456
+ if gray.size == 0:
457
+ return {"error": "Requested OCR region is empty"}
458
+
459
+ effective_area = int(gray.shape[0] * gray.shape[1])
460
+ deadline.check("image conditioning")
461
+ gray_2x = upscale_2x(gray)
462
+ plan = build_ocr_plan(effective_area, single_psm)
463
+ attempts_completed = 0
464
+ try:
465
+ all_results, ocr_errors, best_key, early_exit = run_variant_plan(
466
+ gray_2x, plan, deadline, language, debug_dir
467
+ )
468
+ attempts_completed = len(all_results)
469
+ if not best_key:
470
+ detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
471
+ return {"error": f"Tesseract failed for every bounded OCR attempt: {detail}"}
472
+
473
+ best = all_results[best_key]
474
+ result = {
475
+ "text": best["text"],
476
+ "confidence": best["confidence"],
477
+ "variant": best_key,
478
+ "chars": best["chars"],
479
+ "lines": best["lines"],
480
+ "score": best["score"],
481
+ "image_size": f"{w_orig}x{h_orig}",
482
+ "variants_tested": len(all_results),
483
+ "all_variants": all_results,
484
+ "strategy": {
485
+ "effective_area_px": effective_area,
486
+ "attempts_planned": len(plan),
487
+ "attempts_completed": attempts_completed,
488
+ "early_exit": early_exit,
489
+ "deadline_ms": deadline.deadline_ms,
490
+ },
491
+ }
423
492
 
424
- # Write output files if output_dir specified
425
- if output_dir:
426
- base_name = Path(image_path).stem
427
- files = write_all_outputs(best["text"], base_name, output_dir)
428
- result["output_files"] = files
493
+ if do_regions:
494
+ regions = {}
495
+ region_defs = {"header": (0, 35), "body": (30, 80), "footer": (75, 100)}
496
+ for rname, (y_start, y_end) in region_defs.items():
497
+ deadline.check(f"region {rname}")
498
+ region_gray = extract_region(gray_2x, y_start, y_end)
499
+ # Region requests use at most two high-yield attempts. The
500
+ # main result already provides full-frame coverage.
501
+ region_plan = build_ocr_plan(region_gray.shape[0] * region_gray.shape[1], single_psm)[:2]
502
+ region_results, _errors, region_best_key, _early_exit = run_variant_plan(
503
+ region_gray, region_plan, deadline, language, debug_dir, f"region_{rname}"
504
+ )
505
+ regions[rname] = region_results[region_best_key]["text"] if region_best_key else ""
506
+ result["regions"] = regions
429
507
 
430
- return result
508
+ if debug_dir:
509
+ result["debug_dir"] = debug_dir
510
+ if output_dir:
511
+ base_name = Path(image_path).stem
512
+ result["output_files"] = write_all_outputs(best["text"], base_name, output_dir)
513
+ return result
514
+ except OcrPipelineTimeout as error:
515
+ return {
516
+ "error": str(error),
517
+ "diagnostic": diagnostic("ocr_timeout", str(error), deadline, "pipeline", attempts_completed, len(plan)),
518
+ }
519
+ except OcrPipelineCancelled as error:
520
+ return {
521
+ "error": "Advanced OCR cancelled",
522
+ "diagnostic": diagnostic("ocr_cancelled", str(error), deadline, "pipeline", attempts_completed, len(plan)),
523
+ }
431
524
 
432
525
 
433
- def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
526
+ def run_batch(images_dir, deadline, language="eng", do_regions=False, debug_dir=None,
434
527
  output_dir=None):
435
528
  """Process all images in a directory."""
436
529
  images_dir = os.path.abspath(images_dir)
@@ -450,10 +543,18 @@ def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
450
543
  return {"error": f"No image files found in {images_dir}"}
451
544
 
452
545
  for img_file in image_files:
546
+ try:
547
+ deadline.check("batch scheduling")
548
+ except OcrPipelineTimeout as error:
549
+ return {
550
+ "error": str(error),
551
+ "diagnostic": diagnostic("ocr_timeout", str(error), deadline, "batch", len(batch_results), len(image_files)),
552
+ }
453
553
  img_path = os.path.join(images_dir, img_file)
454
554
  img_debug = os.path.join(debug_dir, Path(img_file).stem) if debug_dir else None
455
555
  result = run_pipeline(
456
556
  img_path,
557
+ deadline,
457
558
  language=language,
458
559
  do_regions=do_regions,
459
560
  debug_dir=img_debug,
@@ -469,6 +570,14 @@ def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
469
570
  "output_files": result.get("output_files"),
470
571
  "error": result.get("error"),
471
572
  }
573
+ if result.get("diagnostic", {}).get("code") in {"ocr_timeout", "ocr_cancelled"}:
574
+ return {
575
+ "error": result["error"],
576
+ "diagnostic": result["diagnostic"],
577
+ "batch": True,
578
+ "images_processed": len(batch_results),
579
+ "results": batch_results,
580
+ }
472
581
 
473
582
  # Write summary
474
583
  summary_path = os.path.join(out_dir, "OCR_PROCESSING_SUMMARY.md")
@@ -497,6 +606,7 @@ def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
497
606
 
498
607
 
499
608
  def main():
609
+ global ACTIVE_DEADLINE
500
610
  parser = argparse.ArgumentParser(
501
611
  description="Advanced multi-variant OCR pipeline for omnius"
502
612
  )
@@ -520,13 +630,20 @@ def main():
520
630
  help="Write TXT + CSV + PDF outputs to this directory")
521
631
  parser.add_argument("--batch", action="store_true",
522
632
  help="Process all images in a directory")
633
+ parser.add_argument("--deadline-ms", type=int, default=MAX_PIPELINE_DEADLINE_MS,
634
+ help="Bound total OCR work; clamped to the managed maximum")
523
635
 
524
636
  args = parser.parse_args()
637
+ deadline = OcrDeadline(args.deadline_ms)
638
+ ACTIVE_DEADLINE = deadline
639
+ signal.signal(signal.SIGTERM, cancellation_signal_handler)
640
+ signal.signal(signal.SIGINT, cancellation_signal_handler)
525
641
 
526
642
  # Batch mode
527
643
  if args.batch or os.path.isdir(args.image):
528
644
  result = run_batch(
529
645
  args.image,
646
+ deadline,
530
647
  language=args.language,
531
648
  do_regions=args.regions,
532
649
  debug_dir=args.debug_dir,
@@ -558,6 +675,7 @@ def main():
558
675
 
559
676
  result = run_pipeline(
560
677
  args.image,
678
+ deadline,
561
679
  language=args.language,
562
680
  do_regions=args.regions,
563
681
  debug_dir=args.debug_dir,
@@ -578,4 +696,19 @@ def main():
578
696
 
579
697
 
580
698
  if __name__ == "__main__":
581
- main()
699
+ try:
700
+ main()
701
+ except OcrPipelineTimeout as error:
702
+ deadline = ACTIVE_DEADLINE or OcrDeadline(MAX_PIPELINE_DEADLINE_MS)
703
+ print(json.dumps({
704
+ "error": str(error),
705
+ "diagnostic": diagnostic("ocr_timeout", str(error), deadline, "entrypoint"),
706
+ }))
707
+ sys.exit(124)
708
+ except OcrPipelineCancelled as error:
709
+ deadline = ACTIVE_DEADLINE or OcrDeadline(MAX_PIPELINE_DEADLINE_MS)
710
+ print(json.dumps({
711
+ "error": "Advanced OCR cancelled",
712
+ "diagnostic": diagnostic("ocr_cancelled", str(error), deadline, "entrypoint"),
713
+ }))
714
+ sys.exit(130)
@@ -8,16 +8,65 @@ stdout: JSON {"text": "...", "duration": null, "segments": []}
8
8
 
9
9
  ASR is GPU-first: CPU fallback is refused unless OMNIUS_ASR_ALLOW_CPU=1.
10
10
  """
11
- import sys, os, json, warnings
11
+ import sys, os, json, math, struct, warnings, wave
12
12
 
13
13
  warnings.filterwarnings("ignore")
14
14
  os.environ.setdefault("PYTHONWARNINGS", "ignore")
15
15
 
16
- try:
17
- import whisper
18
- except Exception as e:
19
- print(json.dumps({"error": "deps_missing", "message": str(e)}))
20
- sys.exit(0)
16
+ PCM16_SILENCE_MAX_PEAK = 4
17
+ PCM16_SILENCE_MAX_RMS = 1.0
18
+ PCM16_ACTIVE_SAMPLE_THRESHOLD = 8
19
+
20
+
21
+ def exact_pcm16_silence(path):
22
+ """Return typed no-speech evidence for exact PCM16/16k/mono WAV only.
23
+
24
+ This is intentionally not VAD. It reads user bytes but never rewrites,
25
+ normalizes, decodes another media format, or instantiates Whisper.
26
+ """
27
+ try:
28
+ with wave.open(path, "rb") as wav:
29
+ if (
30
+ wav.getcomptype() != "NONE"
31
+ or wav.getnchannels() != 1
32
+ or wav.getsampwidth() != 2
33
+ or wav.getframerate() != 16000
34
+ ):
35
+ return None
36
+ frames = wav.readframes(wav.getnframes())
37
+ except (wave.Error, OSError, EOFError):
38
+ return None
39
+ sample_count = len(frames) // 2
40
+ if sample_count == 0:
41
+ metrics = {
42
+ "sampleRateHz": 16000, "channels": 1, "bitsPerSample": 16,
43
+ "sampleCount": 0, "durationMs": 0.0, "peakPcm16": 0,
44
+ "rmsPcm16": 0.0, "activeSampleCount": 0, "activeSampleRatio": 0.0,
45
+ }
46
+ else:
47
+ samples = struct.iter_unpack("<h", frames[: sample_count * 2])
48
+ total_squares = 0
49
+ peak = 0
50
+ active = 0
51
+ for (sample,) in samples:
52
+ magnitude = abs(sample)
53
+ total_squares += sample * sample
54
+ peak = max(peak, magnitude)
55
+ if magnitude >= PCM16_ACTIVE_SAMPLE_THRESHOLD:
56
+ active += 1
57
+ metrics = {
58
+ "sampleRateHz": 16000, "channels": 1, "bitsPerSample": 16,
59
+ "sampleCount": sample_count, "durationMs": sample_count / 16.0,
60
+ "peakPcm16": peak, "rmsPcm16": math.sqrt(total_squares / sample_count),
61
+ "activeSampleCount": active, "activeSampleRatio": active / sample_count,
62
+ }
63
+ if (
64
+ metrics["peakPcm16"] <= PCM16_SILENCE_MAX_PEAK
65
+ and metrics["rmsPcm16"] <= PCM16_SILENCE_MAX_RMS
66
+ and metrics["activeSampleCount"] == 0
67
+ ):
68
+ return {"kind": "no_speech", "reason": "digital_silence", "text": "", "signal": metrics}
69
+ return None
21
70
 
22
71
 
23
72
  def main() -> None:
@@ -33,8 +82,14 @@ def main() -> None:
33
82
  print(json.dumps({"error": "file_not_found", "path": path or ""}))
34
83
  return
35
84
 
85
+ no_speech = exact_pcm16_silence(path)
86
+ if no_speech is not None:
87
+ print(json.dumps({**no_speech, "duration": no_speech["signal"]["durationMs"] / 1000.0, "segments": []}))
88
+ return
89
+
36
90
  model_name = data.get("model") or "tiny"
37
91
  try:
92
+ import whisper
38
93
  device = os.environ.get("OMNIUS_ASR_DEVICE", "cuda").strip().lower() or "cuda"
39
94
  allow_cpu = os.environ.get("OMNIUS_ASR_ALLOW_CPU", "").strip().lower() in ("1", "true", "yes", "on")
40
95
  if device == "auto":
@@ -293080,7 +293080,6 @@ var MAX_IMAGE_BYTES = 8 * 1024 * 1024;
293080
293080
 
293081
293081
  // packages/execution/dist/tools/ocr-image-advanced.js
293082
293082
  init_process_async();
293083
- var OCR_PIPELINE_TIMEOUT_MS = 5 * 6e4;
293084
293083
 
293085
293084
  // packages/execution/dist/tools/browser-action.js
293086
293085
  init_process_lifecycle();