omnius 1.0.638 → 1.0.640
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +1314 -857
- package/dist/scripts/audio-speaker-embedding-worker.py +115 -31
- package/dist/scripts/live-whisper.py +41 -0
- package/dist/scripts/ocr-advanced.py +273 -140
- package/dist/scripts/transcribe-file.py +61 -6
- package/dist/update-worker.js +0 -1
- package/docs/DISCOVERY.json +26 -6
- package/docs/rest/endpoints/chat.md +5 -1
- package/docs/rest/endpoints/voice-vision.md +26 -23
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
|
@@ -37,6 +37,9 @@ import os
|
|
|
37
37
|
import json
|
|
38
38
|
import csv
|
|
39
39
|
import argparse
|
|
40
|
+
import signal
|
|
41
|
+
import time
|
|
42
|
+
from collections import OrderedDict
|
|
40
43
|
from pathlib import Path
|
|
41
44
|
|
|
42
45
|
def check_deps():
|
|
@@ -168,34 +171,105 @@ PSM_MODES = {
|
|
|
168
171
|
11: "sparse",
|
|
169
172
|
}
|
|
170
173
|
|
|
174
|
+
# Advanced OCR is a bounded recovery pipeline, not a blind Cartesian product.
|
|
175
|
+
# A 327x333 crop previously paid for 24 variants/PSMs and two Tesseract child
|
|
176
|
+
# processes per attempt. Small inputs now begin with two high-yield attempts;
|
|
177
|
+
# larger/low-evidence images expand only while budget remains.
|
|
178
|
+
SMALL_CROP_AREA_PX = 512_000
|
|
179
|
+
MEDIUM_IMAGE_AREA_PX = 2_000_000
|
|
180
|
+
MAX_PIPELINE_DEADLINE_MS = 80_000
|
|
181
|
+
MAX_TESSERACT_ATTEMPT_SECONDS = 12.0
|
|
182
|
+
MIN_TESSERACT_ATTEMPT_SECONDS = 0.25
|
|
183
|
+
ACTIVE_DEADLINE = None
|
|
171
184
|
|
|
172
|
-
# ---------------------------------------------------------------------------
|
|
173
|
-
# OCR execution
|
|
174
|
-
# ---------------------------------------------------------------------------
|
|
175
185
|
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
Returns (text, confidence, line_count)."""
|
|
179
|
-
pil_img = Image.fromarray(binary_img)
|
|
180
|
-
config = f"--psm {psm}"
|
|
186
|
+
class OcrPipelineTimeout(RuntimeError):
|
|
187
|
+
pass
|
|
181
188
|
|
|
182
|
-
# Engine failures are not blank documents. Let callers distinguish a real
|
|
183
|
-
# low-information image from a missing language pack or broken Tesseract.
|
|
184
|
-
text = pytesseract.image_to_string(pil_img, lang=language, config=config).strip()
|
|
185
189
|
|
|
186
|
-
|
|
190
|
+
class OcrPipelineCancelled(RuntimeError):
|
|
191
|
+
pass
|
|
187
192
|
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
+
|
|
194
|
+
def cancellation_signal_handler(signum, _frame):
|
|
195
|
+
raise OcrPipelineCancelled(f"received signal {signum}")
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
class OcrDeadline:
|
|
199
|
+
def __init__(self, deadline_ms):
|
|
200
|
+
self.deadline_ms = max(1_000, min(int(deadline_ms), MAX_PIPELINE_DEADLINE_MS))
|
|
201
|
+
self.started = time.monotonic()
|
|
202
|
+
|
|
203
|
+
def remaining_seconds(self):
|
|
204
|
+
return self.deadline_ms / 1000.0 - (time.monotonic() - self.started)
|
|
205
|
+
|
|
206
|
+
def check(self, stage):
|
|
207
|
+
if self.remaining_seconds() <= 0:
|
|
208
|
+
raise OcrPipelineTimeout(f"OCR deadline exceeded during {stage}")
|
|
209
|
+
|
|
210
|
+
def tesseract_timeout_seconds(self):
|
|
211
|
+
self.check("Tesseract scheduling")
|
|
212
|
+
return max(
|
|
213
|
+
MIN_TESSERACT_ATTEMPT_SECONDS,
|
|
214
|
+
min(MAX_TESSERACT_ATTEMPT_SECONDS, self.remaining_seconds()),
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def diagnostic(code, message, deadline, stage, attempts_completed=0, attempts_planned=0):
|
|
219
|
+
return {
|
|
220
|
+
"schema": "omnius.ocr-diagnostic.v1",
|
|
221
|
+
"code": code,
|
|
222
|
+
"message": message,
|
|
223
|
+
"stage": stage,
|
|
224
|
+
"deadline_ms": deadline.deadline_ms,
|
|
225
|
+
"attempts_completed": attempts_completed,
|
|
226
|
+
"attempts_planned": attempts_planned,
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
# ---------------------------------------------------------------------------
|
|
231
|
+
# OCR execution
|
|
232
|
+
# ---------------------------------------------------------------------------
|
|
233
|
+
|
|
234
|
+
def text_from_tesseract_data(data):
|
|
235
|
+
"""Reconstruct line breaks from one TSV pass; do not spawn Tesseract twice."""
|
|
236
|
+
lines = OrderedDict()
|
|
237
|
+
texts = data.get("text", [])
|
|
238
|
+
for index, raw_text in enumerate(texts):
|
|
239
|
+
text = str(raw_text or "").strip()
|
|
240
|
+
if not text:
|
|
241
|
+
continue
|
|
242
|
+
key = tuple(
|
|
243
|
+
int(data.get(field, [0] * len(texts))[index] or 0)
|
|
244
|
+
for field in ("block_num", "par_num", "line_num")
|
|
193
245
|
)
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
avg_conf = 0.0
|
|
246
|
+
lines.setdefault(key, []).append(text)
|
|
247
|
+
return "\n".join(" ".join(words) for words in lines.values()).strip()
|
|
248
|
+
|
|
198
249
|
|
|
250
|
+
def run_tesseract(binary_img, deadline, language="eng", psm=6):
|
|
251
|
+
"""Run one bounded Tesseract TSV pass and derive text plus confidence."""
|
|
252
|
+
deadline.check("Tesseract")
|
|
253
|
+
pil_img = Image.fromarray(binary_img)
|
|
254
|
+
config = f"--psm {psm}"
|
|
255
|
+
data = pytesseract.image_to_data(
|
|
256
|
+
pil_img,
|
|
257
|
+
lang=language,
|
|
258
|
+
config=config,
|
|
259
|
+
output_type=pytesseract.Output.DICT,
|
|
260
|
+
timeout=deadline.tesseract_timeout_seconds(),
|
|
261
|
+
)
|
|
262
|
+
text = text_from_tesseract_data(data)
|
|
263
|
+
confs = []
|
|
264
|
+
for value in data.get("conf", []):
|
|
265
|
+
try:
|
|
266
|
+
confidence = float(value)
|
|
267
|
+
except (TypeError, ValueError):
|
|
268
|
+
continue
|
|
269
|
+
if confidence >= 0:
|
|
270
|
+
confs.append(confidence)
|
|
271
|
+
avg_conf = sum(confs) / len(confs) if confs else 0.0
|
|
272
|
+
line_count = len([line for line in text.split("\n") if line.strip()])
|
|
199
273
|
return text, avg_conf, line_count
|
|
200
274
|
|
|
201
275
|
|
|
@@ -226,6 +300,28 @@ def extract_pixel_region(gray, x, y, w, h):
|
|
|
226
300
|
return gray[y:y+h, x:x+w]
|
|
227
301
|
|
|
228
302
|
|
|
303
|
+
def build_ocr_plan(image_area_px, single_psm=None):
|
|
304
|
+
"""Return the smallest credible variant/PSM plan for the effective image."""
|
|
305
|
+
psm_modes = [single_psm] if single_psm else (
|
|
306
|
+
[6] if image_area_px <= SMALL_CROP_AREA_PX
|
|
307
|
+
else [6, 11] if image_area_px <= MEDIUM_IMAGE_AREA_PX
|
|
308
|
+
else [6, 11, 4]
|
|
309
|
+
)
|
|
310
|
+
variants = (
|
|
311
|
+
["otsu", "adaptive_fine"] if image_area_px <= SMALL_CROP_AREA_PX
|
|
312
|
+
else ["otsu", "adaptive_fine", "denoise", "sharpen_unsharp"]
|
|
313
|
+
if image_area_px <= MEDIUM_IMAGE_AREA_PX
|
|
314
|
+
else list(ALL_VARIANTS.keys())
|
|
315
|
+
)
|
|
316
|
+
# OTSU/PSM6 is deliberately first: a legible result ends the search
|
|
317
|
+
# instead of paying for redundant variants that cannot improve the answer.
|
|
318
|
+
return [(variant, psm) for psm in psm_modes for variant in variants]
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def has_sufficient_evidence(text, confidence, line_count):
|
|
322
|
+
return len(text.strip()) >= 8 and confidence >= 70.0 and line_count >= 1
|
|
323
|
+
|
|
324
|
+
|
|
229
325
|
# ---------------------------------------------------------------------------
|
|
230
326
|
# Output writers
|
|
231
327
|
# ---------------------------------------------------------------------------
|
|
@@ -295,142 +391,139 @@ def write_all_outputs(text, base_name, output_dir):
|
|
|
295
391
|
# Main pipeline
|
|
296
392
|
# ---------------------------------------------------------------------------
|
|
297
393
|
|
|
298
|
-
def
|
|
299
|
-
|
|
300
|
-
"""Run the full multi-variant, multi-PSM OCR pipeline."""
|
|
301
|
-
|
|
302
|
-
# Load image
|
|
303
|
-
img = cv2.imread(image_path)
|
|
304
|
-
if img is None:
|
|
305
|
-
return {"error": f"Could not load image: {image_path}"}
|
|
306
|
-
|
|
307
|
-
h_orig, w_orig = img.shape[:2]
|
|
308
|
-
gray = to_grayscale(img)
|
|
309
|
-
|
|
310
|
-
# Upscale 2x
|
|
311
|
-
gray_2x = upscale_2x(gray)
|
|
312
|
-
|
|
313
|
-
# If a pixel region is specified, crop before processing
|
|
314
|
-
if pixel_region:
|
|
315
|
-
rx, ry, rw, rh = pixel_region
|
|
316
|
-
# Scale region coords to match 2x upscale
|
|
317
|
-
gray_2x = extract_pixel_region(gray_2x, rx * 2, ry * 2, rw * 2, rh * 2)
|
|
318
|
-
|
|
319
|
-
# Determine PSM modes to test
|
|
320
|
-
psm_modes = [single_psm] if single_psm else [4, 6, 11]
|
|
321
|
-
|
|
322
|
-
# Generate all variants and run OCR
|
|
394
|
+
def run_variant_plan(gray, plan, deadline, language, debug_dir=None, debug_prefix="full"):
|
|
395
|
+
"""Run a bounded plan, stopping early once a legible result is proven."""
|
|
323
396
|
all_results = {}
|
|
324
397
|
ocr_errors = []
|
|
325
398
|
best_key = None
|
|
326
399
|
best_score = -1
|
|
400
|
+
early_exit = False
|
|
327
401
|
|
|
328
|
-
for vname,
|
|
402
|
+
for vname, psm in plan:
|
|
403
|
+
deadline.check("preprocessing")
|
|
404
|
+
key = f"{vname}_psm{psm}"
|
|
329
405
|
try:
|
|
330
|
-
binary =
|
|
331
|
-
except Exception:
|
|
406
|
+
binary = ALL_VARIANTS[vname](gray)
|
|
407
|
+
except Exception as error:
|
|
408
|
+
ocr_errors.append(f"{key}: preprocessing failed: {error}")
|
|
332
409
|
continue
|
|
333
|
-
|
|
334
|
-
# Save debug images
|
|
335
410
|
if debug_dir:
|
|
336
411
|
os.makedirs(debug_dir, exist_ok=True)
|
|
337
|
-
cv2.imwrite(os.path.join(debug_dir, f"
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
}
|
|
356
|
-
|
|
357
|
-
if score > best_score:
|
|
358
|
-
best_score = score
|
|
359
|
-
best_key = key
|
|
360
|
-
|
|
361
|
-
if not best_key:
|
|
362
|
-
detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
|
|
363
|
-
return {"error": f"Tesseract failed for every OCR variant: {detail}"}
|
|
364
|
-
|
|
365
|
-
best = all_results[best_key]
|
|
366
|
-
result = {
|
|
367
|
-
"text": best["text"],
|
|
368
|
-
"confidence": best["confidence"],
|
|
369
|
-
"variant": best_key,
|
|
370
|
-
"chars": best["chars"],
|
|
371
|
-
"lines": best["lines"],
|
|
372
|
-
"score": best["score"],
|
|
373
|
-
"image_size": f"{w_orig}x{h_orig}",
|
|
374
|
-
"variants_tested": len(all_results),
|
|
375
|
-
"all_variants": all_results,
|
|
376
|
-
}
|
|
377
|
-
|
|
378
|
-
# Region-based OCR
|
|
379
|
-
if do_regions:
|
|
380
|
-
regions = {}
|
|
381
|
-
region_defs = {
|
|
382
|
-
"header": (0, 35),
|
|
383
|
-
"body": (30, 80),
|
|
384
|
-
"footer": (75, 100),
|
|
412
|
+
cv2.imwrite(os.path.join(debug_dir, f"{debug_prefix}_{vname}.png"), binary)
|
|
413
|
+
try:
|
|
414
|
+
text, confidence, line_count = run_tesseract(binary, deadline, language, psm)
|
|
415
|
+
except OcrPipelineTimeout:
|
|
416
|
+
raise
|
|
417
|
+
except OcrPipelineCancelled:
|
|
418
|
+
raise
|
|
419
|
+
except Exception as error:
|
|
420
|
+
ocr_errors.append(f"{key}: {error}")
|
|
421
|
+
continue
|
|
422
|
+
char_count = len(text)
|
|
423
|
+
score = compute_score(text, confidence, line_count)
|
|
424
|
+
all_results[key] = {
|
|
425
|
+
"text": text,
|
|
426
|
+
"chars": char_count,
|
|
427
|
+
"lines": line_count,
|
|
428
|
+
"confidence": round(confidence, 1),
|
|
429
|
+
"score": round(score, 1),
|
|
385
430
|
}
|
|
431
|
+
if score > best_score:
|
|
432
|
+
best_score = score
|
|
433
|
+
best_key = key
|
|
434
|
+
if has_sufficient_evidence(text, confidence, line_count):
|
|
435
|
+
early_exit = True
|
|
436
|
+
break
|
|
386
437
|
|
|
387
|
-
|
|
388
|
-
region_gray = extract_region(gray_2x, y_start, y_end)
|
|
389
|
-
|
|
390
|
-
if debug_dir:
|
|
391
|
-
cv2.imwrite(os.path.join(debug_dir, f"region_{rname}.png"), region_gray)
|
|
392
|
-
|
|
393
|
-
# Test all variants on each region for best accuracy
|
|
394
|
-
region_best = ""
|
|
395
|
-
region_best_score = -1
|
|
396
|
-
|
|
397
|
-
for vname in ["otsu", "denoise", "adaptive_fine", "sharpen_unsharp"]:
|
|
398
|
-
if vname not in ALL_VARIANTS:
|
|
399
|
-
continue
|
|
400
|
-
try:
|
|
401
|
-
binary = ALL_VARIANTS[vname](region_gray)
|
|
402
|
-
except Exception:
|
|
403
|
-
continue
|
|
438
|
+
return all_results, ocr_errors, best_key, early_exit
|
|
404
439
|
|
|
405
|
-
if debug_dir:
|
|
406
|
-
cv2.imwrite(os.path.join(debug_dir, f"region_{rname}_{vname}.png"), binary)
|
|
407
440
|
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
region_best = text
|
|
416
|
-
|
|
417
|
-
regions[rname] = region_best
|
|
418
|
-
|
|
419
|
-
result["regions"] = regions
|
|
441
|
+
def run_pipeline(image_path, deadline, language="eng", do_regions=False, debug_dir=None,
|
|
442
|
+
single_psm=None, pixel_region=None, output_dir=None):
|
|
443
|
+
"""Run bounded OCR. The effective crop controls plan size, not the full frame."""
|
|
444
|
+
deadline.check("image load")
|
|
445
|
+
img = cv2.imread(image_path)
|
|
446
|
+
if img is None:
|
|
447
|
+
return {"error": f"Could not load image: {image_path}"}
|
|
420
448
|
|
|
421
|
-
|
|
422
|
-
|
|
449
|
+
h_orig, w_orig = img.shape[:2]
|
|
450
|
+
gray = to_grayscale(img)
|
|
451
|
+
if pixel_region:
|
|
452
|
+
rx, ry, rw, rh = pixel_region
|
|
453
|
+
if rx >= w_orig or ry >= h_orig:
|
|
454
|
+
return {"error": "Requested OCR region lies outside the image"}
|
|
455
|
+
gray = extract_pixel_region(gray, rx, ry, rw, rh)
|
|
456
|
+
if gray.size == 0:
|
|
457
|
+
return {"error": "Requested OCR region is empty"}
|
|
458
|
+
|
|
459
|
+
effective_area = int(gray.shape[0] * gray.shape[1])
|
|
460
|
+
deadline.check("image conditioning")
|
|
461
|
+
gray_2x = upscale_2x(gray)
|
|
462
|
+
plan = build_ocr_plan(effective_area, single_psm)
|
|
463
|
+
attempts_completed = 0
|
|
464
|
+
try:
|
|
465
|
+
all_results, ocr_errors, best_key, early_exit = run_variant_plan(
|
|
466
|
+
gray_2x, plan, deadline, language, debug_dir
|
|
467
|
+
)
|
|
468
|
+
attempts_completed = len(all_results)
|
|
469
|
+
if not best_key:
|
|
470
|
+
detail = ocr_errors[-1] if ocr_errors else "no preprocessing variant completed"
|
|
471
|
+
return {"error": f"Tesseract failed for every bounded OCR attempt: {detail}"}
|
|
472
|
+
|
|
473
|
+
best = all_results[best_key]
|
|
474
|
+
result = {
|
|
475
|
+
"text": best["text"],
|
|
476
|
+
"confidence": best["confidence"],
|
|
477
|
+
"variant": best_key,
|
|
478
|
+
"chars": best["chars"],
|
|
479
|
+
"lines": best["lines"],
|
|
480
|
+
"score": best["score"],
|
|
481
|
+
"image_size": f"{w_orig}x{h_orig}",
|
|
482
|
+
"variants_tested": len(all_results),
|
|
483
|
+
"all_variants": all_results,
|
|
484
|
+
"strategy": {
|
|
485
|
+
"effective_area_px": effective_area,
|
|
486
|
+
"attempts_planned": len(plan),
|
|
487
|
+
"attempts_completed": attempts_completed,
|
|
488
|
+
"early_exit": early_exit,
|
|
489
|
+
"deadline_ms": deadline.deadline_ms,
|
|
490
|
+
},
|
|
491
|
+
}
|
|
423
492
|
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
493
|
+
if do_regions:
|
|
494
|
+
regions = {}
|
|
495
|
+
region_defs = {"header": (0, 35), "body": (30, 80), "footer": (75, 100)}
|
|
496
|
+
for rname, (y_start, y_end) in region_defs.items():
|
|
497
|
+
deadline.check(f"region {rname}")
|
|
498
|
+
region_gray = extract_region(gray_2x, y_start, y_end)
|
|
499
|
+
# Region requests use at most two high-yield attempts. The
|
|
500
|
+
# main result already provides full-frame coverage.
|
|
501
|
+
region_plan = build_ocr_plan(region_gray.shape[0] * region_gray.shape[1], single_psm)[:2]
|
|
502
|
+
region_results, _errors, region_best_key, _early_exit = run_variant_plan(
|
|
503
|
+
region_gray, region_plan, deadline, language, debug_dir, f"region_{rname}"
|
|
504
|
+
)
|
|
505
|
+
regions[rname] = region_results[region_best_key]["text"] if region_best_key else ""
|
|
506
|
+
result["regions"] = regions
|
|
429
507
|
|
|
430
|
-
|
|
508
|
+
if debug_dir:
|
|
509
|
+
result["debug_dir"] = debug_dir
|
|
510
|
+
if output_dir:
|
|
511
|
+
base_name = Path(image_path).stem
|
|
512
|
+
result["output_files"] = write_all_outputs(best["text"], base_name, output_dir)
|
|
513
|
+
return result
|
|
514
|
+
except OcrPipelineTimeout as error:
|
|
515
|
+
return {
|
|
516
|
+
"error": str(error),
|
|
517
|
+
"diagnostic": diagnostic("ocr_timeout", str(error), deadline, "pipeline", attempts_completed, len(plan)),
|
|
518
|
+
}
|
|
519
|
+
except OcrPipelineCancelled as error:
|
|
520
|
+
return {
|
|
521
|
+
"error": "Advanced OCR cancelled",
|
|
522
|
+
"diagnostic": diagnostic("ocr_cancelled", str(error), deadline, "pipeline", attempts_completed, len(plan)),
|
|
523
|
+
}
|
|
431
524
|
|
|
432
525
|
|
|
433
|
-
def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
|
|
526
|
+
def run_batch(images_dir, deadline, language="eng", do_regions=False, debug_dir=None,
|
|
434
527
|
output_dir=None):
|
|
435
528
|
"""Process all images in a directory."""
|
|
436
529
|
images_dir = os.path.abspath(images_dir)
|
|
@@ -450,10 +543,18 @@ def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
|
|
|
450
543
|
return {"error": f"No image files found in {images_dir}"}
|
|
451
544
|
|
|
452
545
|
for img_file in image_files:
|
|
546
|
+
try:
|
|
547
|
+
deadline.check("batch scheduling")
|
|
548
|
+
except OcrPipelineTimeout as error:
|
|
549
|
+
return {
|
|
550
|
+
"error": str(error),
|
|
551
|
+
"diagnostic": diagnostic("ocr_timeout", str(error), deadline, "batch", len(batch_results), len(image_files)),
|
|
552
|
+
}
|
|
453
553
|
img_path = os.path.join(images_dir, img_file)
|
|
454
554
|
img_debug = os.path.join(debug_dir, Path(img_file).stem) if debug_dir else None
|
|
455
555
|
result = run_pipeline(
|
|
456
556
|
img_path,
|
|
557
|
+
deadline,
|
|
457
558
|
language=language,
|
|
458
559
|
do_regions=do_regions,
|
|
459
560
|
debug_dir=img_debug,
|
|
@@ -469,6 +570,14 @@ def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
|
|
|
469
570
|
"output_files": result.get("output_files"),
|
|
470
571
|
"error": result.get("error"),
|
|
471
572
|
}
|
|
573
|
+
if result.get("diagnostic", {}).get("code") in {"ocr_timeout", "ocr_cancelled"}:
|
|
574
|
+
return {
|
|
575
|
+
"error": result["error"],
|
|
576
|
+
"diagnostic": result["diagnostic"],
|
|
577
|
+
"batch": True,
|
|
578
|
+
"images_processed": len(batch_results),
|
|
579
|
+
"results": batch_results,
|
|
580
|
+
}
|
|
472
581
|
|
|
473
582
|
# Write summary
|
|
474
583
|
summary_path = os.path.join(out_dir, "OCR_PROCESSING_SUMMARY.md")
|
|
@@ -497,6 +606,7 @@ def run_batch(images_dir, language="eng", do_regions=False, debug_dir=None,
|
|
|
497
606
|
|
|
498
607
|
|
|
499
608
|
def main():
|
|
609
|
+
global ACTIVE_DEADLINE
|
|
500
610
|
parser = argparse.ArgumentParser(
|
|
501
611
|
description="Advanced multi-variant OCR pipeline for omnius"
|
|
502
612
|
)
|
|
@@ -520,13 +630,20 @@ def main():
|
|
|
520
630
|
help="Write TXT + CSV + PDF outputs to this directory")
|
|
521
631
|
parser.add_argument("--batch", action="store_true",
|
|
522
632
|
help="Process all images in a directory")
|
|
633
|
+
parser.add_argument("--deadline-ms", type=int, default=MAX_PIPELINE_DEADLINE_MS,
|
|
634
|
+
help="Bound total OCR work; clamped to the managed maximum")
|
|
523
635
|
|
|
524
636
|
args = parser.parse_args()
|
|
637
|
+
deadline = OcrDeadline(args.deadline_ms)
|
|
638
|
+
ACTIVE_DEADLINE = deadline
|
|
639
|
+
signal.signal(signal.SIGTERM, cancellation_signal_handler)
|
|
640
|
+
signal.signal(signal.SIGINT, cancellation_signal_handler)
|
|
525
641
|
|
|
526
642
|
# Batch mode
|
|
527
643
|
if args.batch or os.path.isdir(args.image):
|
|
528
644
|
result = run_batch(
|
|
529
645
|
args.image,
|
|
646
|
+
deadline,
|
|
530
647
|
language=args.language,
|
|
531
648
|
do_regions=args.regions,
|
|
532
649
|
debug_dir=args.debug_dir,
|
|
@@ -558,6 +675,7 @@ def main():
|
|
|
558
675
|
|
|
559
676
|
result = run_pipeline(
|
|
560
677
|
args.image,
|
|
678
|
+
deadline,
|
|
561
679
|
language=args.language,
|
|
562
680
|
do_regions=args.regions,
|
|
563
681
|
debug_dir=args.debug_dir,
|
|
@@ -578,4 +696,19 @@ def main():
|
|
|
578
696
|
|
|
579
697
|
|
|
580
698
|
if __name__ == "__main__":
|
|
581
|
-
|
|
699
|
+
try:
|
|
700
|
+
main()
|
|
701
|
+
except OcrPipelineTimeout as error:
|
|
702
|
+
deadline = ACTIVE_DEADLINE or OcrDeadline(MAX_PIPELINE_DEADLINE_MS)
|
|
703
|
+
print(json.dumps({
|
|
704
|
+
"error": str(error),
|
|
705
|
+
"diagnostic": diagnostic("ocr_timeout", str(error), deadline, "entrypoint"),
|
|
706
|
+
}))
|
|
707
|
+
sys.exit(124)
|
|
708
|
+
except OcrPipelineCancelled as error:
|
|
709
|
+
deadline = ACTIVE_DEADLINE or OcrDeadline(MAX_PIPELINE_DEADLINE_MS)
|
|
710
|
+
print(json.dumps({
|
|
711
|
+
"error": "Advanced OCR cancelled",
|
|
712
|
+
"diagnostic": diagnostic("ocr_cancelled", str(error), deadline, "entrypoint"),
|
|
713
|
+
}))
|
|
714
|
+
sys.exit(130)
|
|
@@ -8,16 +8,65 @@ stdout: JSON {"text": "...", "duration": null, "segments": []}
|
|
|
8
8
|
|
|
9
9
|
ASR is GPU-first: CPU fallback is refused unless OMNIUS_ASR_ALLOW_CPU=1.
|
|
10
10
|
"""
|
|
11
|
-
import sys, os, json, warnings
|
|
11
|
+
import sys, os, json, math, struct, warnings, wave
|
|
12
12
|
|
|
13
13
|
warnings.filterwarnings("ignore")
|
|
14
14
|
os.environ.setdefault("PYTHONWARNINGS", "ignore")
|
|
15
15
|
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
16
|
+
PCM16_SILENCE_MAX_PEAK = 4
|
|
17
|
+
PCM16_SILENCE_MAX_RMS = 1.0
|
|
18
|
+
PCM16_ACTIVE_SAMPLE_THRESHOLD = 8
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def exact_pcm16_silence(path):
|
|
22
|
+
"""Return typed no-speech evidence for exact PCM16/16k/mono WAV only.
|
|
23
|
+
|
|
24
|
+
This is intentionally not VAD. It reads user bytes but never rewrites,
|
|
25
|
+
normalizes, decodes another media format, or instantiates Whisper.
|
|
26
|
+
"""
|
|
27
|
+
try:
|
|
28
|
+
with wave.open(path, "rb") as wav:
|
|
29
|
+
if (
|
|
30
|
+
wav.getcomptype() != "NONE"
|
|
31
|
+
or wav.getnchannels() != 1
|
|
32
|
+
or wav.getsampwidth() != 2
|
|
33
|
+
or wav.getframerate() != 16000
|
|
34
|
+
):
|
|
35
|
+
return None
|
|
36
|
+
frames = wav.readframes(wav.getnframes())
|
|
37
|
+
except (wave.Error, OSError, EOFError):
|
|
38
|
+
return None
|
|
39
|
+
sample_count = len(frames) // 2
|
|
40
|
+
if sample_count == 0:
|
|
41
|
+
metrics = {
|
|
42
|
+
"sampleRateHz": 16000, "channels": 1, "bitsPerSample": 16,
|
|
43
|
+
"sampleCount": 0, "durationMs": 0.0, "peakPcm16": 0,
|
|
44
|
+
"rmsPcm16": 0.0, "activeSampleCount": 0, "activeSampleRatio": 0.0,
|
|
45
|
+
}
|
|
46
|
+
else:
|
|
47
|
+
samples = struct.iter_unpack("<h", frames[: sample_count * 2])
|
|
48
|
+
total_squares = 0
|
|
49
|
+
peak = 0
|
|
50
|
+
active = 0
|
|
51
|
+
for (sample,) in samples:
|
|
52
|
+
magnitude = abs(sample)
|
|
53
|
+
total_squares += sample * sample
|
|
54
|
+
peak = max(peak, magnitude)
|
|
55
|
+
if magnitude >= PCM16_ACTIVE_SAMPLE_THRESHOLD:
|
|
56
|
+
active += 1
|
|
57
|
+
metrics = {
|
|
58
|
+
"sampleRateHz": 16000, "channels": 1, "bitsPerSample": 16,
|
|
59
|
+
"sampleCount": sample_count, "durationMs": sample_count / 16.0,
|
|
60
|
+
"peakPcm16": peak, "rmsPcm16": math.sqrt(total_squares / sample_count),
|
|
61
|
+
"activeSampleCount": active, "activeSampleRatio": active / sample_count,
|
|
62
|
+
}
|
|
63
|
+
if (
|
|
64
|
+
metrics["peakPcm16"] <= PCM16_SILENCE_MAX_PEAK
|
|
65
|
+
and metrics["rmsPcm16"] <= PCM16_SILENCE_MAX_RMS
|
|
66
|
+
and metrics["activeSampleCount"] == 0
|
|
67
|
+
):
|
|
68
|
+
return {"kind": "no_speech", "reason": "digital_silence", "text": "", "signal": metrics}
|
|
69
|
+
return None
|
|
21
70
|
|
|
22
71
|
|
|
23
72
|
def main() -> None:
|
|
@@ -33,8 +82,14 @@ def main() -> None:
|
|
|
33
82
|
print(json.dumps({"error": "file_not_found", "path": path or ""}))
|
|
34
83
|
return
|
|
35
84
|
|
|
85
|
+
no_speech = exact_pcm16_silence(path)
|
|
86
|
+
if no_speech is not None:
|
|
87
|
+
print(json.dumps({**no_speech, "duration": no_speech["signal"]["durationMs"] / 1000.0, "segments": []}))
|
|
88
|
+
return
|
|
89
|
+
|
|
36
90
|
model_name = data.get("model") or "tiny"
|
|
37
91
|
try:
|
|
92
|
+
import whisper
|
|
38
93
|
device = os.environ.get("OMNIUS_ASR_DEVICE", "cuda").strip().lower() or "cuda"
|
|
39
94
|
allow_cpu = os.environ.get("OMNIUS_ASR_ALLOW_CPU", "").strip().lower() in ("1", "true", "yes", "on")
|
|
40
95
|
if device == "auto":
|
package/dist/update-worker.js
CHANGED
|
@@ -293080,7 +293080,6 @@ var MAX_IMAGE_BYTES = 8 * 1024 * 1024;
|
|
|
293080
293080
|
|
|
293081
293081
|
// packages/execution/dist/tools/ocr-image-advanced.js
|
|
293082
293082
|
init_process_async();
|
|
293083
|
-
var OCR_PIPELINE_TIMEOUT_MS = 5 * 6e4;
|
|
293084
293083
|
|
|
293085
293084
|
// packages/execution/dist/tools/browser-action.js
|
|
293086
293085
|
init_process_lifecycle();
|