graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,1581 @@
1
+ from __future__ import annotations
2
+
3
+ """Workflow-first OCR ingest helpers for image and PDF sources.
4
+
5
+ This module sits at the boundary between raw OCR and the reusable workflow
6
+ ingest pipeline. It is responsible for three ideas that are easy to conflate:
7
+
8
+ 1. **Page materialization**
9
+ - Accept already-rendered page images, or render a PDF into page images.
10
+ - Keep the page images around on disk so humans can inspect what was OCR'd.
11
+
12
+ 2. **OCR execution and resume state**
13
+ - Run one OCR provider call per page through a pluggable callable hook.
14
+ - Persist page-level progress in ``ocr-state.sqlite`` so reruns can skip
15
+ completed pages and rebuild state if the DB is missing.
16
+ - Save legacy-compatible ``page_N.json`` artifacts alongside the images.
17
+
18
+ 3. **Workflow normalization**
19
+ - Convert the serialized OCR pages into ``WorkflowIngestInput``.
20
+ - Hand that normalized input to the downstream semantic workflow parser.
21
+
22
+ Important concepts used throughout the file:
23
+
24
+ - ``OCRRunner``: a callable that OCRs exactly one page image and returns one
25
+ structured OCR page response.
26
+ - ``PDFRasterizer``: a callable that turns a PDF into a list of page image paths.
27
+ - ``OCRWorkflowStateStore``: the SQLite-backed resume store for page render/OCR
28
+ state.
29
+ - ``OCRWorkflowArtifacts``: the inspectable bundle returned after OCR prep so
30
+ tests and manual runs can open the generated files.
31
+
32
+ The public entrypoints are intentionally explicit so tests and manual runs can
33
+ inspect intermediate folders without needing to understand the legacy OCR code.
34
+ """
35
+
36
+ import base64
37
+ import contextlib
38
+ import hashlib
39
+ import json
40
+ import logging
41
+ import sqlite3
42
+ import shutil
43
+ import time
44
+ from dataclasses import dataclass
45
+ from pathlib import Path
46
+ from typing import Any, Callable, Sequence
47
+
48
+ from langchain_core.messages import HumanMessage, SystemMessage
49
+ from pydantic import BaseModel, Field
50
+ from PIL import Image
51
+ from pypdf import PdfReader
52
+
53
+ from ..models import OCRClusterResponse, SplitPage, SplitPageMeta
54
+
55
+ from .adapters import OCRPageJSON, normalize_ocr_pages
56
+ from .models import WorkflowIngestInput
57
+ from .providers import ProviderEndpointConfig, WorkflowProviderSettings, build_chat_model_for_role
58
+ from .probe import WorkflowProbe, emit_probe_event
59
+ from .service import run_ingest_workflow
60
+
61
+ _LOGGER = logging.getLogger(__name__)
62
+
63
+
64
+ class OCRImagePayload(BaseModel):
65
+ """Single OCR page input.
66
+
67
+ A payload may already exist on disk, or it can be materialized from bytes
68
+ into the working directory. The resulting file path is what the OCR model
69
+ and the legacy artifact writer operate on.
70
+ """
71
+
72
+ page_number: int | None = None
73
+ image_path: str | None = None
74
+ image_bytes_b64: str | None = None
75
+ filename: str | None = None
76
+
77
+
78
+ class OCRPageProgress(BaseModel):
79
+ page_number: int
80
+ image_path: str
81
+ image_sha256: str
82
+ json_path: str
83
+ status: str = "completed"
84
+
85
+
86
+ class OCRWorkflowProgress(BaseModel):
87
+ document_id: str
88
+ title: str
89
+ source_kind: str
90
+ total_pages: int
91
+ pages: dict[str, OCRPageProgress] = Field(default_factory=dict)
92
+
93
+
94
+ @dataclass(slots=True)
95
+ class OCRWorkflowArtifacts:
96
+ """Inspectable result bundle produced by the OCR preparation phase.
97
+
98
+ This is the bridge object between the OCR world and the workflow ingest
99
+ world. It keeps both the normalized workflow input and the on-disk
100
+ artifacts that were used to produce it.
101
+ """
102
+ workflow_input: WorkflowIngestInput
103
+ ocr_pages: list[OCRPageJSON]
104
+ legacy_dir: Path
105
+ rendered_dir: Path
106
+ state_db_path: Path
107
+ progress_path: Path
108
+ summary_path: Path
109
+ completed_pages: list[int]
110
+ reused_pages: list[int]
111
+
112
+
113
+ # Pluggable hook for "OCR one page image and return the structured OCR model".
114
+ # Signature: (image_path, page_number, provider_settings) -> OCRClusterResponse
115
+ OCRRunner = Callable[[Path, int, WorkflowProviderSettings], OCRClusterResponse]
116
+ # Pluggable hook for "render a PDF into page image paths inside a destination dir".
117
+ # Signature: (pdf_path, rendered_dir) -> list[Path]
118
+ PDFRasterizer = Callable[[Path, Path], list[Path]]
119
+
120
+ _OCR_STATE_SCHEMA_VERSION = 1
121
+
122
+
123
+ @dataclass(slots=True)
124
+ class OCRPageStateSnapshot:
125
+ """One page/stage snapshot from the SQLite resume store."""
126
+ page_number: int
127
+ stage: str
128
+ attempt_count: int
129
+ status: str
130
+ content_hash: str | None
131
+ artifact_path: str | None
132
+ last_error: str | None
133
+ last_model: str | None
134
+ last_attempted_ts: float | None
135
+
136
+
137
+ @dataclass(slots=True)
138
+ class OCRSourcePlan:
139
+ """Resolved file-system plan for one OCR ingest run.
140
+
141
+ This captures the immutable inputs needed to process a document:
142
+ where the pages live, where legacy JSON should be written, and how the
143
+ page sources should be iterated.
144
+ """
145
+
146
+ pdf_source: Path | None
147
+ rendered_dir: Path
148
+ legacy_dir: Path
149
+ page_sources: list[tuple[int, Path]]
150
+ source_kind: str
151
+ input_fingerprint: str
152
+
153
+
154
+ @dataclass(slots=True)
155
+ class OCRPageProcessingContext:
156
+ """Immutable per-run OCR processing context.
157
+
158
+ This groups the values that every page in a run needs so the per-page loop
159
+ can stay focused on the page-specific state rather than carrying a very
160
+ long parameter list.
161
+ """
162
+
163
+ document_id: str
164
+ title: str
165
+ source_kind: str
166
+ total_pages: int
167
+ input_fingerprint: str
168
+ candidate_models: list[str]
169
+ provider_settings: WorkflowProviderSettings
170
+ ocr_page_runner: OCRRunner
171
+ state_store: OCRWorkflowStateStore
172
+ probe: WorkflowProbe | None
173
+
174
+
175
+ def _resolve_ocr_source_plan(
176
+ *,
177
+ document_id: str,
178
+ output_dir: Path,
179
+ image_payloads: Sequence[OCRImagePayload] | None,
180
+ pdf_path: Path | None,
181
+ pdf_rasterizer: PDFRasterizer | None,
182
+ ) -> OCRSourcePlan:
183
+ rendered_root = output_dir / "rendered_pages"
184
+ legacy_root = output_dir / "legacy_split_pages"
185
+
186
+ pdf_source = Path(pdf_path) if pdf_path is not None else None
187
+ if pdf_source is not None:
188
+ # Render once per PDF so downstream OCR can work page-by-page against
189
+ # stable page image files and the state store can resume individual pages.
190
+ rendered_dir = rendered_root / pdf_source.stem
191
+ pdf_page_rasterizer = pdf_rasterizer or _render_pdf_to_images
192
+ rendered_paths = pdf_page_rasterizer(pdf_source, rendered_dir)
193
+ page_sources = list(enumerate(rendered_paths, start=1))
194
+ source_kind = "pdf"
195
+ input_fingerprint = _compute_input_fingerprint(pdf_path=pdf_source)
196
+ else:
197
+ rendered_dir = rendered_root / document_id
198
+ page_sources = _materialize_image_payloads(list(image_payloads or []), rendered_dir)
199
+ source_kind = "image"
200
+ input_fingerprint = _compute_input_fingerprint(page_sources=page_sources)
201
+
202
+ legacy_dir = legacy_root / _legacy_folder_name(document_id=document_id, pdf_path=pdf_source)
203
+ legacy_dir.mkdir(parents=True, exist_ok=True)
204
+ return OCRSourcePlan(
205
+ pdf_source=pdf_source,
206
+ rendered_dir=rendered_dir,
207
+ legacy_dir=legacy_dir,
208
+ page_sources=page_sources,
209
+ source_kind=source_kind,
210
+ input_fingerprint=input_fingerprint,
211
+ )
212
+
213
+
214
+ def _process_ocr_page(
215
+ *,
216
+ context: OCRPageProcessingContext,
217
+ done_count: int,
218
+ page_number: int,
219
+ image_path: Path,
220
+ image_sha256: str,
221
+ page_json_path: Path,
222
+ page_image_path: Path,
223
+ raw_pages: list[OCRPageJSON],
224
+ completed_pages: list[int],
225
+ reused_pages: list[int],
226
+ page_failures: list[int],
227
+ ) -> None:
228
+ """Process one page image, either by resuming or by running OCR models."""
229
+
230
+ state_store = context.state_store
231
+ document_id = context.document_id
232
+ title = context.title
233
+ source_kind = context.source_kind
234
+ total_pages = context.total_pages
235
+ probe = context.probe
236
+
237
+ state_store.ensure_document(
238
+ document_id=document_id,
239
+ title=title,
240
+ source_kind=source_kind,
241
+ total_pages=total_pages,
242
+ input_fingerprint=context.input_fingerprint,
243
+ )
244
+ if not state_store.should_skip_page(
245
+ document_id=document_id,
246
+ page_number=page_number,
247
+ stage="render",
248
+ content_hash=image_sha256,
249
+ artifact_path=image_path,
250
+ ):
251
+ _emit_ocr_event(
252
+ probe,
253
+ "ocr.render_started",
254
+ document_id=document_id,
255
+ page_number=page_number,
256
+ image_path=str(image_path),
257
+ )
258
+ state_store.record_attempt(
259
+ document_id=document_id,
260
+ page_number=page_number,
261
+ stage="render",
262
+ content_hash=image_sha256,
263
+ model_name="local-materialize",
264
+ artifact_path=image_path,
265
+ )
266
+ state_store.record_page_completed(
267
+ document_id=document_id,
268
+ page_number=page_number,
269
+ stage="render",
270
+ content_hash=image_sha256,
271
+ artifact_path=image_path,
272
+ model_name="local-materialize",
273
+ attempt_index=1 if state_store.get_page_state(document_id=document_id, page_number=page_number, stage="render") else None,
274
+ )
275
+ _emit_ocr_event(
276
+ probe,
277
+ "ocr.render_completed",
278
+ document_id=document_id,
279
+ page_number=page_number,
280
+ image_path=str(image_path),
281
+ )
282
+
283
+ # Resume boundary: if the OCR page artifact already matches the image,
284
+ # reuse the serialized page instead of calling the model again.
285
+ if state_store.should_skip_page(
286
+ document_id=document_id,
287
+ page_number=page_number,
288
+ stage="ocr",
289
+ content_hash=image_sha256,
290
+ artifact_path=page_json_path,
291
+ ):
292
+ split_page = SplitPage.model_validate(json.loads(page_json_path.read_text(encoding="utf-8")))
293
+ raw_pages.append(split_page.dump_supercede_parse())
294
+ completed_pages.append(page_number)
295
+ reused_pages.append(page_number)
296
+ _emit_ocr_event(
297
+ probe,
298
+ "ocr.page_reused",
299
+ document_id=document_id,
300
+ page_number=page_number,
301
+ page_json_path=str(page_json_path),
302
+ )
303
+ _LOGGER.info(
304
+ "ocr resume | %s/%s | %s | reused page %s",
305
+ done_count,
306
+ total_pages,
307
+ _progress_bar(done_count, total_pages),
308
+ page_number,
309
+ )
310
+ return
311
+
312
+ last_exc: Exception | None = None
313
+ page_completed = False
314
+ _emit_ocr_event(
315
+ probe,
316
+ "ocr.page_started",
317
+ document_id=document_id,
318
+ page_number=page_number,
319
+ page_json_path=str(page_json_path),
320
+ )
321
+ # Try candidate OCR models one-by-one for this page; the first grounded
322
+ # success wins and gets persisted as the canonical page artifact.
323
+ for candidate_model in context.candidate_models:
324
+ candidate_settings: WorkflowProviderSettings = context.provider_settings.model_copy(
325
+ update={
326
+ "ocr": context.provider_settings.ocr.model_copy(update={"model": candidate_model}),
327
+ }
328
+ )
329
+ _emit_ocr_event(
330
+ probe,
331
+ "ocr.candidate_started",
332
+ document_id=document_id,
333
+ page_number=page_number,
334
+ model_name=candidate_model,
335
+ )
336
+ attempt_index = state_store.record_attempt(
337
+ document_id=document_id,
338
+ page_number=page_number,
339
+ stage="ocr",
340
+ content_hash=image_sha256,
341
+ model_name=candidate_model,
342
+ artifact_path=page_json_path,
343
+ )
344
+ try:
345
+ response = context.ocr_page_runner(image_path, page_number, candidate_settings)
346
+ split_page: SplitPage = SplitPage(
347
+ pdf_page_num=page_number,
348
+ metadata=SplitPageMeta(
349
+ ocr_model_name=candidate_model,
350
+ ocr_datetime=0.0,
351
+ ocr_json_version="workflow_ingest_v1",
352
+ ),
353
+ **response.model_dump(dump_format="python"),
354
+ )
355
+ shutil.copy2(image_path, page_image_path)
356
+ page_json_path.write_text(
357
+ json.dumps(split_page.dump_raw(dump_format="json"), indent=2),
358
+ encoding="utf-8",
359
+ )
360
+ state_store.record_page_completed(
361
+ document_id=document_id,
362
+ page_number=page_number,
363
+ stage="ocr",
364
+ content_hash=image_sha256,
365
+ artifact_path=page_json_path,
366
+ model_name=candidate_model,
367
+ attempt_index=attempt_index,
368
+ )
369
+ # The adapter consumes plain dict pages, not the Pydantic page
370
+ # model, so we serialize the page into the legacy-compatible shape here.
371
+ raw_pages.append(split_page.dump_supercede_parse())
372
+ completed_pages.append(page_number)
373
+ page_completed = True
374
+ _emit_ocr_event(
375
+ probe,
376
+ "ocr.candidate_completed",
377
+ document_id=document_id,
378
+ page_number=page_number,
379
+ model_name=candidate_model,
380
+ page_json_path=str(page_json_path),
381
+ )
382
+ _LOGGER.info(
383
+ "ocr progress | %s/%s | %s | completed page %s | model=%s",
384
+ done_count,
385
+ total_pages,
386
+ _progress_bar(done_count, total_pages),
387
+ page_number,
388
+ candidate_model,
389
+ )
390
+ break
391
+ except Exception as exc: # noqa: BLE001
392
+ last_exc = exc
393
+ state_store.record_page_failed(
394
+ document_id=document_id,
395
+ page_number=page_number,
396
+ stage="ocr",
397
+ content_hash=image_sha256,
398
+ error_message=str(exc),
399
+ model_name=candidate_model,
400
+ attempt_index=attempt_index,
401
+ )
402
+ _LOGGER.warning(
403
+ "ocr candidate failed | page=%s model=%s attempt=%s error=%s",
404
+ page_number,
405
+ candidate_model,
406
+ attempt_index,
407
+ exc,
408
+ )
409
+ _emit_ocr_event(
410
+ probe,
411
+ "ocr.candidate_failed",
412
+ document_id=document_id,
413
+ page_number=page_number,
414
+ model_name=candidate_model,
415
+ attempt_index=attempt_index,
416
+ error=str(exc),
417
+ )
418
+ if not page_completed:
419
+ page_failures.append(page_number)
420
+ _emit_ocr_event(
421
+ probe,
422
+ "ocr.page_failed",
423
+ document_id=document_id,
424
+ page_number=page_number,
425
+ candidate_models=list(context.candidate_models),
426
+ )
427
+ _LOGGER.warning(
428
+ "ocr progress | %s/%s | %s | failed page %s after %s candidate(s)",
429
+ done_count,
430
+ total_pages,
431
+ _progress_bar(done_count, total_pages),
432
+ page_number,
433
+ len(context.candidate_models),
434
+ )
435
+ if last_exc is not None:
436
+ _LOGGER.warning("last ocr error for page %s: %s", page_number, last_exc)
437
+
438
+
439
+ def _finalize_ocr_workflow_artifacts(
440
+ *,
441
+ document_id: str,
442
+ title: str,
443
+ raw_pages: list[OCRPageJSON],
444
+ completed_pages: list[int],
445
+ reused_pages: list[int],
446
+ rendered_dir: Path,
447
+ legacy_dir: Path,
448
+ state_db_path: Path,
449
+ progress_path: Path,
450
+ summary_path: Path,
451
+ provider_settings: WorkflowProviderSettings,
452
+ state_store: OCRWorkflowStateStore,
453
+ ) -> OCRWorkflowArtifacts:
454
+ workflow_input: WorkflowIngestInput = normalize_ocr_pages(
455
+ document_id=document_id,
456
+ title=title,
457
+ pages=raw_pages,
458
+ )
459
+ artifacts = OCRWorkflowArtifacts(
460
+ workflow_input=workflow_input,
461
+ ocr_pages=raw_pages,
462
+ legacy_dir=legacy_dir,
463
+ rendered_dir=rendered_dir,
464
+ state_db_path=state_db_path,
465
+ progress_path=progress_path,
466
+ summary_path=summary_path,
467
+ completed_pages=completed_pages,
468
+ reused_pages=reused_pages,
469
+ )
470
+ _write_summary(
471
+ artifacts,
472
+ document_id=document_id,
473
+ provider_settings=provider_settings,
474
+ state_store=state_store,
475
+ )
476
+ return artifacts
477
+
478
+
479
+ class OCRWorkflowStateStore:
480
+ """SQLite-backed OCR/render state for one artifact root."""
481
+
482
+ def __init__(self, db_path: Path) -> None:
483
+ self.db_path = Path(db_path)
484
+ self.db_path.parent.mkdir(parents=True, exist_ok=True)
485
+ self._initialize()
486
+
487
+ @classmethod
488
+ def open_or_rebuild(
489
+ cls,
490
+ *,
491
+ db_path: Path,
492
+ document_id: str,
493
+ title: str,
494
+ source_kind: str,
495
+ total_pages: int,
496
+ input_fingerprint: str,
497
+ rendered_dir: Path,
498
+ legacy_dir: Path,
499
+ progress_path: Path,
500
+ ) -> "OCRWorkflowStateStore":
501
+ store = cls(db_path)
502
+ rebuilt = False
503
+ if not db_path.exists() or store._is_empty():
504
+ rebuilt = store.rebuild_from_artifacts(
505
+ document_id=document_id,
506
+ title=title,
507
+ source_kind=source_kind,
508
+ total_pages=total_pages,
509
+ input_fingerprint=input_fingerprint,
510
+ rendered_dir=rendered_dir,
511
+ legacy_dir=legacy_dir,
512
+ progress_path=progress_path,
513
+ )
514
+ else:
515
+ store.ensure_document(
516
+ document_id=document_id,
517
+ title=title,
518
+ source_kind=source_kind,
519
+ total_pages=total_pages,
520
+ input_fingerprint=input_fingerprint,
521
+ )
522
+ if rebuilt:
523
+ _LOGGER.info("ocr state rebuilt from artifacts | db=%s", db_path)
524
+ return store
525
+
526
+ def _connect(self) -> sqlite3.Connection:
527
+ conn = sqlite3.connect(str(self.db_path), timeout=10)
528
+ conn.row_factory = sqlite3.Row
529
+ return conn
530
+
531
+ @contextlib.contextmanager
532
+ def _session(self):
533
+ conn = self._connect()
534
+ try:
535
+ yield conn
536
+ conn.commit()
537
+ finally:
538
+ conn.close()
539
+
540
+ def _initialize(self) -> None:
541
+ with self._session() as conn:
542
+ conn.executescript(
543
+ """
544
+ CREATE TABLE IF NOT EXISTS document_state (
545
+ document_id TEXT PRIMARY KEY,
546
+ title TEXT NOT NULL,
547
+ source_kind TEXT NOT NULL,
548
+ input_fingerprint TEXT NOT NULL,
549
+ schema_version INTEGER NOT NULL,
550
+ total_pages INTEGER NOT NULL,
551
+ is_completed INTEGER NOT NULL DEFAULT 0,
552
+ last_updated_ts REAL NOT NULL
553
+ );
554
+
555
+ CREATE TABLE IF NOT EXISTS page_state (
556
+ document_id TEXT NOT NULL,
557
+ page_number INTEGER NOT NULL,
558
+ stage TEXT NOT NULL,
559
+ attempt_count INTEGER NOT NULL DEFAULT 0,
560
+ status TEXT NOT NULL DEFAULT 'pending',
561
+ content_hash TEXT,
562
+ artifact_path TEXT,
563
+ last_error TEXT,
564
+ last_model TEXT,
565
+ last_attempted_ts REAL,
566
+ PRIMARY KEY (document_id, page_number, stage)
567
+ );
568
+
569
+ CREATE TABLE IF NOT EXISTS model_attempts (
570
+ document_id TEXT NOT NULL,
571
+ page_number INTEGER NOT NULL,
572
+ stage TEXT NOT NULL,
573
+ attempt_index INTEGER NOT NULL,
574
+ model_name TEXT NOT NULL,
575
+ status TEXT NOT NULL,
576
+ error_message TEXT,
577
+ attempted_ts REAL NOT NULL,
578
+ PRIMARY KEY (document_id, page_number, stage, attempt_index)
579
+ );
580
+ """
581
+ )
582
+
583
+ def _is_empty(self) -> bool:
584
+ with self._session() as conn:
585
+ row = conn.execute("SELECT COUNT(*) AS count FROM document_state").fetchone()
586
+ return bool(row is None or int(row["count"]) == 0)
587
+
588
+ def ensure_document(
589
+ self,
590
+ *,
591
+ document_id: str,
592
+ title: str,
593
+ source_kind: str,
594
+ total_pages: int,
595
+ input_fingerprint: str,
596
+ ) -> None:
597
+ now = time.time()
598
+ with self._session() as conn:
599
+ existing = conn.execute(
600
+ "SELECT input_fingerprint FROM document_state WHERE document_id = ?",
601
+ (document_id,),
602
+ ).fetchone()
603
+ if existing is not None and str(existing["input_fingerprint"]) != input_fingerprint:
604
+ conn.execute("DELETE FROM model_attempts WHERE document_id = ?", (document_id,))
605
+ conn.execute("DELETE FROM page_state WHERE document_id = ?", (document_id,))
606
+ conn.execute("DELETE FROM document_state WHERE document_id = ?", (document_id,))
607
+ conn.execute(
608
+ """
609
+ INSERT INTO document_state (
610
+ document_id, title, source_kind, input_fingerprint,
611
+ schema_version, total_pages, is_completed, last_updated_ts
612
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?)
613
+ ON CONFLICT(document_id) DO UPDATE SET
614
+ title = excluded.title,
615
+ source_kind = excluded.source_kind,
616
+ input_fingerprint = excluded.input_fingerprint,
617
+ schema_version = excluded.schema_version,
618
+ total_pages = excluded.total_pages,
619
+ last_updated_ts = excluded.last_updated_ts
620
+ """,
621
+ (
622
+ document_id,
623
+ title,
624
+ source_kind,
625
+ input_fingerprint,
626
+ _OCR_STATE_SCHEMA_VERSION,
627
+ total_pages,
628
+ 0,
629
+ now,
630
+ ),
631
+ )
632
+
633
+ def rebuild_from_artifacts(
634
+ self,
635
+ *,
636
+ document_id: str,
637
+ title: str,
638
+ source_kind: str,
639
+ total_pages: int,
640
+ input_fingerprint: str,
641
+ rendered_dir: Path,
642
+ legacy_dir: Path,
643
+ progress_path: Path,
644
+ ) -> bool:
645
+ self.ensure_document(
646
+ document_id=document_id,
647
+ title=title,
648
+ source_kind=source_kind,
649
+ total_pages=total_pages,
650
+ input_fingerprint=input_fingerprint,
651
+ )
652
+ rebuilt_any = False
653
+ progress_payload: dict[str, Any] = {}
654
+ if progress_path.exists():
655
+ progress_payload = json.loads(progress_path.read_text(encoding="utf-8"))
656
+ progress_pages = progress_payload.get("pages", {})
657
+ for page_number, path in _scan_rendered_pages(rendered_dir):
658
+ rebuilt_any = True
659
+ self._upsert_page_state(
660
+ document_id=document_id,
661
+ page_number=page_number,
662
+ stage="render",
663
+ attempt_count=1,
664
+ status="completed",
665
+ content_hash=_sha256_file(path),
666
+ artifact_path=str(path),
667
+ last_error=None,
668
+ last_model=None,
669
+ )
670
+ for page_number, json_path in _scan_legacy_json_pages(legacy_dir):
671
+ content_hash = None
672
+ record = progress_pages.get(str(page_number))
673
+ if isinstance(record, dict):
674
+ content_hash = record.get("image_sha256")
675
+ if content_hash is None:
676
+ content_hash = _find_matching_render_hash(rendered_dir, legacy_dir, page_number)
677
+ rebuilt_any = True
678
+ self._upsert_page_state(
679
+ document_id=document_id,
680
+ page_number=page_number,
681
+ stage="ocr",
682
+ attempt_count=1,
683
+ status="completed",
684
+ content_hash=content_hash,
685
+ artifact_path=str(json_path),
686
+ last_error=None,
687
+ last_model=None,
688
+ )
689
+ self.refresh_document_completion(document_id=document_id)
690
+ return rebuilt_any
691
+
692
+ def _upsert_page_state(
693
+ self,
694
+ *,
695
+ document_id: str,
696
+ page_number: int,
697
+ stage: str,
698
+ attempt_count: int,
699
+ status: str,
700
+ content_hash: str | None,
701
+ artifact_path: str | None,
702
+ last_error: str | None,
703
+ last_model: str | None,
704
+ ) -> None:
705
+ now = time.time()
706
+ with self._session() as conn:
707
+ conn.execute(
708
+ """
709
+ INSERT INTO page_state (
710
+ document_id, page_number, stage, attempt_count, status,
711
+ content_hash, artifact_path, last_error, last_model, last_attempted_ts
712
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
713
+ ON CONFLICT(document_id, page_number, stage) DO UPDATE SET
714
+ attempt_count = excluded.attempt_count,
715
+ status = excluded.status,
716
+ content_hash = excluded.content_hash,
717
+ artifact_path = excluded.artifact_path,
718
+ last_error = excluded.last_error,
719
+ last_model = excluded.last_model,
720
+ last_attempted_ts = excluded.last_attempted_ts
721
+ """,
722
+ (
723
+ document_id,
724
+ page_number,
725
+ stage,
726
+ attempt_count,
727
+ status,
728
+ content_hash,
729
+ artifact_path,
730
+ last_error,
731
+ last_model,
732
+ now,
733
+ ),
734
+ )
735
+
736
+ def get_page_state(self, *, document_id: str, page_number: int, stage: str) -> OCRPageStateSnapshot | None:
737
+ with self._session() as conn:
738
+ row = conn.execute(
739
+ """
740
+ SELECT page_number, stage, attempt_count, status, content_hash, artifact_path,
741
+ last_error, last_model, last_attempted_ts
742
+ FROM page_state
743
+ WHERE document_id = ? AND page_number = ? AND stage = ?
744
+ """,
745
+ (document_id, page_number, stage),
746
+ ).fetchone()
747
+ if row is None:
748
+ return None
749
+ return OCRPageStateSnapshot(
750
+ page_number=int(row["page_number"]),
751
+ stage=str(row["stage"]),
752
+ attempt_count=int(row["attempt_count"]),
753
+ status=str(row["status"]),
754
+ content_hash=row["content_hash"],
755
+ artifact_path=row["artifact_path"],
756
+ last_error=row["last_error"],
757
+ last_model=row["last_model"],
758
+ last_attempted_ts=row["last_attempted_ts"],
759
+ )
760
+
761
+ def should_skip_page(
762
+ self,
763
+ *,
764
+ document_id: str,
765
+ page_number: int,
766
+ stage: str,
767
+ content_hash: str,
768
+ artifact_path: Path | None = None,
769
+ ) -> bool:
770
+ state = self.get_page_state(document_id=document_id, page_number=page_number, stage=stage)
771
+ if state is None or state.status != "completed" or state.content_hash != content_hash:
772
+ return False
773
+ if artifact_path is not None and not artifact_path.exists():
774
+ return False
775
+ return True
776
+
777
+ def record_attempt(
778
+ self,
779
+ *,
780
+ document_id: str,
781
+ page_number: int,
782
+ stage: str,
783
+ content_hash: str | None,
784
+ model_name: str,
785
+ artifact_path: Path | None = None,
786
+ ) -> int:
787
+ now = time.time()
788
+ with self._session() as conn:
789
+ row = conn.execute(
790
+ """
791
+ SELECT attempt_count
792
+ FROM page_state
793
+ WHERE document_id = ? AND page_number = ? AND stage = ?
794
+ """,
795
+ (document_id, page_number, stage),
796
+ ).fetchone()
797
+ attempt_count = int(row["attempt_count"]) + 1 if row is not None else 1
798
+ conn.execute(
799
+ """
800
+ INSERT INTO page_state (
801
+ document_id, page_number, stage, attempt_count, status,
802
+ content_hash, artifact_path, last_error, last_model, last_attempted_ts
803
+ ) VALUES (?, ?, ?, ?, 'pending', ?, ?, NULL, ?, ?)
804
+ ON CONFLICT(document_id, page_number, stage) DO UPDATE SET
805
+ attempt_count = excluded.attempt_count,
806
+ status = excluded.status,
807
+ content_hash = excluded.content_hash,
808
+ artifact_path = excluded.artifact_path,
809
+ last_error = excluded.last_error,
810
+ last_model = excluded.last_model,
811
+ last_attempted_ts = excluded.last_attempted_ts
812
+ """,
813
+ (
814
+ document_id,
815
+ page_number,
816
+ stage,
817
+ attempt_count,
818
+ content_hash,
819
+ str(artifact_path) if artifact_path is not None else None,
820
+ model_name,
821
+ now,
822
+ ),
823
+ )
824
+ conn.execute(
825
+ """
826
+ INSERT INTO model_attempts (
827
+ document_id, page_number, stage, attempt_index, model_name, status, error_message, attempted_ts
828
+ ) VALUES (?, ?, ?, ?, ?, 'pending', NULL, ?)
829
+ """,
830
+ (document_id, page_number, stage, attempt_count, model_name, now),
831
+ )
832
+ return attempt_count
833
+
834
+ def record_page_completed(
835
+ self,
836
+ *,
837
+ document_id: str,
838
+ page_number: int,
839
+ stage: str,
840
+ content_hash: str | None,
841
+ artifact_path: Path | None,
842
+ model_name: str | None,
843
+ attempt_index: int | None = None,
844
+ ) -> None:
845
+ now = time.time()
846
+ with self._session() as conn:
847
+ conn.execute(
848
+ """
849
+ UPDATE page_state
850
+ SET status = 'completed',
851
+ content_hash = ?,
852
+ artifact_path = ?,
853
+ last_error = NULL,
854
+ last_model = ?,
855
+ last_attempted_ts = ?
856
+ WHERE document_id = ? AND page_number = ? AND stage = ?
857
+ """,
858
+ (
859
+ content_hash,
860
+ str(artifact_path) if artifact_path is not None else None,
861
+ model_name,
862
+ now,
863
+ document_id,
864
+ page_number,
865
+ stage,
866
+ ),
867
+ )
868
+ if attempt_index is not None:
869
+ conn.execute(
870
+ """
871
+ UPDATE model_attempts
872
+ SET status = 'completed', error_message = NULL
873
+ WHERE document_id = ? AND page_number = ? AND stage = ? AND attempt_index = ?
874
+ """,
875
+ (document_id, page_number, stage, attempt_index),
876
+ )
877
+
878
+ def record_page_failed(
879
+ self,
880
+ *,
881
+ document_id: str,
882
+ page_number: int,
883
+ stage: str,
884
+ content_hash: str | None,
885
+ error_message: str,
886
+ model_name: str | None,
887
+ attempt_index: int | None = None,
888
+ ) -> None:
889
+ now = time.time()
890
+ with self._session() as conn:
891
+ conn.execute(
892
+ """
893
+ UPDATE page_state
894
+ SET status = 'failed',
895
+ content_hash = ?,
896
+ last_error = ?,
897
+ last_model = ?,
898
+ last_attempted_ts = ?
899
+ WHERE document_id = ? AND page_number = ? AND stage = ?
900
+ """,
901
+ (
902
+ content_hash,
903
+ error_message,
904
+ model_name,
905
+ now,
906
+ document_id,
907
+ page_number,
908
+ stage,
909
+ ),
910
+ )
911
+ if attempt_index is not None:
912
+ conn.execute(
913
+ """
914
+ UPDATE model_attempts
915
+ SET status = 'failed', error_message = ?
916
+ WHERE document_id = ? AND page_number = ? AND stage = ? AND attempt_index = ?
917
+ """,
918
+ (error_message, document_id, page_number, stage, attempt_index),
919
+ )
920
+
921
+ def mark_document_completed(self, *, document_id: str, is_completed: bool) -> None:
922
+ with self._session() as conn:
923
+ conn.execute(
924
+ """
925
+ UPDATE document_state
926
+ SET is_completed = ?, last_updated_ts = ?
927
+ WHERE document_id = ?
928
+ """,
929
+ (1 if is_completed else 0, time.time(), document_id),
930
+ )
931
+
932
+ def refresh_document_completion(self, *, document_id: str) -> bool:
933
+ with self._session() as conn:
934
+ row = conn.execute(
935
+ """
936
+ SELECT total_pages
937
+ FROM document_state
938
+ WHERE document_id = ?
939
+ """,
940
+ (document_id,),
941
+ ).fetchone()
942
+ total_pages = int(row["total_pages"]) if row is not None else 0
943
+ completed = conn.execute(
944
+ """
945
+ SELECT COUNT(*) AS count
946
+ FROM page_state
947
+ WHERE document_id = ? AND stage = 'ocr' AND status = 'completed'
948
+ """,
949
+ (document_id,),
950
+ ).fetchone()
951
+ is_completed = total_pages > 0 and int(completed["count"]) == total_pages
952
+ self.mark_document_completed(document_id=document_id, is_completed=is_completed)
953
+ return is_completed
954
+
955
+ def list_model_attempts(self, *, document_id: str, page_number: int, stage: str = "ocr") -> list[dict[str, Any]]:
956
+ with self._session() as conn:
957
+ rows = conn.execute(
958
+ """
959
+ SELECT attempt_index, model_name, status, error_message, attempted_ts
960
+ FROM model_attempts
961
+ WHERE document_id = ? AND page_number = ? AND stage = ?
962
+ ORDER BY attempt_index
963
+ """,
964
+ (document_id, page_number, stage),
965
+ ).fetchall()
966
+ return [dict(row) for row in rows]
967
+
968
+ def export_progress_payload(
969
+ self,
970
+ *,
971
+ document_id: str,
972
+ title: str,
973
+ source_kind: str,
974
+ total_pages: int,
975
+ ) -> OCRWorkflowProgress:
976
+ pages: dict[str, OCRPageProgress] = {}
977
+ with self._session() as conn:
978
+ rows = conn.execute(
979
+ """
980
+ SELECT page_number, content_hash, artifact_path, status
981
+ FROM page_state
982
+ WHERE document_id = ? AND stage = 'ocr'
983
+ ORDER BY page_number
984
+ """,
985
+ (document_id,),
986
+ ).fetchall()
987
+ for row in rows:
988
+ if row["artifact_path"] is None:
989
+ continue
990
+ pages[str(int(row["page_number"]))] = OCRPageProgress(
991
+ page_number=int(row["page_number"]),
992
+ image_path="",
993
+ image_sha256=row["content_hash"] or "",
994
+ json_path=str(row["artifact_path"]),
995
+ status=str(row["status"]),
996
+ )
997
+ return OCRWorkflowProgress(
998
+ document_id=document_id,
999
+ title=title,
1000
+ source_kind=source_kind,
1001
+ total_pages=total_pages,
1002
+ pages=pages,
1003
+ )
1004
+
1005
+ def read_document_completed(self, *, document_id: str) -> bool:
1006
+ with self._session() as conn:
1007
+ row = conn.execute(
1008
+ "SELECT is_completed FROM document_state WHERE document_id = ?",
1009
+ (document_id,),
1010
+ ).fetchone()
1011
+ return bool(row is not None and int(row["is_completed"]) == 1)
1012
+
1013
+
1014
+ def _scan_rendered_pages(rendered_dir: Path) -> list[tuple[int, Path]]:
1015
+ pairs: list[tuple[int, Path]] = []
1016
+ if not rendered_dir.exists():
1017
+ return pairs
1018
+ for path in sorted(rendered_dir.glob("page_*")):
1019
+ stem = path.stem
1020
+ suffix = stem.rsplit("_", 1)
1021
+ if len(suffix) != 2 or not suffix[1].isdigit():
1022
+ continue
1023
+ pairs.append((int(suffix[1]), path))
1024
+ return pairs
1025
+
1026
+
1027
+ def _scan_legacy_json_pages(legacy_dir: Path) -> list[tuple[int, Path]]:
1028
+ pairs: list[tuple[int, Path]] = []
1029
+ if not legacy_dir.exists():
1030
+ return pairs
1031
+ for path in sorted(legacy_dir.glob("page_*.json")):
1032
+ stem = path.stem
1033
+ suffix = stem.rsplit("_", 1)
1034
+ if len(suffix) != 2 or not suffix[1].isdigit():
1035
+ continue
1036
+ pairs.append((int(suffix[1]), path))
1037
+ return pairs
1038
+
1039
+
1040
+ def _find_matching_render_hash(rendered_dir: Path, legacy_dir: Path, page_number: int) -> str | None:
1041
+ candidates = list(rendered_dir.glob(f"page_{page_number}.*")) + list(legacy_dir.glob(f"page_{page_number}.*"))
1042
+ for candidate in candidates:
1043
+ if candidate.suffix.lower() == ".json" or not candidate.exists():
1044
+ continue
1045
+ return _sha256_file(candidate)
1046
+ return None
1047
+
1048
+
1049
+ def _sha256_file(path: Path) -> str:
1050
+ hasher = hashlib.sha256()
1051
+ with path.open("rb") as handle:
1052
+ while True:
1053
+ chunk = handle.read(1024 * 1024)
1054
+ if not chunk:
1055
+ break
1056
+ hasher.update(chunk)
1057
+ return hasher.hexdigest()
1058
+
1059
+
1060
+ def _image_to_data_url(path: Path) -> str:
1061
+ encoded = base64.b64encode(path.read_bytes()).decode("ascii")
1062
+ suffix = path.suffix.lower()
1063
+ mime = "image/png"
1064
+ if suffix in {".jpg", ".jpeg"}:
1065
+ mime = "image/jpeg"
1066
+ elif suffix == ".webp":
1067
+ mime = "image/webp"
1068
+ return f"data:{mime};base64,{encoded}"
1069
+
1070
+
1071
+ def _progress_bar(done: int, total: int, width: int = 20) -> str:
1072
+ if total <= 0:
1073
+ return "." * width
1074
+ filled = min(width, max(0, round((done / total) * width)))
1075
+ return ("#" * filled) + ("." * (width - filled))
1076
+
1077
+
1078
+ def _materialize_image_payloads(image_payloads: Sequence[OCRImagePayload], rendered_dir: Path) -> list[tuple[int, Path]]:
1079
+ rendered_dir.mkdir(parents=True, exist_ok=True)
1080
+ resolved: list[tuple[int, Path]] = []
1081
+ for index, payload in enumerate(image_payloads, start=1):
1082
+ page_number = int(payload.page_number or index)
1083
+ suffix = ".png"
1084
+ if payload.image_path:
1085
+ src = Path(payload.image_path)
1086
+ suffix = src.suffix or ".png"
1087
+ dst = rendered_dir / f"page_{page_number}{suffix}"
1088
+ if src.resolve() != dst.resolve():
1089
+ shutil.copy2(src, dst)
1090
+ else:
1091
+ dst = src
1092
+ elif payload.image_bytes_b64:
1093
+ raw = base64.b64decode(payload.image_bytes_b64)
1094
+ filename = payload.filename or f"page_{page_number}.png"
1095
+ dst = rendered_dir / filename
1096
+ dst.write_bytes(raw)
1097
+ else:
1098
+ raise ValueError("OCRImagePayload requires image_path or image_bytes_b64")
1099
+ resolved.append((page_number, dst))
1100
+ return resolved
1101
+
1102
+
1103
+ def _compute_input_fingerprint(*, page_sources: Sequence[tuple[int, Path]] | None = None, pdf_path: Path | None = None) -> str:
1104
+ payload: dict[str, Any] = {}
1105
+ if pdf_path is not None:
1106
+ payload["pdf_sha256"] = _sha256_file(pdf_path)
1107
+ if page_sources is not None:
1108
+ payload["pages"] = [
1109
+ {"page_number": page_number, "sha256": _sha256_file(Path(path))}
1110
+ for page_number, path in page_sources
1111
+ ]
1112
+ encoded = json.dumps(payload, sort_keys=True).encode("utf-8")
1113
+ return hashlib.sha256(encoded).hexdigest()
1114
+
1115
+
1116
+ def _write_progress(progress: OCRWorkflowProgress, progress_path: Path) -> None:
1117
+ progress_path.parent.mkdir(parents=True, exist_ok=True)
1118
+ progress_path.write_text(
1119
+ json.dumps(progress.model_dump(mode="json"), indent=2),
1120
+ encoding="utf-8",
1121
+ )
1122
+
1123
+
1124
+ def _sync_progress_from_state(
1125
+ *,
1126
+ state_store: OCRWorkflowStateStore,
1127
+ document_id: str,
1128
+ title: str,
1129
+ source_kind: str,
1130
+ total_pages: int,
1131
+ progress_path: Path,
1132
+ ) -> OCRWorkflowProgress:
1133
+ progress = state_store.export_progress_payload(
1134
+ document_id=document_id,
1135
+ title=title,
1136
+ source_kind=source_kind,
1137
+ total_pages=total_pages,
1138
+ )
1139
+ _write_progress(progress, progress_path)
1140
+ return progress
1141
+
1142
+
1143
+ def _render_pdf_to_images(pdf_path: Path, rendered_dir: Path) -> list[Path]:
1144
+ from pdf2image import convert_from_path
1145
+
1146
+ rendered_dir.mkdir(parents=True, exist_ok=True)
1147
+ reader = PdfReader(str(pdf_path))
1148
+ output_paths: list[Path] = []
1149
+ for page_number in range(1, len(reader.pages) + 1):
1150
+ output_path = rendered_dir / f"page_{page_number}.png"
1151
+ if not output_path.exists():
1152
+ images = convert_from_path(
1153
+ str(pdf_path),
1154
+ first_page=page_number,
1155
+ last_page=page_number,
1156
+ fmt="png",
1157
+ )
1158
+ if not images:
1159
+ raise RuntimeError(f"pdf rasterizer returned no image for page {page_number}")
1160
+ images[0].save(output_path, "PNG")
1161
+ output_paths.append(output_path)
1162
+ return output_paths
1163
+
1164
+
1165
+ def _coerce_ocr_response(payload: Any) -> OCRClusterResponse:
1166
+ """Coerce a structured-output payload into the OCR model."""
1167
+ if isinstance(payload, OCRClusterResponse):
1168
+ return payload
1169
+ if isinstance(payload, dict):
1170
+ parsed = payload.get("parsed", payload)
1171
+ if parsed is not None:
1172
+ return OCRClusterResponse.model_validate(parsed)
1173
+ raw = payload.get("raw")
1174
+ if raw is not None:
1175
+ content = getattr(raw, "content", raw)
1176
+ if isinstance(content, str):
1177
+ return OCRClusterResponse.model_validate(json.loads(content))
1178
+ if isinstance(content, list):
1179
+ text = "".join(part.get("text", "") for part in content if isinstance(part, dict))
1180
+ if text.strip():
1181
+ return OCRClusterResponse.model_validate(json.loads(text))
1182
+ return OCRClusterResponse.model_validate(payload)
1183
+
1184
+
1185
+ def _extract_message_text(raw: Any) -> str:
1186
+ """Extract plain text from LangChain raw message content."""
1187
+ content = getattr(raw, "content", raw)
1188
+ if isinstance(content, str):
1189
+ return content.strip()
1190
+ if isinstance(content, list):
1191
+ parts: list[str] = []
1192
+ for item in content:
1193
+ if isinstance(item, str):
1194
+ parts.append(item)
1195
+ continue
1196
+ if isinstance(item, dict):
1197
+ text = item.get("text")
1198
+ if isinstance(text, str):
1199
+ parts.append(text)
1200
+ return "\n".join(part for part in parts if part).strip()
1201
+ return str(content).strip()
1202
+
1203
+
1204
+ def _minimal_ocr_response_from_text(*, text: str, page_number: int, image_path: Path) -> OCRClusterResponse:
1205
+ """Create a coarse page-wide OCR result when only raw text is available."""
1206
+ normalized_text = text.strip()
1207
+ with Image.open(image_path) as image:
1208
+ width, height = image.size
1209
+ if not normalized_text or normalized_text == "{}":
1210
+ return OCRClusterResponse(
1211
+ OCR_text_clusters=[],
1212
+ non_text_objects=[],
1213
+ is_empty_page=True,
1214
+ printed_page_number=str(page_number),
1215
+ meaningful_ordering=[],
1216
+ page_x_min=0.0,
1217
+ page_x_max=float(width),
1218
+ page_y_min=0.0,
1219
+ page_y_max=float(height),
1220
+ estimated_rotation_degrees=0.0,
1221
+ incomplete_words_on_edge=False,
1222
+ incomplete_text=False,
1223
+ data_loss_likelihood=0.0,
1224
+ scan_quality="medium",
1225
+ contains_table=False,
1226
+ )
1227
+ return OCRClusterResponse(
1228
+ OCR_text_clusters=[
1229
+ {
1230
+ "text": normalized_text,
1231
+ "bb_x_min": 0.0,
1232
+ "bb_x_max": float(width),
1233
+ "bb_y_min": 0.0,
1234
+ "bb_y_max": float(height),
1235
+ "cluster_number": 0,
1236
+ }
1237
+ ],
1238
+ non_text_objects=[],
1239
+ is_empty_page=False,
1240
+ printed_page_number=str(page_number),
1241
+ meaningful_ordering=[0],
1242
+ page_x_min=0.0,
1243
+ page_x_max=float(width),
1244
+ page_y_min=0.0,
1245
+ page_y_max=float(height),
1246
+ estimated_rotation_degrees=0.0,
1247
+ incomplete_words_on_edge=False,
1248
+ incomplete_text=False,
1249
+ data_loss_likelihood=0.0,
1250
+ scan_quality="medium",
1251
+ contains_table=False,
1252
+ )
1253
+
1254
+
1255
+ def _run_live_ocr_page(image_path: Path, page_number: int, provider_settings: WorkflowProviderSettings) -> OCRClusterResponse:
1256
+ """Run one page image through the configured OCR provider.
1257
+
1258
+ This is the default implementation behind the OCRRunner hook. Tests can
1259
+ replace it with a fake runner, but the callable contract stays the same:
1260
+ page image in, structured OCR model out.
1261
+ """
1262
+ chat = build_chat_model_for_role("ocr", provider_settings)
1263
+ structured = chat.with_structured_output(OCRClusterResponse, include_raw=True)
1264
+ prompt = (
1265
+ "You are an OCR model for workflow ingest.\n"
1266
+ "Return structured OCR for one page image.\n"
1267
+ "Preserve reading order, cluster numbering, and non-text regions.\n"
1268
+ "Do not invent missing text. If the page is empty, mark it as empty."
1269
+ )
1270
+ response = structured.invoke(
1271
+ [
1272
+ SystemMessage(content=prompt),
1273
+ HumanMessage(
1274
+ content=[
1275
+ {"type": "text", "text": f"OCR page {page_number} and return the structured schema."},
1276
+ {"type": "image_url", "image_url": {"url": _image_to_data_url(image_path)}},
1277
+ ]
1278
+ ),
1279
+ ]
1280
+ )
1281
+ try:
1282
+ return _coerce_ocr_response(response)
1283
+ except Exception as exc: # noqa: BLE001
1284
+ if isinstance(response, dict):
1285
+ raw = response.get("raw")
1286
+ parsing_error = response.get("parsing_error")
1287
+ raw_text = _extract_message_text(raw) if raw is not None else ""
1288
+ if raw_text:
1289
+ _LOGGER.info(
1290
+ "ocr structured parse fallback | provider=%s model=%s page=%s | parsing_error=%s",
1291
+ provider_settings.ocr.provider,
1292
+ provider_settings.ocr.model,
1293
+ page_number,
1294
+ parsing_error,
1295
+ )
1296
+ return _minimal_ocr_response_from_text(
1297
+ text=raw_text,
1298
+ page_number=page_number,
1299
+ image_path=image_path,
1300
+ )
1301
+ raise RuntimeError(
1302
+ f"ocr provider returned an unreadable response for page {page_number}: {exc}"
1303
+ ) from exc
1304
+
1305
+
1306
+ def _legacy_folder_name(*, document_id: str, pdf_path: Path | None) -> str:
1307
+ if pdf_path is not None:
1308
+ return pdf_path.name
1309
+ return document_id
1310
+
1311
+
1312
+ def _write_summary(
1313
+ artifacts: OCRWorkflowArtifacts,
1314
+ *,
1315
+ document_id: str,
1316
+ provider_settings: WorkflowProviderSettings,
1317
+ state_store: OCRWorkflowStateStore,
1318
+ ) -> None:
1319
+ payload: dict[str, Any] = {
1320
+ "document_id": document_id,
1321
+ "ocr_provider": provider_settings.ocr.provider,
1322
+ "ocr_model": provider_settings.ocr.model,
1323
+ "legacy_dir": str(artifacts.legacy_dir),
1324
+ "rendered_dir": str(artifacts.rendered_dir),
1325
+ "state_db_path": str(artifacts.state_db_path),
1326
+ "progress_path": str(artifacts.progress_path),
1327
+ "document_completed": state_store.read_document_completed(document_id=document_id),
1328
+ "completed_pages": artifacts.completed_pages,
1329
+ "reused_pages": artifacts.reused_pages,
1330
+ "model_attempts": {
1331
+ str(page_number): state_store.list_model_attempts(document_id=document_id, page_number=page_number)
1332
+ for page_number in artifacts.completed_pages
1333
+ },
1334
+ "page_count": len(artifacts.ocr_pages),
1335
+ }
1336
+ artifacts.summary_path.write_text(json.dumps(payload, indent=2), encoding="utf-8")
1337
+
1338
+
1339
+ def _emit_ocr_event(probe: WorkflowProbe | None, kind: str, /, **payload: Any) -> None:
1340
+ emit_probe_event(probe, kind, **payload)
1341
+
1342
+
1343
+ def prepare_ocr_workflow_input(
1344
+ *,
1345
+ document_id: str,
1346
+ title: str,
1347
+ output_dir: str | Path,
1348
+ image_payloads: Sequence[OCRImagePayload] | None = None,
1349
+ pdf_path: str | Path | None = None,
1350
+ provider_settings: WorkflowProviderSettings | None = None,
1351
+ ocr_runner: OCRRunner | None = None,
1352
+ pdf_rasterizer: PDFRasterizer | None = None,
1353
+ ocr_candidate_models: Sequence[str] | None = None,
1354
+ probe: WorkflowProbe | None = None,
1355
+ ) -> OCRWorkflowArtifacts:
1356
+ """Prepare OCR pages and normalize them into workflow ingest input.
1357
+
1358
+ The output directory is intentionally inspectable. It keeps:
1359
+
1360
+ - rendered page images
1361
+ - legacy-compatible ``page_N.json`` files
1362
+ - ``ocr-state.sqlite`` as the authoritative state store
1363
+ - a progress manifest mirrored from SQLite for quick inspection
1364
+ - a short summary file for manual debugging
1365
+
1366
+ Hook contracts:
1367
+ - ``ocr_runner`` must accept ``(image_path, page_number, provider_settings)``
1368
+ and return one structured OCR page result.
1369
+ - ``pdf_rasterizer`` must accept ``(pdf_path, rendered_dir)`` and return the
1370
+ list of rendered page image paths.
1371
+ """
1372
+
1373
+ # Require exactly one input shape so the rest of the function can stay linear.
1374
+ if bool(image_payloads) == bool(pdf_path):
1375
+ raise ValueError("provide exactly one of image_payloads or pdf_path")
1376
+
1377
+ # Load provider defaults only when the caller did not pass them explicitly.
1378
+ provider_settings = provider_settings or WorkflowProviderSettings.from_env()
1379
+
1380
+ # Record the start of OCR preparation for probes and debug traces.
1381
+ _emit_ocr_event(
1382
+ probe,
1383
+ "ocr.prepare_started",
1384
+ document_id=document_id,
1385
+ title=title,
1386
+ output_dir=str(output_dir),
1387
+ source_kind="pdf" if pdf_path is not None else "image",
1388
+ )
1389
+ # Resolve the output folder and the inspectable files we will write.
1390
+ output_dir = Path(output_dir)
1391
+ output_dir.mkdir(parents=True, exist_ok=True)
1392
+ summary_path = output_dir / "ocr-summary.json"
1393
+ progress_path = output_dir / "ocr-progress.json"
1394
+ state_db_path = output_dir / "ocr-state.sqlite"
1395
+
1396
+ # Turn the source input into a concrete page-by-page plan.
1397
+ source_plan = _resolve_ocr_source_plan(
1398
+ document_id=document_id,
1399
+ output_dir=output_dir,
1400
+ image_payloads=image_payloads,
1401
+ pdf_path=Path(pdf_path) if pdf_path is not None else None,
1402
+ pdf_rasterizer=pdf_rasterizer,
1403
+ )
1404
+
1405
+ # Open or rebuild the SQLite resume store for this exact document input.
1406
+ state_store = OCRWorkflowStateStore.open_or_rebuild(
1407
+ db_path=state_db_path,
1408
+ document_id=document_id,
1409
+ title=title,
1410
+ source_kind=source_plan.source_kind,
1411
+ total_pages=len(source_plan.page_sources),
1412
+ input_fingerprint=source_plan.input_fingerprint,
1413
+ rendered_dir=source_plan.rendered_dir,
1414
+ legacy_dir=source_plan.legacy_dir,
1415
+ progress_path=progress_path,
1416
+ )
1417
+
1418
+ # Emit a trace that the state store and inspectable folders are ready.
1419
+ _emit_ocr_event(
1420
+ probe,
1421
+ "ocr.state_ready",
1422
+ document_id=document_id,
1423
+ state_db_path=str(state_db_path),
1424
+ progress_path=str(progress_path),
1425
+ rendered_dir=str(source_plan.rendered_dir),
1426
+ legacy_dir=str(source_plan.legacy_dir),
1427
+ )
1428
+
1429
+ # Accumulate the serialized OCR pages that will later be normalized into workflow input.
1430
+ raw_pages: list[OCRPageJSON] = []
1431
+
1432
+ # Track which pages completed, were reused, or still failed after retries.
1433
+ completed_pages: list[int] = []
1434
+ reused_pages: list[int] = []
1435
+
1436
+ # Select candidate OCR models, falling back to the configured default model.
1437
+ candidate_models: list[str] = list(ocr_candidate_models or [provider_settings.ocr.model])
1438
+
1439
+ # Collect pages that still fail after exhausting every candidate model.
1440
+ page_failures: list[int] = []
1441
+
1442
+ # Bundle the reusable per-run OCR context so the per-page loop stays small.
1443
+ page_context = OCRPageProcessingContext(
1444
+ document_id=document_id,
1445
+ title=title,
1446
+ source_kind=source_plan.source_kind,
1447
+ total_pages=len(source_plan.page_sources),
1448
+ input_fingerprint=source_plan.input_fingerprint,
1449
+ candidate_models=candidate_models,
1450
+ provider_settings=provider_settings,
1451
+ ocr_page_runner=ocr_runner or _run_live_ocr_page,
1452
+ state_store=state_store,
1453
+ probe=probe,
1454
+ )
1455
+
1456
+ # Process every rendered page one-by-one.
1457
+ for done_count, (page_number, image_path) in enumerate(source_plan.page_sources, start=1):
1458
+ image_path = Path(image_path)
1459
+ image_sha256 = _sha256_file(image_path)
1460
+ page_json_path = source_plan.legacy_dir / f"page_{page_number}.json"
1461
+ page_image_path = source_plan.legacy_dir / f"page_{page_number}{image_path.suffix or '.png'}"
1462
+
1463
+ # Run the page through the resume-aware OCR processor.
1464
+ _process_ocr_page(
1465
+ context=page_context,
1466
+ done_count=done_count,
1467
+ page_number=page_number,
1468
+ image_path=image_path,
1469
+ image_sha256=image_sha256,
1470
+ page_json_path=page_json_path,
1471
+ page_image_path=page_image_path,
1472
+ raw_pages=raw_pages,
1473
+ completed_pages=completed_pages,
1474
+ reused_pages=reused_pages,
1475
+ page_failures=page_failures,
1476
+ )
1477
+
1478
+ # Mirror the SQLite state into the human-readable progress file.
1479
+ _sync_progress_from_state(
1480
+ state_store=state_store,
1481
+ document_id=document_id,
1482
+ title=title,
1483
+ source_kind=source_plan.source_kind,
1484
+ total_pages=len(source_plan.page_sources),
1485
+ progress_path=progress_path,
1486
+ )
1487
+
1488
+ # Mark the whole document complete once every page has been processed.
1489
+ state_store.refresh_document_completion(document_id=document_id)
1490
+
1491
+ # Write one final mirrored progress snapshot after completion.
1492
+ _sync_progress_from_state(
1493
+ state_store=state_store,
1494
+ document_id=document_id,
1495
+ title=title,
1496
+ source_kind=source_plan.source_kind,
1497
+ total_pages=len(source_plan.page_sources),
1498
+ progress_path=progress_path,
1499
+ )
1500
+
1501
+ # Emit a final probe event for tooling and debug traces.
1502
+ _emit_ocr_event(
1503
+ probe,
1504
+ "ocr.prepare_finished",
1505
+ document_id=document_id,
1506
+ status="failed" if page_failures else "succeeded",
1507
+ completed_pages=completed_pages,
1508
+ reused_pages=reused_pages,
1509
+ failed_pages=page_failures,
1510
+ state_db_path=str(state_db_path),
1511
+ )
1512
+
1513
+ # Fail the run if any page still has no successful OCR artifact.
1514
+ if page_failures:
1515
+ raise RuntimeError(
1516
+ f"OCR preparation incomplete for pages {page_failures}; rerun will retry failed pages."
1517
+ )
1518
+
1519
+ # Convert the serialized OCR page dicts into the workflow ingest source model.
1520
+ return _finalize_ocr_workflow_artifacts(
1521
+ document_id=document_id,
1522
+ title=title,
1523
+ raw_pages=raw_pages,
1524
+ completed_pages=completed_pages,
1525
+ reused_pages=reused_pages,
1526
+ rendered_dir=source_plan.rendered_dir,
1527
+ legacy_dir=source_plan.legacy_dir,
1528
+ state_db_path=state_db_path,
1529
+ progress_path=progress_path,
1530
+ summary_path=summary_path,
1531
+ provider_settings=provider_settings,
1532
+ state_store=state_store,
1533
+ )
1534
+
1535
+
1536
+ def run_ocr_ingest_workflow(
1537
+ *,
1538
+ document_id: str,
1539
+ title: str,
1540
+ output_dir: str | Path,
1541
+ workflow_engine: Any,
1542
+ conversation_engine: Any,
1543
+ knowledge_engine: Any | None = None,
1544
+ image_payloads: Sequence[OCRImagePayload] | None = None,
1545
+ pdf_path: str | Path | None = None,
1546
+ provider_settings: WorkflowProviderSettings | None = None,
1547
+ ocr_runner: OCRRunner | None = None,
1548
+ pdf_rasterizer: PDFRasterizer | None = None,
1549
+ ocr_candidate_models: Sequence[str] | None = None,
1550
+ deps: dict[str, Any] | None = None,
1551
+ probe: WorkflowProbe | None = None,
1552
+ ):
1553
+ """Run OCR preparation and then feed the normalized result into workflow ingest.
1554
+
1555
+ This is the high-level two-stage orchestration:
1556
+ 1. produce or reuse per-page OCR artifacts
1557
+ 2. normalize those pages and hand them to the semantic workflow parser
1558
+ """
1559
+
1560
+ from .parsing import parse_ocr_document
1561
+
1562
+ artifacts = parse_ocr_document(
1563
+ document_id=document_id,
1564
+ title=title,
1565
+ output_dir=output_dir,
1566
+ image_payloads=image_payloads,
1567
+ pdf_path=pdf_path,
1568
+ provider_settings=provider_settings,
1569
+ ocr_runner=ocr_runner,
1570
+ pdf_rasterizer=pdf_rasterizer,
1571
+ ocr_candidate_models=ocr_candidate_models,
1572
+ probe=probe,
1573
+ )
1574
+ run, bundle = run_ingest_workflow(
1575
+ inp=artifacts.workflow_input,
1576
+ workflow_engine=workflow_engine,
1577
+ conversation_engine=conversation_engine,
1578
+ knowledge_engine=knowledge_engine,
1579
+ deps=deps,
1580
+ )
1581
+ return run, bundle, artifacts