graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,1581 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
"""Workflow-first OCR ingest helpers for image and PDF sources.
|
|
4
|
+
|
|
5
|
+
This module sits at the boundary between raw OCR and the reusable workflow
|
|
6
|
+
ingest pipeline. It is responsible for three ideas that are easy to conflate:
|
|
7
|
+
|
|
8
|
+
1. **Page materialization**
|
|
9
|
+
- Accept already-rendered page images, or render a PDF into page images.
|
|
10
|
+
- Keep the page images around on disk so humans can inspect what was OCR'd.
|
|
11
|
+
|
|
12
|
+
2. **OCR execution and resume state**
|
|
13
|
+
- Run one OCR provider call per page through a pluggable callable hook.
|
|
14
|
+
- Persist page-level progress in ``ocr-state.sqlite`` so reruns can skip
|
|
15
|
+
completed pages and rebuild state if the DB is missing.
|
|
16
|
+
- Save legacy-compatible ``page_N.json`` artifacts alongside the images.
|
|
17
|
+
|
|
18
|
+
3. **Workflow normalization**
|
|
19
|
+
- Convert the serialized OCR pages into ``WorkflowIngestInput``.
|
|
20
|
+
- Hand that normalized input to the downstream semantic workflow parser.
|
|
21
|
+
|
|
22
|
+
Important concepts used throughout the file:
|
|
23
|
+
|
|
24
|
+
- ``OCRRunner``: a callable that OCRs exactly one page image and returns one
|
|
25
|
+
structured OCR page response.
|
|
26
|
+
- ``PDFRasterizer``: a callable that turns a PDF into a list of page image paths.
|
|
27
|
+
- ``OCRWorkflowStateStore``: the SQLite-backed resume store for page render/OCR
|
|
28
|
+
state.
|
|
29
|
+
- ``OCRWorkflowArtifacts``: the inspectable bundle returned after OCR prep so
|
|
30
|
+
tests and manual runs can open the generated files.
|
|
31
|
+
|
|
32
|
+
The public entrypoints are intentionally explicit so tests and manual runs can
|
|
33
|
+
inspect intermediate folders without needing to understand the legacy OCR code.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
import base64
|
|
37
|
+
import contextlib
|
|
38
|
+
import hashlib
|
|
39
|
+
import json
|
|
40
|
+
import logging
|
|
41
|
+
import sqlite3
|
|
42
|
+
import shutil
|
|
43
|
+
import time
|
|
44
|
+
from dataclasses import dataclass
|
|
45
|
+
from pathlib import Path
|
|
46
|
+
from typing import Any, Callable, Sequence
|
|
47
|
+
|
|
48
|
+
from langchain_core.messages import HumanMessage, SystemMessage
|
|
49
|
+
from pydantic import BaseModel, Field
|
|
50
|
+
from PIL import Image
|
|
51
|
+
from pypdf import PdfReader
|
|
52
|
+
|
|
53
|
+
from ..models import OCRClusterResponse, SplitPage, SplitPageMeta
|
|
54
|
+
|
|
55
|
+
from .adapters import OCRPageJSON, normalize_ocr_pages
|
|
56
|
+
from .models import WorkflowIngestInput
|
|
57
|
+
from .providers import ProviderEndpointConfig, WorkflowProviderSettings, build_chat_model_for_role
|
|
58
|
+
from .probe import WorkflowProbe, emit_probe_event
|
|
59
|
+
from .service import run_ingest_workflow
|
|
60
|
+
|
|
61
|
+
_LOGGER = logging.getLogger(__name__)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class OCRImagePayload(BaseModel):
|
|
65
|
+
"""Single OCR page input.
|
|
66
|
+
|
|
67
|
+
A payload may already exist on disk, or it can be materialized from bytes
|
|
68
|
+
into the working directory. The resulting file path is what the OCR model
|
|
69
|
+
and the legacy artifact writer operate on.
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
page_number: int | None = None
|
|
73
|
+
image_path: str | None = None
|
|
74
|
+
image_bytes_b64: str | None = None
|
|
75
|
+
filename: str | None = None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class OCRPageProgress(BaseModel):
|
|
79
|
+
page_number: int
|
|
80
|
+
image_path: str
|
|
81
|
+
image_sha256: str
|
|
82
|
+
json_path: str
|
|
83
|
+
status: str = "completed"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class OCRWorkflowProgress(BaseModel):
|
|
87
|
+
document_id: str
|
|
88
|
+
title: str
|
|
89
|
+
source_kind: str
|
|
90
|
+
total_pages: int
|
|
91
|
+
pages: dict[str, OCRPageProgress] = Field(default_factory=dict)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass(slots=True)
|
|
95
|
+
class OCRWorkflowArtifacts:
|
|
96
|
+
"""Inspectable result bundle produced by the OCR preparation phase.
|
|
97
|
+
|
|
98
|
+
This is the bridge object between the OCR world and the workflow ingest
|
|
99
|
+
world. It keeps both the normalized workflow input and the on-disk
|
|
100
|
+
artifacts that were used to produce it.
|
|
101
|
+
"""
|
|
102
|
+
workflow_input: WorkflowIngestInput
|
|
103
|
+
ocr_pages: list[OCRPageJSON]
|
|
104
|
+
legacy_dir: Path
|
|
105
|
+
rendered_dir: Path
|
|
106
|
+
state_db_path: Path
|
|
107
|
+
progress_path: Path
|
|
108
|
+
summary_path: Path
|
|
109
|
+
completed_pages: list[int]
|
|
110
|
+
reused_pages: list[int]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# Pluggable hook for "OCR one page image and return the structured OCR model".
|
|
114
|
+
# Signature: (image_path, page_number, provider_settings) -> OCRClusterResponse
|
|
115
|
+
OCRRunner = Callable[[Path, int, WorkflowProviderSettings], OCRClusterResponse]
|
|
116
|
+
# Pluggable hook for "render a PDF into page image paths inside a destination dir".
|
|
117
|
+
# Signature: (pdf_path, rendered_dir) -> list[Path]
|
|
118
|
+
PDFRasterizer = Callable[[Path, Path], list[Path]]
|
|
119
|
+
|
|
120
|
+
_OCR_STATE_SCHEMA_VERSION = 1
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@dataclass(slots=True)
|
|
124
|
+
class OCRPageStateSnapshot:
|
|
125
|
+
"""One page/stage snapshot from the SQLite resume store."""
|
|
126
|
+
page_number: int
|
|
127
|
+
stage: str
|
|
128
|
+
attempt_count: int
|
|
129
|
+
status: str
|
|
130
|
+
content_hash: str | None
|
|
131
|
+
artifact_path: str | None
|
|
132
|
+
last_error: str | None
|
|
133
|
+
last_model: str | None
|
|
134
|
+
last_attempted_ts: float | None
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass(slots=True)
|
|
138
|
+
class OCRSourcePlan:
|
|
139
|
+
"""Resolved file-system plan for one OCR ingest run.
|
|
140
|
+
|
|
141
|
+
This captures the immutable inputs needed to process a document:
|
|
142
|
+
where the pages live, where legacy JSON should be written, and how the
|
|
143
|
+
page sources should be iterated.
|
|
144
|
+
"""
|
|
145
|
+
|
|
146
|
+
pdf_source: Path | None
|
|
147
|
+
rendered_dir: Path
|
|
148
|
+
legacy_dir: Path
|
|
149
|
+
page_sources: list[tuple[int, Path]]
|
|
150
|
+
source_kind: str
|
|
151
|
+
input_fingerprint: str
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@dataclass(slots=True)
|
|
155
|
+
class OCRPageProcessingContext:
|
|
156
|
+
"""Immutable per-run OCR processing context.
|
|
157
|
+
|
|
158
|
+
This groups the values that every page in a run needs so the per-page loop
|
|
159
|
+
can stay focused on the page-specific state rather than carrying a very
|
|
160
|
+
long parameter list.
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
document_id: str
|
|
164
|
+
title: str
|
|
165
|
+
source_kind: str
|
|
166
|
+
total_pages: int
|
|
167
|
+
input_fingerprint: str
|
|
168
|
+
candidate_models: list[str]
|
|
169
|
+
provider_settings: WorkflowProviderSettings
|
|
170
|
+
ocr_page_runner: OCRRunner
|
|
171
|
+
state_store: OCRWorkflowStateStore
|
|
172
|
+
probe: WorkflowProbe | None
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _resolve_ocr_source_plan(
|
|
176
|
+
*,
|
|
177
|
+
document_id: str,
|
|
178
|
+
output_dir: Path,
|
|
179
|
+
image_payloads: Sequence[OCRImagePayload] | None,
|
|
180
|
+
pdf_path: Path | None,
|
|
181
|
+
pdf_rasterizer: PDFRasterizer | None,
|
|
182
|
+
) -> OCRSourcePlan:
|
|
183
|
+
rendered_root = output_dir / "rendered_pages"
|
|
184
|
+
legacy_root = output_dir / "legacy_split_pages"
|
|
185
|
+
|
|
186
|
+
pdf_source = Path(pdf_path) if pdf_path is not None else None
|
|
187
|
+
if pdf_source is not None:
|
|
188
|
+
# Render once per PDF so downstream OCR can work page-by-page against
|
|
189
|
+
# stable page image files and the state store can resume individual pages.
|
|
190
|
+
rendered_dir = rendered_root / pdf_source.stem
|
|
191
|
+
pdf_page_rasterizer = pdf_rasterizer or _render_pdf_to_images
|
|
192
|
+
rendered_paths = pdf_page_rasterizer(pdf_source, rendered_dir)
|
|
193
|
+
page_sources = list(enumerate(rendered_paths, start=1))
|
|
194
|
+
source_kind = "pdf"
|
|
195
|
+
input_fingerprint = _compute_input_fingerprint(pdf_path=pdf_source)
|
|
196
|
+
else:
|
|
197
|
+
rendered_dir = rendered_root / document_id
|
|
198
|
+
page_sources = _materialize_image_payloads(list(image_payloads or []), rendered_dir)
|
|
199
|
+
source_kind = "image"
|
|
200
|
+
input_fingerprint = _compute_input_fingerprint(page_sources=page_sources)
|
|
201
|
+
|
|
202
|
+
legacy_dir = legacy_root / _legacy_folder_name(document_id=document_id, pdf_path=pdf_source)
|
|
203
|
+
legacy_dir.mkdir(parents=True, exist_ok=True)
|
|
204
|
+
return OCRSourcePlan(
|
|
205
|
+
pdf_source=pdf_source,
|
|
206
|
+
rendered_dir=rendered_dir,
|
|
207
|
+
legacy_dir=legacy_dir,
|
|
208
|
+
page_sources=page_sources,
|
|
209
|
+
source_kind=source_kind,
|
|
210
|
+
input_fingerprint=input_fingerprint,
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _process_ocr_page(
|
|
215
|
+
*,
|
|
216
|
+
context: OCRPageProcessingContext,
|
|
217
|
+
done_count: int,
|
|
218
|
+
page_number: int,
|
|
219
|
+
image_path: Path,
|
|
220
|
+
image_sha256: str,
|
|
221
|
+
page_json_path: Path,
|
|
222
|
+
page_image_path: Path,
|
|
223
|
+
raw_pages: list[OCRPageJSON],
|
|
224
|
+
completed_pages: list[int],
|
|
225
|
+
reused_pages: list[int],
|
|
226
|
+
page_failures: list[int],
|
|
227
|
+
) -> None:
|
|
228
|
+
"""Process one page image, either by resuming or by running OCR models."""
|
|
229
|
+
|
|
230
|
+
state_store = context.state_store
|
|
231
|
+
document_id = context.document_id
|
|
232
|
+
title = context.title
|
|
233
|
+
source_kind = context.source_kind
|
|
234
|
+
total_pages = context.total_pages
|
|
235
|
+
probe = context.probe
|
|
236
|
+
|
|
237
|
+
state_store.ensure_document(
|
|
238
|
+
document_id=document_id,
|
|
239
|
+
title=title,
|
|
240
|
+
source_kind=source_kind,
|
|
241
|
+
total_pages=total_pages,
|
|
242
|
+
input_fingerprint=context.input_fingerprint,
|
|
243
|
+
)
|
|
244
|
+
if not state_store.should_skip_page(
|
|
245
|
+
document_id=document_id,
|
|
246
|
+
page_number=page_number,
|
|
247
|
+
stage="render",
|
|
248
|
+
content_hash=image_sha256,
|
|
249
|
+
artifact_path=image_path,
|
|
250
|
+
):
|
|
251
|
+
_emit_ocr_event(
|
|
252
|
+
probe,
|
|
253
|
+
"ocr.render_started",
|
|
254
|
+
document_id=document_id,
|
|
255
|
+
page_number=page_number,
|
|
256
|
+
image_path=str(image_path),
|
|
257
|
+
)
|
|
258
|
+
state_store.record_attempt(
|
|
259
|
+
document_id=document_id,
|
|
260
|
+
page_number=page_number,
|
|
261
|
+
stage="render",
|
|
262
|
+
content_hash=image_sha256,
|
|
263
|
+
model_name="local-materialize",
|
|
264
|
+
artifact_path=image_path,
|
|
265
|
+
)
|
|
266
|
+
state_store.record_page_completed(
|
|
267
|
+
document_id=document_id,
|
|
268
|
+
page_number=page_number,
|
|
269
|
+
stage="render",
|
|
270
|
+
content_hash=image_sha256,
|
|
271
|
+
artifact_path=image_path,
|
|
272
|
+
model_name="local-materialize",
|
|
273
|
+
attempt_index=1 if state_store.get_page_state(document_id=document_id, page_number=page_number, stage="render") else None,
|
|
274
|
+
)
|
|
275
|
+
_emit_ocr_event(
|
|
276
|
+
probe,
|
|
277
|
+
"ocr.render_completed",
|
|
278
|
+
document_id=document_id,
|
|
279
|
+
page_number=page_number,
|
|
280
|
+
image_path=str(image_path),
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
# Resume boundary: if the OCR page artifact already matches the image,
|
|
284
|
+
# reuse the serialized page instead of calling the model again.
|
|
285
|
+
if state_store.should_skip_page(
|
|
286
|
+
document_id=document_id,
|
|
287
|
+
page_number=page_number,
|
|
288
|
+
stage="ocr",
|
|
289
|
+
content_hash=image_sha256,
|
|
290
|
+
artifact_path=page_json_path,
|
|
291
|
+
):
|
|
292
|
+
split_page = SplitPage.model_validate(json.loads(page_json_path.read_text(encoding="utf-8")))
|
|
293
|
+
raw_pages.append(split_page.dump_supercede_parse())
|
|
294
|
+
completed_pages.append(page_number)
|
|
295
|
+
reused_pages.append(page_number)
|
|
296
|
+
_emit_ocr_event(
|
|
297
|
+
probe,
|
|
298
|
+
"ocr.page_reused",
|
|
299
|
+
document_id=document_id,
|
|
300
|
+
page_number=page_number,
|
|
301
|
+
page_json_path=str(page_json_path),
|
|
302
|
+
)
|
|
303
|
+
_LOGGER.info(
|
|
304
|
+
"ocr resume | %s/%s | %s | reused page %s",
|
|
305
|
+
done_count,
|
|
306
|
+
total_pages,
|
|
307
|
+
_progress_bar(done_count, total_pages),
|
|
308
|
+
page_number,
|
|
309
|
+
)
|
|
310
|
+
return
|
|
311
|
+
|
|
312
|
+
last_exc: Exception | None = None
|
|
313
|
+
page_completed = False
|
|
314
|
+
_emit_ocr_event(
|
|
315
|
+
probe,
|
|
316
|
+
"ocr.page_started",
|
|
317
|
+
document_id=document_id,
|
|
318
|
+
page_number=page_number,
|
|
319
|
+
page_json_path=str(page_json_path),
|
|
320
|
+
)
|
|
321
|
+
# Try candidate OCR models one-by-one for this page; the first grounded
|
|
322
|
+
# success wins and gets persisted as the canonical page artifact.
|
|
323
|
+
for candidate_model in context.candidate_models:
|
|
324
|
+
candidate_settings: WorkflowProviderSettings = context.provider_settings.model_copy(
|
|
325
|
+
update={
|
|
326
|
+
"ocr": context.provider_settings.ocr.model_copy(update={"model": candidate_model}),
|
|
327
|
+
}
|
|
328
|
+
)
|
|
329
|
+
_emit_ocr_event(
|
|
330
|
+
probe,
|
|
331
|
+
"ocr.candidate_started",
|
|
332
|
+
document_id=document_id,
|
|
333
|
+
page_number=page_number,
|
|
334
|
+
model_name=candidate_model,
|
|
335
|
+
)
|
|
336
|
+
attempt_index = state_store.record_attempt(
|
|
337
|
+
document_id=document_id,
|
|
338
|
+
page_number=page_number,
|
|
339
|
+
stage="ocr",
|
|
340
|
+
content_hash=image_sha256,
|
|
341
|
+
model_name=candidate_model,
|
|
342
|
+
artifact_path=page_json_path,
|
|
343
|
+
)
|
|
344
|
+
try:
|
|
345
|
+
response = context.ocr_page_runner(image_path, page_number, candidate_settings)
|
|
346
|
+
split_page: SplitPage = SplitPage(
|
|
347
|
+
pdf_page_num=page_number,
|
|
348
|
+
metadata=SplitPageMeta(
|
|
349
|
+
ocr_model_name=candidate_model,
|
|
350
|
+
ocr_datetime=0.0,
|
|
351
|
+
ocr_json_version="workflow_ingest_v1",
|
|
352
|
+
),
|
|
353
|
+
**response.model_dump(dump_format="python"),
|
|
354
|
+
)
|
|
355
|
+
shutil.copy2(image_path, page_image_path)
|
|
356
|
+
page_json_path.write_text(
|
|
357
|
+
json.dumps(split_page.dump_raw(dump_format="json"), indent=2),
|
|
358
|
+
encoding="utf-8",
|
|
359
|
+
)
|
|
360
|
+
state_store.record_page_completed(
|
|
361
|
+
document_id=document_id,
|
|
362
|
+
page_number=page_number,
|
|
363
|
+
stage="ocr",
|
|
364
|
+
content_hash=image_sha256,
|
|
365
|
+
artifact_path=page_json_path,
|
|
366
|
+
model_name=candidate_model,
|
|
367
|
+
attempt_index=attempt_index,
|
|
368
|
+
)
|
|
369
|
+
# The adapter consumes plain dict pages, not the Pydantic page
|
|
370
|
+
# model, so we serialize the page into the legacy-compatible shape here.
|
|
371
|
+
raw_pages.append(split_page.dump_supercede_parse())
|
|
372
|
+
completed_pages.append(page_number)
|
|
373
|
+
page_completed = True
|
|
374
|
+
_emit_ocr_event(
|
|
375
|
+
probe,
|
|
376
|
+
"ocr.candidate_completed",
|
|
377
|
+
document_id=document_id,
|
|
378
|
+
page_number=page_number,
|
|
379
|
+
model_name=candidate_model,
|
|
380
|
+
page_json_path=str(page_json_path),
|
|
381
|
+
)
|
|
382
|
+
_LOGGER.info(
|
|
383
|
+
"ocr progress | %s/%s | %s | completed page %s | model=%s",
|
|
384
|
+
done_count,
|
|
385
|
+
total_pages,
|
|
386
|
+
_progress_bar(done_count, total_pages),
|
|
387
|
+
page_number,
|
|
388
|
+
candidate_model,
|
|
389
|
+
)
|
|
390
|
+
break
|
|
391
|
+
except Exception as exc: # noqa: BLE001
|
|
392
|
+
last_exc = exc
|
|
393
|
+
state_store.record_page_failed(
|
|
394
|
+
document_id=document_id,
|
|
395
|
+
page_number=page_number,
|
|
396
|
+
stage="ocr",
|
|
397
|
+
content_hash=image_sha256,
|
|
398
|
+
error_message=str(exc),
|
|
399
|
+
model_name=candidate_model,
|
|
400
|
+
attempt_index=attempt_index,
|
|
401
|
+
)
|
|
402
|
+
_LOGGER.warning(
|
|
403
|
+
"ocr candidate failed | page=%s model=%s attempt=%s error=%s",
|
|
404
|
+
page_number,
|
|
405
|
+
candidate_model,
|
|
406
|
+
attempt_index,
|
|
407
|
+
exc,
|
|
408
|
+
)
|
|
409
|
+
_emit_ocr_event(
|
|
410
|
+
probe,
|
|
411
|
+
"ocr.candidate_failed",
|
|
412
|
+
document_id=document_id,
|
|
413
|
+
page_number=page_number,
|
|
414
|
+
model_name=candidate_model,
|
|
415
|
+
attempt_index=attempt_index,
|
|
416
|
+
error=str(exc),
|
|
417
|
+
)
|
|
418
|
+
if not page_completed:
|
|
419
|
+
page_failures.append(page_number)
|
|
420
|
+
_emit_ocr_event(
|
|
421
|
+
probe,
|
|
422
|
+
"ocr.page_failed",
|
|
423
|
+
document_id=document_id,
|
|
424
|
+
page_number=page_number,
|
|
425
|
+
candidate_models=list(context.candidate_models),
|
|
426
|
+
)
|
|
427
|
+
_LOGGER.warning(
|
|
428
|
+
"ocr progress | %s/%s | %s | failed page %s after %s candidate(s)",
|
|
429
|
+
done_count,
|
|
430
|
+
total_pages,
|
|
431
|
+
_progress_bar(done_count, total_pages),
|
|
432
|
+
page_number,
|
|
433
|
+
len(context.candidate_models),
|
|
434
|
+
)
|
|
435
|
+
if last_exc is not None:
|
|
436
|
+
_LOGGER.warning("last ocr error for page %s: %s", page_number, last_exc)
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def _finalize_ocr_workflow_artifacts(
|
|
440
|
+
*,
|
|
441
|
+
document_id: str,
|
|
442
|
+
title: str,
|
|
443
|
+
raw_pages: list[OCRPageJSON],
|
|
444
|
+
completed_pages: list[int],
|
|
445
|
+
reused_pages: list[int],
|
|
446
|
+
rendered_dir: Path,
|
|
447
|
+
legacy_dir: Path,
|
|
448
|
+
state_db_path: Path,
|
|
449
|
+
progress_path: Path,
|
|
450
|
+
summary_path: Path,
|
|
451
|
+
provider_settings: WorkflowProviderSettings,
|
|
452
|
+
state_store: OCRWorkflowStateStore,
|
|
453
|
+
) -> OCRWorkflowArtifacts:
|
|
454
|
+
workflow_input: WorkflowIngestInput = normalize_ocr_pages(
|
|
455
|
+
document_id=document_id,
|
|
456
|
+
title=title,
|
|
457
|
+
pages=raw_pages,
|
|
458
|
+
)
|
|
459
|
+
artifacts = OCRWorkflowArtifacts(
|
|
460
|
+
workflow_input=workflow_input,
|
|
461
|
+
ocr_pages=raw_pages,
|
|
462
|
+
legacy_dir=legacy_dir,
|
|
463
|
+
rendered_dir=rendered_dir,
|
|
464
|
+
state_db_path=state_db_path,
|
|
465
|
+
progress_path=progress_path,
|
|
466
|
+
summary_path=summary_path,
|
|
467
|
+
completed_pages=completed_pages,
|
|
468
|
+
reused_pages=reused_pages,
|
|
469
|
+
)
|
|
470
|
+
_write_summary(
|
|
471
|
+
artifacts,
|
|
472
|
+
document_id=document_id,
|
|
473
|
+
provider_settings=provider_settings,
|
|
474
|
+
state_store=state_store,
|
|
475
|
+
)
|
|
476
|
+
return artifacts
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
class OCRWorkflowStateStore:
|
|
480
|
+
"""SQLite-backed OCR/render state for one artifact root."""
|
|
481
|
+
|
|
482
|
+
def __init__(self, db_path: Path) -> None:
|
|
483
|
+
self.db_path = Path(db_path)
|
|
484
|
+
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
485
|
+
self._initialize()
|
|
486
|
+
|
|
487
|
+
@classmethod
|
|
488
|
+
def open_or_rebuild(
|
|
489
|
+
cls,
|
|
490
|
+
*,
|
|
491
|
+
db_path: Path,
|
|
492
|
+
document_id: str,
|
|
493
|
+
title: str,
|
|
494
|
+
source_kind: str,
|
|
495
|
+
total_pages: int,
|
|
496
|
+
input_fingerprint: str,
|
|
497
|
+
rendered_dir: Path,
|
|
498
|
+
legacy_dir: Path,
|
|
499
|
+
progress_path: Path,
|
|
500
|
+
) -> "OCRWorkflowStateStore":
|
|
501
|
+
store = cls(db_path)
|
|
502
|
+
rebuilt = False
|
|
503
|
+
if not db_path.exists() or store._is_empty():
|
|
504
|
+
rebuilt = store.rebuild_from_artifacts(
|
|
505
|
+
document_id=document_id,
|
|
506
|
+
title=title,
|
|
507
|
+
source_kind=source_kind,
|
|
508
|
+
total_pages=total_pages,
|
|
509
|
+
input_fingerprint=input_fingerprint,
|
|
510
|
+
rendered_dir=rendered_dir,
|
|
511
|
+
legacy_dir=legacy_dir,
|
|
512
|
+
progress_path=progress_path,
|
|
513
|
+
)
|
|
514
|
+
else:
|
|
515
|
+
store.ensure_document(
|
|
516
|
+
document_id=document_id,
|
|
517
|
+
title=title,
|
|
518
|
+
source_kind=source_kind,
|
|
519
|
+
total_pages=total_pages,
|
|
520
|
+
input_fingerprint=input_fingerprint,
|
|
521
|
+
)
|
|
522
|
+
if rebuilt:
|
|
523
|
+
_LOGGER.info("ocr state rebuilt from artifacts | db=%s", db_path)
|
|
524
|
+
return store
|
|
525
|
+
|
|
526
|
+
def _connect(self) -> sqlite3.Connection:
|
|
527
|
+
conn = sqlite3.connect(str(self.db_path), timeout=10)
|
|
528
|
+
conn.row_factory = sqlite3.Row
|
|
529
|
+
return conn
|
|
530
|
+
|
|
531
|
+
@contextlib.contextmanager
|
|
532
|
+
def _session(self):
|
|
533
|
+
conn = self._connect()
|
|
534
|
+
try:
|
|
535
|
+
yield conn
|
|
536
|
+
conn.commit()
|
|
537
|
+
finally:
|
|
538
|
+
conn.close()
|
|
539
|
+
|
|
540
|
+
def _initialize(self) -> None:
|
|
541
|
+
with self._session() as conn:
|
|
542
|
+
conn.executescript(
|
|
543
|
+
"""
|
|
544
|
+
CREATE TABLE IF NOT EXISTS document_state (
|
|
545
|
+
document_id TEXT PRIMARY KEY,
|
|
546
|
+
title TEXT NOT NULL,
|
|
547
|
+
source_kind TEXT NOT NULL,
|
|
548
|
+
input_fingerprint TEXT NOT NULL,
|
|
549
|
+
schema_version INTEGER NOT NULL,
|
|
550
|
+
total_pages INTEGER NOT NULL,
|
|
551
|
+
is_completed INTEGER NOT NULL DEFAULT 0,
|
|
552
|
+
last_updated_ts REAL NOT NULL
|
|
553
|
+
);
|
|
554
|
+
|
|
555
|
+
CREATE TABLE IF NOT EXISTS page_state (
|
|
556
|
+
document_id TEXT NOT NULL,
|
|
557
|
+
page_number INTEGER NOT NULL,
|
|
558
|
+
stage TEXT NOT NULL,
|
|
559
|
+
attempt_count INTEGER NOT NULL DEFAULT 0,
|
|
560
|
+
status TEXT NOT NULL DEFAULT 'pending',
|
|
561
|
+
content_hash TEXT,
|
|
562
|
+
artifact_path TEXT,
|
|
563
|
+
last_error TEXT,
|
|
564
|
+
last_model TEXT,
|
|
565
|
+
last_attempted_ts REAL,
|
|
566
|
+
PRIMARY KEY (document_id, page_number, stage)
|
|
567
|
+
);
|
|
568
|
+
|
|
569
|
+
CREATE TABLE IF NOT EXISTS model_attempts (
|
|
570
|
+
document_id TEXT NOT NULL,
|
|
571
|
+
page_number INTEGER NOT NULL,
|
|
572
|
+
stage TEXT NOT NULL,
|
|
573
|
+
attempt_index INTEGER NOT NULL,
|
|
574
|
+
model_name TEXT NOT NULL,
|
|
575
|
+
status TEXT NOT NULL,
|
|
576
|
+
error_message TEXT,
|
|
577
|
+
attempted_ts REAL NOT NULL,
|
|
578
|
+
PRIMARY KEY (document_id, page_number, stage, attempt_index)
|
|
579
|
+
);
|
|
580
|
+
"""
|
|
581
|
+
)
|
|
582
|
+
|
|
583
|
+
def _is_empty(self) -> bool:
|
|
584
|
+
with self._session() as conn:
|
|
585
|
+
row = conn.execute("SELECT COUNT(*) AS count FROM document_state").fetchone()
|
|
586
|
+
return bool(row is None or int(row["count"]) == 0)
|
|
587
|
+
|
|
588
|
+
def ensure_document(
|
|
589
|
+
self,
|
|
590
|
+
*,
|
|
591
|
+
document_id: str,
|
|
592
|
+
title: str,
|
|
593
|
+
source_kind: str,
|
|
594
|
+
total_pages: int,
|
|
595
|
+
input_fingerprint: str,
|
|
596
|
+
) -> None:
|
|
597
|
+
now = time.time()
|
|
598
|
+
with self._session() as conn:
|
|
599
|
+
existing = conn.execute(
|
|
600
|
+
"SELECT input_fingerprint FROM document_state WHERE document_id = ?",
|
|
601
|
+
(document_id,),
|
|
602
|
+
).fetchone()
|
|
603
|
+
if existing is not None and str(existing["input_fingerprint"]) != input_fingerprint:
|
|
604
|
+
conn.execute("DELETE FROM model_attempts WHERE document_id = ?", (document_id,))
|
|
605
|
+
conn.execute("DELETE FROM page_state WHERE document_id = ?", (document_id,))
|
|
606
|
+
conn.execute("DELETE FROM document_state WHERE document_id = ?", (document_id,))
|
|
607
|
+
conn.execute(
|
|
608
|
+
"""
|
|
609
|
+
INSERT INTO document_state (
|
|
610
|
+
document_id, title, source_kind, input_fingerprint,
|
|
611
|
+
schema_version, total_pages, is_completed, last_updated_ts
|
|
612
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
613
|
+
ON CONFLICT(document_id) DO UPDATE SET
|
|
614
|
+
title = excluded.title,
|
|
615
|
+
source_kind = excluded.source_kind,
|
|
616
|
+
input_fingerprint = excluded.input_fingerprint,
|
|
617
|
+
schema_version = excluded.schema_version,
|
|
618
|
+
total_pages = excluded.total_pages,
|
|
619
|
+
last_updated_ts = excluded.last_updated_ts
|
|
620
|
+
""",
|
|
621
|
+
(
|
|
622
|
+
document_id,
|
|
623
|
+
title,
|
|
624
|
+
source_kind,
|
|
625
|
+
input_fingerprint,
|
|
626
|
+
_OCR_STATE_SCHEMA_VERSION,
|
|
627
|
+
total_pages,
|
|
628
|
+
0,
|
|
629
|
+
now,
|
|
630
|
+
),
|
|
631
|
+
)
|
|
632
|
+
|
|
633
|
+
def rebuild_from_artifacts(
|
|
634
|
+
self,
|
|
635
|
+
*,
|
|
636
|
+
document_id: str,
|
|
637
|
+
title: str,
|
|
638
|
+
source_kind: str,
|
|
639
|
+
total_pages: int,
|
|
640
|
+
input_fingerprint: str,
|
|
641
|
+
rendered_dir: Path,
|
|
642
|
+
legacy_dir: Path,
|
|
643
|
+
progress_path: Path,
|
|
644
|
+
) -> bool:
|
|
645
|
+
self.ensure_document(
|
|
646
|
+
document_id=document_id,
|
|
647
|
+
title=title,
|
|
648
|
+
source_kind=source_kind,
|
|
649
|
+
total_pages=total_pages,
|
|
650
|
+
input_fingerprint=input_fingerprint,
|
|
651
|
+
)
|
|
652
|
+
rebuilt_any = False
|
|
653
|
+
progress_payload: dict[str, Any] = {}
|
|
654
|
+
if progress_path.exists():
|
|
655
|
+
progress_payload = json.loads(progress_path.read_text(encoding="utf-8"))
|
|
656
|
+
progress_pages = progress_payload.get("pages", {})
|
|
657
|
+
for page_number, path in _scan_rendered_pages(rendered_dir):
|
|
658
|
+
rebuilt_any = True
|
|
659
|
+
self._upsert_page_state(
|
|
660
|
+
document_id=document_id,
|
|
661
|
+
page_number=page_number,
|
|
662
|
+
stage="render",
|
|
663
|
+
attempt_count=1,
|
|
664
|
+
status="completed",
|
|
665
|
+
content_hash=_sha256_file(path),
|
|
666
|
+
artifact_path=str(path),
|
|
667
|
+
last_error=None,
|
|
668
|
+
last_model=None,
|
|
669
|
+
)
|
|
670
|
+
for page_number, json_path in _scan_legacy_json_pages(legacy_dir):
|
|
671
|
+
content_hash = None
|
|
672
|
+
record = progress_pages.get(str(page_number))
|
|
673
|
+
if isinstance(record, dict):
|
|
674
|
+
content_hash = record.get("image_sha256")
|
|
675
|
+
if content_hash is None:
|
|
676
|
+
content_hash = _find_matching_render_hash(rendered_dir, legacy_dir, page_number)
|
|
677
|
+
rebuilt_any = True
|
|
678
|
+
self._upsert_page_state(
|
|
679
|
+
document_id=document_id,
|
|
680
|
+
page_number=page_number,
|
|
681
|
+
stage="ocr",
|
|
682
|
+
attempt_count=1,
|
|
683
|
+
status="completed",
|
|
684
|
+
content_hash=content_hash,
|
|
685
|
+
artifact_path=str(json_path),
|
|
686
|
+
last_error=None,
|
|
687
|
+
last_model=None,
|
|
688
|
+
)
|
|
689
|
+
self.refresh_document_completion(document_id=document_id)
|
|
690
|
+
return rebuilt_any
|
|
691
|
+
|
|
692
|
+
def _upsert_page_state(
|
|
693
|
+
self,
|
|
694
|
+
*,
|
|
695
|
+
document_id: str,
|
|
696
|
+
page_number: int,
|
|
697
|
+
stage: str,
|
|
698
|
+
attempt_count: int,
|
|
699
|
+
status: str,
|
|
700
|
+
content_hash: str | None,
|
|
701
|
+
artifact_path: str | None,
|
|
702
|
+
last_error: str | None,
|
|
703
|
+
last_model: str | None,
|
|
704
|
+
) -> None:
|
|
705
|
+
now = time.time()
|
|
706
|
+
with self._session() as conn:
|
|
707
|
+
conn.execute(
|
|
708
|
+
"""
|
|
709
|
+
INSERT INTO page_state (
|
|
710
|
+
document_id, page_number, stage, attempt_count, status,
|
|
711
|
+
content_hash, artifact_path, last_error, last_model, last_attempted_ts
|
|
712
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
713
|
+
ON CONFLICT(document_id, page_number, stage) DO UPDATE SET
|
|
714
|
+
attempt_count = excluded.attempt_count,
|
|
715
|
+
status = excluded.status,
|
|
716
|
+
content_hash = excluded.content_hash,
|
|
717
|
+
artifact_path = excluded.artifact_path,
|
|
718
|
+
last_error = excluded.last_error,
|
|
719
|
+
last_model = excluded.last_model,
|
|
720
|
+
last_attempted_ts = excluded.last_attempted_ts
|
|
721
|
+
""",
|
|
722
|
+
(
|
|
723
|
+
document_id,
|
|
724
|
+
page_number,
|
|
725
|
+
stage,
|
|
726
|
+
attempt_count,
|
|
727
|
+
status,
|
|
728
|
+
content_hash,
|
|
729
|
+
artifact_path,
|
|
730
|
+
last_error,
|
|
731
|
+
last_model,
|
|
732
|
+
now,
|
|
733
|
+
),
|
|
734
|
+
)
|
|
735
|
+
|
|
736
|
+
def get_page_state(self, *, document_id: str, page_number: int, stage: str) -> OCRPageStateSnapshot | None:
|
|
737
|
+
with self._session() as conn:
|
|
738
|
+
row = conn.execute(
|
|
739
|
+
"""
|
|
740
|
+
SELECT page_number, stage, attempt_count, status, content_hash, artifact_path,
|
|
741
|
+
last_error, last_model, last_attempted_ts
|
|
742
|
+
FROM page_state
|
|
743
|
+
WHERE document_id = ? AND page_number = ? AND stage = ?
|
|
744
|
+
""",
|
|
745
|
+
(document_id, page_number, stage),
|
|
746
|
+
).fetchone()
|
|
747
|
+
if row is None:
|
|
748
|
+
return None
|
|
749
|
+
return OCRPageStateSnapshot(
|
|
750
|
+
page_number=int(row["page_number"]),
|
|
751
|
+
stage=str(row["stage"]),
|
|
752
|
+
attempt_count=int(row["attempt_count"]),
|
|
753
|
+
status=str(row["status"]),
|
|
754
|
+
content_hash=row["content_hash"],
|
|
755
|
+
artifact_path=row["artifact_path"],
|
|
756
|
+
last_error=row["last_error"],
|
|
757
|
+
last_model=row["last_model"],
|
|
758
|
+
last_attempted_ts=row["last_attempted_ts"],
|
|
759
|
+
)
|
|
760
|
+
|
|
761
|
+
def should_skip_page(
|
|
762
|
+
self,
|
|
763
|
+
*,
|
|
764
|
+
document_id: str,
|
|
765
|
+
page_number: int,
|
|
766
|
+
stage: str,
|
|
767
|
+
content_hash: str,
|
|
768
|
+
artifact_path: Path | None = None,
|
|
769
|
+
) -> bool:
|
|
770
|
+
state = self.get_page_state(document_id=document_id, page_number=page_number, stage=stage)
|
|
771
|
+
if state is None or state.status != "completed" or state.content_hash != content_hash:
|
|
772
|
+
return False
|
|
773
|
+
if artifact_path is not None and not artifact_path.exists():
|
|
774
|
+
return False
|
|
775
|
+
return True
|
|
776
|
+
|
|
777
|
+
def record_attempt(
|
|
778
|
+
self,
|
|
779
|
+
*,
|
|
780
|
+
document_id: str,
|
|
781
|
+
page_number: int,
|
|
782
|
+
stage: str,
|
|
783
|
+
content_hash: str | None,
|
|
784
|
+
model_name: str,
|
|
785
|
+
artifact_path: Path | None = None,
|
|
786
|
+
) -> int:
|
|
787
|
+
now = time.time()
|
|
788
|
+
with self._session() as conn:
|
|
789
|
+
row = conn.execute(
|
|
790
|
+
"""
|
|
791
|
+
SELECT attempt_count
|
|
792
|
+
FROM page_state
|
|
793
|
+
WHERE document_id = ? AND page_number = ? AND stage = ?
|
|
794
|
+
""",
|
|
795
|
+
(document_id, page_number, stage),
|
|
796
|
+
).fetchone()
|
|
797
|
+
attempt_count = int(row["attempt_count"]) + 1 if row is not None else 1
|
|
798
|
+
conn.execute(
|
|
799
|
+
"""
|
|
800
|
+
INSERT INTO page_state (
|
|
801
|
+
document_id, page_number, stage, attempt_count, status,
|
|
802
|
+
content_hash, artifact_path, last_error, last_model, last_attempted_ts
|
|
803
|
+
) VALUES (?, ?, ?, ?, 'pending', ?, ?, NULL, ?, ?)
|
|
804
|
+
ON CONFLICT(document_id, page_number, stage) DO UPDATE SET
|
|
805
|
+
attempt_count = excluded.attempt_count,
|
|
806
|
+
status = excluded.status,
|
|
807
|
+
content_hash = excluded.content_hash,
|
|
808
|
+
artifact_path = excluded.artifact_path,
|
|
809
|
+
last_error = excluded.last_error,
|
|
810
|
+
last_model = excluded.last_model,
|
|
811
|
+
last_attempted_ts = excluded.last_attempted_ts
|
|
812
|
+
""",
|
|
813
|
+
(
|
|
814
|
+
document_id,
|
|
815
|
+
page_number,
|
|
816
|
+
stage,
|
|
817
|
+
attempt_count,
|
|
818
|
+
content_hash,
|
|
819
|
+
str(artifact_path) if artifact_path is not None else None,
|
|
820
|
+
model_name,
|
|
821
|
+
now,
|
|
822
|
+
),
|
|
823
|
+
)
|
|
824
|
+
conn.execute(
|
|
825
|
+
"""
|
|
826
|
+
INSERT INTO model_attempts (
|
|
827
|
+
document_id, page_number, stage, attempt_index, model_name, status, error_message, attempted_ts
|
|
828
|
+
) VALUES (?, ?, ?, ?, ?, 'pending', NULL, ?)
|
|
829
|
+
""",
|
|
830
|
+
(document_id, page_number, stage, attempt_count, model_name, now),
|
|
831
|
+
)
|
|
832
|
+
return attempt_count
|
|
833
|
+
|
|
834
|
+
def record_page_completed(
|
|
835
|
+
self,
|
|
836
|
+
*,
|
|
837
|
+
document_id: str,
|
|
838
|
+
page_number: int,
|
|
839
|
+
stage: str,
|
|
840
|
+
content_hash: str | None,
|
|
841
|
+
artifact_path: Path | None,
|
|
842
|
+
model_name: str | None,
|
|
843
|
+
attempt_index: int | None = None,
|
|
844
|
+
) -> None:
|
|
845
|
+
now = time.time()
|
|
846
|
+
with self._session() as conn:
|
|
847
|
+
conn.execute(
|
|
848
|
+
"""
|
|
849
|
+
UPDATE page_state
|
|
850
|
+
SET status = 'completed',
|
|
851
|
+
content_hash = ?,
|
|
852
|
+
artifact_path = ?,
|
|
853
|
+
last_error = NULL,
|
|
854
|
+
last_model = ?,
|
|
855
|
+
last_attempted_ts = ?
|
|
856
|
+
WHERE document_id = ? AND page_number = ? AND stage = ?
|
|
857
|
+
""",
|
|
858
|
+
(
|
|
859
|
+
content_hash,
|
|
860
|
+
str(artifact_path) if artifact_path is not None else None,
|
|
861
|
+
model_name,
|
|
862
|
+
now,
|
|
863
|
+
document_id,
|
|
864
|
+
page_number,
|
|
865
|
+
stage,
|
|
866
|
+
),
|
|
867
|
+
)
|
|
868
|
+
if attempt_index is not None:
|
|
869
|
+
conn.execute(
|
|
870
|
+
"""
|
|
871
|
+
UPDATE model_attempts
|
|
872
|
+
SET status = 'completed', error_message = NULL
|
|
873
|
+
WHERE document_id = ? AND page_number = ? AND stage = ? AND attempt_index = ?
|
|
874
|
+
""",
|
|
875
|
+
(document_id, page_number, stage, attempt_index),
|
|
876
|
+
)
|
|
877
|
+
|
|
878
|
+
def record_page_failed(
|
|
879
|
+
self,
|
|
880
|
+
*,
|
|
881
|
+
document_id: str,
|
|
882
|
+
page_number: int,
|
|
883
|
+
stage: str,
|
|
884
|
+
content_hash: str | None,
|
|
885
|
+
error_message: str,
|
|
886
|
+
model_name: str | None,
|
|
887
|
+
attempt_index: int | None = None,
|
|
888
|
+
) -> None:
|
|
889
|
+
now = time.time()
|
|
890
|
+
with self._session() as conn:
|
|
891
|
+
conn.execute(
|
|
892
|
+
"""
|
|
893
|
+
UPDATE page_state
|
|
894
|
+
SET status = 'failed',
|
|
895
|
+
content_hash = ?,
|
|
896
|
+
last_error = ?,
|
|
897
|
+
last_model = ?,
|
|
898
|
+
last_attempted_ts = ?
|
|
899
|
+
WHERE document_id = ? AND page_number = ? AND stage = ?
|
|
900
|
+
""",
|
|
901
|
+
(
|
|
902
|
+
content_hash,
|
|
903
|
+
error_message,
|
|
904
|
+
model_name,
|
|
905
|
+
now,
|
|
906
|
+
document_id,
|
|
907
|
+
page_number,
|
|
908
|
+
stage,
|
|
909
|
+
),
|
|
910
|
+
)
|
|
911
|
+
if attempt_index is not None:
|
|
912
|
+
conn.execute(
|
|
913
|
+
"""
|
|
914
|
+
UPDATE model_attempts
|
|
915
|
+
SET status = 'failed', error_message = ?
|
|
916
|
+
WHERE document_id = ? AND page_number = ? AND stage = ? AND attempt_index = ?
|
|
917
|
+
""",
|
|
918
|
+
(error_message, document_id, page_number, stage, attempt_index),
|
|
919
|
+
)
|
|
920
|
+
|
|
921
|
+
def mark_document_completed(self, *, document_id: str, is_completed: bool) -> None:
|
|
922
|
+
with self._session() as conn:
|
|
923
|
+
conn.execute(
|
|
924
|
+
"""
|
|
925
|
+
UPDATE document_state
|
|
926
|
+
SET is_completed = ?, last_updated_ts = ?
|
|
927
|
+
WHERE document_id = ?
|
|
928
|
+
""",
|
|
929
|
+
(1 if is_completed else 0, time.time(), document_id),
|
|
930
|
+
)
|
|
931
|
+
|
|
932
|
+
def refresh_document_completion(self, *, document_id: str) -> bool:
|
|
933
|
+
with self._session() as conn:
|
|
934
|
+
row = conn.execute(
|
|
935
|
+
"""
|
|
936
|
+
SELECT total_pages
|
|
937
|
+
FROM document_state
|
|
938
|
+
WHERE document_id = ?
|
|
939
|
+
""",
|
|
940
|
+
(document_id,),
|
|
941
|
+
).fetchone()
|
|
942
|
+
total_pages = int(row["total_pages"]) if row is not None else 0
|
|
943
|
+
completed = conn.execute(
|
|
944
|
+
"""
|
|
945
|
+
SELECT COUNT(*) AS count
|
|
946
|
+
FROM page_state
|
|
947
|
+
WHERE document_id = ? AND stage = 'ocr' AND status = 'completed'
|
|
948
|
+
""",
|
|
949
|
+
(document_id,),
|
|
950
|
+
).fetchone()
|
|
951
|
+
is_completed = total_pages > 0 and int(completed["count"]) == total_pages
|
|
952
|
+
self.mark_document_completed(document_id=document_id, is_completed=is_completed)
|
|
953
|
+
return is_completed
|
|
954
|
+
|
|
955
|
+
def list_model_attempts(self, *, document_id: str, page_number: int, stage: str = "ocr") -> list[dict[str, Any]]:
|
|
956
|
+
with self._session() as conn:
|
|
957
|
+
rows = conn.execute(
|
|
958
|
+
"""
|
|
959
|
+
SELECT attempt_index, model_name, status, error_message, attempted_ts
|
|
960
|
+
FROM model_attempts
|
|
961
|
+
WHERE document_id = ? AND page_number = ? AND stage = ?
|
|
962
|
+
ORDER BY attempt_index
|
|
963
|
+
""",
|
|
964
|
+
(document_id, page_number, stage),
|
|
965
|
+
).fetchall()
|
|
966
|
+
return [dict(row) for row in rows]
|
|
967
|
+
|
|
968
|
+
def export_progress_payload(
|
|
969
|
+
self,
|
|
970
|
+
*,
|
|
971
|
+
document_id: str,
|
|
972
|
+
title: str,
|
|
973
|
+
source_kind: str,
|
|
974
|
+
total_pages: int,
|
|
975
|
+
) -> OCRWorkflowProgress:
|
|
976
|
+
pages: dict[str, OCRPageProgress] = {}
|
|
977
|
+
with self._session() as conn:
|
|
978
|
+
rows = conn.execute(
|
|
979
|
+
"""
|
|
980
|
+
SELECT page_number, content_hash, artifact_path, status
|
|
981
|
+
FROM page_state
|
|
982
|
+
WHERE document_id = ? AND stage = 'ocr'
|
|
983
|
+
ORDER BY page_number
|
|
984
|
+
""",
|
|
985
|
+
(document_id,),
|
|
986
|
+
).fetchall()
|
|
987
|
+
for row in rows:
|
|
988
|
+
if row["artifact_path"] is None:
|
|
989
|
+
continue
|
|
990
|
+
pages[str(int(row["page_number"]))] = OCRPageProgress(
|
|
991
|
+
page_number=int(row["page_number"]),
|
|
992
|
+
image_path="",
|
|
993
|
+
image_sha256=row["content_hash"] or "",
|
|
994
|
+
json_path=str(row["artifact_path"]),
|
|
995
|
+
status=str(row["status"]),
|
|
996
|
+
)
|
|
997
|
+
return OCRWorkflowProgress(
|
|
998
|
+
document_id=document_id,
|
|
999
|
+
title=title,
|
|
1000
|
+
source_kind=source_kind,
|
|
1001
|
+
total_pages=total_pages,
|
|
1002
|
+
pages=pages,
|
|
1003
|
+
)
|
|
1004
|
+
|
|
1005
|
+
def read_document_completed(self, *, document_id: str) -> bool:
|
|
1006
|
+
with self._session() as conn:
|
|
1007
|
+
row = conn.execute(
|
|
1008
|
+
"SELECT is_completed FROM document_state WHERE document_id = ?",
|
|
1009
|
+
(document_id,),
|
|
1010
|
+
).fetchone()
|
|
1011
|
+
return bool(row is not None and int(row["is_completed"]) == 1)
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def _scan_rendered_pages(rendered_dir: Path) -> list[tuple[int, Path]]:
|
|
1015
|
+
pairs: list[tuple[int, Path]] = []
|
|
1016
|
+
if not rendered_dir.exists():
|
|
1017
|
+
return pairs
|
|
1018
|
+
for path in sorted(rendered_dir.glob("page_*")):
|
|
1019
|
+
stem = path.stem
|
|
1020
|
+
suffix = stem.rsplit("_", 1)
|
|
1021
|
+
if len(suffix) != 2 or not suffix[1].isdigit():
|
|
1022
|
+
continue
|
|
1023
|
+
pairs.append((int(suffix[1]), path))
|
|
1024
|
+
return pairs
|
|
1025
|
+
|
|
1026
|
+
|
|
1027
|
+
def _scan_legacy_json_pages(legacy_dir: Path) -> list[tuple[int, Path]]:
|
|
1028
|
+
pairs: list[tuple[int, Path]] = []
|
|
1029
|
+
if not legacy_dir.exists():
|
|
1030
|
+
return pairs
|
|
1031
|
+
for path in sorted(legacy_dir.glob("page_*.json")):
|
|
1032
|
+
stem = path.stem
|
|
1033
|
+
suffix = stem.rsplit("_", 1)
|
|
1034
|
+
if len(suffix) != 2 or not suffix[1].isdigit():
|
|
1035
|
+
continue
|
|
1036
|
+
pairs.append((int(suffix[1]), path))
|
|
1037
|
+
return pairs
|
|
1038
|
+
|
|
1039
|
+
|
|
1040
|
+
def _find_matching_render_hash(rendered_dir: Path, legacy_dir: Path, page_number: int) -> str | None:
|
|
1041
|
+
candidates = list(rendered_dir.glob(f"page_{page_number}.*")) + list(legacy_dir.glob(f"page_{page_number}.*"))
|
|
1042
|
+
for candidate in candidates:
|
|
1043
|
+
if candidate.suffix.lower() == ".json" or not candidate.exists():
|
|
1044
|
+
continue
|
|
1045
|
+
return _sha256_file(candidate)
|
|
1046
|
+
return None
|
|
1047
|
+
|
|
1048
|
+
|
|
1049
|
+
def _sha256_file(path: Path) -> str:
|
|
1050
|
+
hasher = hashlib.sha256()
|
|
1051
|
+
with path.open("rb") as handle:
|
|
1052
|
+
while True:
|
|
1053
|
+
chunk = handle.read(1024 * 1024)
|
|
1054
|
+
if not chunk:
|
|
1055
|
+
break
|
|
1056
|
+
hasher.update(chunk)
|
|
1057
|
+
return hasher.hexdigest()
|
|
1058
|
+
|
|
1059
|
+
|
|
1060
|
+
def _image_to_data_url(path: Path) -> str:
|
|
1061
|
+
encoded = base64.b64encode(path.read_bytes()).decode("ascii")
|
|
1062
|
+
suffix = path.suffix.lower()
|
|
1063
|
+
mime = "image/png"
|
|
1064
|
+
if suffix in {".jpg", ".jpeg"}:
|
|
1065
|
+
mime = "image/jpeg"
|
|
1066
|
+
elif suffix == ".webp":
|
|
1067
|
+
mime = "image/webp"
|
|
1068
|
+
return f"data:{mime};base64,{encoded}"
|
|
1069
|
+
|
|
1070
|
+
|
|
1071
|
+
def _progress_bar(done: int, total: int, width: int = 20) -> str:
|
|
1072
|
+
if total <= 0:
|
|
1073
|
+
return "." * width
|
|
1074
|
+
filled = min(width, max(0, round((done / total) * width)))
|
|
1075
|
+
return ("#" * filled) + ("." * (width - filled))
|
|
1076
|
+
|
|
1077
|
+
|
|
1078
|
+
def _materialize_image_payloads(image_payloads: Sequence[OCRImagePayload], rendered_dir: Path) -> list[tuple[int, Path]]:
|
|
1079
|
+
rendered_dir.mkdir(parents=True, exist_ok=True)
|
|
1080
|
+
resolved: list[tuple[int, Path]] = []
|
|
1081
|
+
for index, payload in enumerate(image_payloads, start=1):
|
|
1082
|
+
page_number = int(payload.page_number or index)
|
|
1083
|
+
suffix = ".png"
|
|
1084
|
+
if payload.image_path:
|
|
1085
|
+
src = Path(payload.image_path)
|
|
1086
|
+
suffix = src.suffix or ".png"
|
|
1087
|
+
dst = rendered_dir / f"page_{page_number}{suffix}"
|
|
1088
|
+
if src.resolve() != dst.resolve():
|
|
1089
|
+
shutil.copy2(src, dst)
|
|
1090
|
+
else:
|
|
1091
|
+
dst = src
|
|
1092
|
+
elif payload.image_bytes_b64:
|
|
1093
|
+
raw = base64.b64decode(payload.image_bytes_b64)
|
|
1094
|
+
filename = payload.filename or f"page_{page_number}.png"
|
|
1095
|
+
dst = rendered_dir / filename
|
|
1096
|
+
dst.write_bytes(raw)
|
|
1097
|
+
else:
|
|
1098
|
+
raise ValueError("OCRImagePayload requires image_path or image_bytes_b64")
|
|
1099
|
+
resolved.append((page_number, dst))
|
|
1100
|
+
return resolved
|
|
1101
|
+
|
|
1102
|
+
|
|
1103
|
+
def _compute_input_fingerprint(*, page_sources: Sequence[tuple[int, Path]] | None = None, pdf_path: Path | None = None) -> str:
|
|
1104
|
+
payload: dict[str, Any] = {}
|
|
1105
|
+
if pdf_path is not None:
|
|
1106
|
+
payload["pdf_sha256"] = _sha256_file(pdf_path)
|
|
1107
|
+
if page_sources is not None:
|
|
1108
|
+
payload["pages"] = [
|
|
1109
|
+
{"page_number": page_number, "sha256": _sha256_file(Path(path))}
|
|
1110
|
+
for page_number, path in page_sources
|
|
1111
|
+
]
|
|
1112
|
+
encoded = json.dumps(payload, sort_keys=True).encode("utf-8")
|
|
1113
|
+
return hashlib.sha256(encoded).hexdigest()
|
|
1114
|
+
|
|
1115
|
+
|
|
1116
|
+
def _write_progress(progress: OCRWorkflowProgress, progress_path: Path) -> None:
|
|
1117
|
+
progress_path.parent.mkdir(parents=True, exist_ok=True)
|
|
1118
|
+
progress_path.write_text(
|
|
1119
|
+
json.dumps(progress.model_dump(mode="json"), indent=2),
|
|
1120
|
+
encoding="utf-8",
|
|
1121
|
+
)
|
|
1122
|
+
|
|
1123
|
+
|
|
1124
|
+
def _sync_progress_from_state(
|
|
1125
|
+
*,
|
|
1126
|
+
state_store: OCRWorkflowStateStore,
|
|
1127
|
+
document_id: str,
|
|
1128
|
+
title: str,
|
|
1129
|
+
source_kind: str,
|
|
1130
|
+
total_pages: int,
|
|
1131
|
+
progress_path: Path,
|
|
1132
|
+
) -> OCRWorkflowProgress:
|
|
1133
|
+
progress = state_store.export_progress_payload(
|
|
1134
|
+
document_id=document_id,
|
|
1135
|
+
title=title,
|
|
1136
|
+
source_kind=source_kind,
|
|
1137
|
+
total_pages=total_pages,
|
|
1138
|
+
)
|
|
1139
|
+
_write_progress(progress, progress_path)
|
|
1140
|
+
return progress
|
|
1141
|
+
|
|
1142
|
+
|
|
1143
|
+
def _render_pdf_to_images(pdf_path: Path, rendered_dir: Path) -> list[Path]:
|
|
1144
|
+
from pdf2image import convert_from_path
|
|
1145
|
+
|
|
1146
|
+
rendered_dir.mkdir(parents=True, exist_ok=True)
|
|
1147
|
+
reader = PdfReader(str(pdf_path))
|
|
1148
|
+
output_paths: list[Path] = []
|
|
1149
|
+
for page_number in range(1, len(reader.pages) + 1):
|
|
1150
|
+
output_path = rendered_dir / f"page_{page_number}.png"
|
|
1151
|
+
if not output_path.exists():
|
|
1152
|
+
images = convert_from_path(
|
|
1153
|
+
str(pdf_path),
|
|
1154
|
+
first_page=page_number,
|
|
1155
|
+
last_page=page_number,
|
|
1156
|
+
fmt="png",
|
|
1157
|
+
)
|
|
1158
|
+
if not images:
|
|
1159
|
+
raise RuntimeError(f"pdf rasterizer returned no image for page {page_number}")
|
|
1160
|
+
images[0].save(output_path, "PNG")
|
|
1161
|
+
output_paths.append(output_path)
|
|
1162
|
+
return output_paths
|
|
1163
|
+
|
|
1164
|
+
|
|
1165
|
+
def _coerce_ocr_response(payload: Any) -> OCRClusterResponse:
|
|
1166
|
+
"""Coerce a structured-output payload into the OCR model."""
|
|
1167
|
+
if isinstance(payload, OCRClusterResponse):
|
|
1168
|
+
return payload
|
|
1169
|
+
if isinstance(payload, dict):
|
|
1170
|
+
parsed = payload.get("parsed", payload)
|
|
1171
|
+
if parsed is not None:
|
|
1172
|
+
return OCRClusterResponse.model_validate(parsed)
|
|
1173
|
+
raw = payload.get("raw")
|
|
1174
|
+
if raw is not None:
|
|
1175
|
+
content = getattr(raw, "content", raw)
|
|
1176
|
+
if isinstance(content, str):
|
|
1177
|
+
return OCRClusterResponse.model_validate(json.loads(content))
|
|
1178
|
+
if isinstance(content, list):
|
|
1179
|
+
text = "".join(part.get("text", "") for part in content if isinstance(part, dict))
|
|
1180
|
+
if text.strip():
|
|
1181
|
+
return OCRClusterResponse.model_validate(json.loads(text))
|
|
1182
|
+
return OCRClusterResponse.model_validate(payload)
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
def _extract_message_text(raw: Any) -> str:
|
|
1186
|
+
"""Extract plain text from LangChain raw message content."""
|
|
1187
|
+
content = getattr(raw, "content", raw)
|
|
1188
|
+
if isinstance(content, str):
|
|
1189
|
+
return content.strip()
|
|
1190
|
+
if isinstance(content, list):
|
|
1191
|
+
parts: list[str] = []
|
|
1192
|
+
for item in content:
|
|
1193
|
+
if isinstance(item, str):
|
|
1194
|
+
parts.append(item)
|
|
1195
|
+
continue
|
|
1196
|
+
if isinstance(item, dict):
|
|
1197
|
+
text = item.get("text")
|
|
1198
|
+
if isinstance(text, str):
|
|
1199
|
+
parts.append(text)
|
|
1200
|
+
return "\n".join(part for part in parts if part).strip()
|
|
1201
|
+
return str(content).strip()
|
|
1202
|
+
|
|
1203
|
+
|
|
1204
|
+
def _minimal_ocr_response_from_text(*, text: str, page_number: int, image_path: Path) -> OCRClusterResponse:
|
|
1205
|
+
"""Create a coarse page-wide OCR result when only raw text is available."""
|
|
1206
|
+
normalized_text = text.strip()
|
|
1207
|
+
with Image.open(image_path) as image:
|
|
1208
|
+
width, height = image.size
|
|
1209
|
+
if not normalized_text or normalized_text == "{}":
|
|
1210
|
+
return OCRClusterResponse(
|
|
1211
|
+
OCR_text_clusters=[],
|
|
1212
|
+
non_text_objects=[],
|
|
1213
|
+
is_empty_page=True,
|
|
1214
|
+
printed_page_number=str(page_number),
|
|
1215
|
+
meaningful_ordering=[],
|
|
1216
|
+
page_x_min=0.0,
|
|
1217
|
+
page_x_max=float(width),
|
|
1218
|
+
page_y_min=0.0,
|
|
1219
|
+
page_y_max=float(height),
|
|
1220
|
+
estimated_rotation_degrees=0.0,
|
|
1221
|
+
incomplete_words_on_edge=False,
|
|
1222
|
+
incomplete_text=False,
|
|
1223
|
+
data_loss_likelihood=0.0,
|
|
1224
|
+
scan_quality="medium",
|
|
1225
|
+
contains_table=False,
|
|
1226
|
+
)
|
|
1227
|
+
return OCRClusterResponse(
|
|
1228
|
+
OCR_text_clusters=[
|
|
1229
|
+
{
|
|
1230
|
+
"text": normalized_text,
|
|
1231
|
+
"bb_x_min": 0.0,
|
|
1232
|
+
"bb_x_max": float(width),
|
|
1233
|
+
"bb_y_min": 0.0,
|
|
1234
|
+
"bb_y_max": float(height),
|
|
1235
|
+
"cluster_number": 0,
|
|
1236
|
+
}
|
|
1237
|
+
],
|
|
1238
|
+
non_text_objects=[],
|
|
1239
|
+
is_empty_page=False,
|
|
1240
|
+
printed_page_number=str(page_number),
|
|
1241
|
+
meaningful_ordering=[0],
|
|
1242
|
+
page_x_min=0.0,
|
|
1243
|
+
page_x_max=float(width),
|
|
1244
|
+
page_y_min=0.0,
|
|
1245
|
+
page_y_max=float(height),
|
|
1246
|
+
estimated_rotation_degrees=0.0,
|
|
1247
|
+
incomplete_words_on_edge=False,
|
|
1248
|
+
incomplete_text=False,
|
|
1249
|
+
data_loss_likelihood=0.0,
|
|
1250
|
+
scan_quality="medium",
|
|
1251
|
+
contains_table=False,
|
|
1252
|
+
)
|
|
1253
|
+
|
|
1254
|
+
|
|
1255
|
+
def _run_live_ocr_page(image_path: Path, page_number: int, provider_settings: WorkflowProviderSettings) -> OCRClusterResponse:
|
|
1256
|
+
"""Run one page image through the configured OCR provider.
|
|
1257
|
+
|
|
1258
|
+
This is the default implementation behind the OCRRunner hook. Tests can
|
|
1259
|
+
replace it with a fake runner, but the callable contract stays the same:
|
|
1260
|
+
page image in, structured OCR model out.
|
|
1261
|
+
"""
|
|
1262
|
+
chat = build_chat_model_for_role("ocr", provider_settings)
|
|
1263
|
+
structured = chat.with_structured_output(OCRClusterResponse, include_raw=True)
|
|
1264
|
+
prompt = (
|
|
1265
|
+
"You are an OCR model for workflow ingest.\n"
|
|
1266
|
+
"Return structured OCR for one page image.\n"
|
|
1267
|
+
"Preserve reading order, cluster numbering, and non-text regions.\n"
|
|
1268
|
+
"Do not invent missing text. If the page is empty, mark it as empty."
|
|
1269
|
+
)
|
|
1270
|
+
response = structured.invoke(
|
|
1271
|
+
[
|
|
1272
|
+
SystemMessage(content=prompt),
|
|
1273
|
+
HumanMessage(
|
|
1274
|
+
content=[
|
|
1275
|
+
{"type": "text", "text": f"OCR page {page_number} and return the structured schema."},
|
|
1276
|
+
{"type": "image_url", "image_url": {"url": _image_to_data_url(image_path)}},
|
|
1277
|
+
]
|
|
1278
|
+
),
|
|
1279
|
+
]
|
|
1280
|
+
)
|
|
1281
|
+
try:
|
|
1282
|
+
return _coerce_ocr_response(response)
|
|
1283
|
+
except Exception as exc: # noqa: BLE001
|
|
1284
|
+
if isinstance(response, dict):
|
|
1285
|
+
raw = response.get("raw")
|
|
1286
|
+
parsing_error = response.get("parsing_error")
|
|
1287
|
+
raw_text = _extract_message_text(raw) if raw is not None else ""
|
|
1288
|
+
if raw_text:
|
|
1289
|
+
_LOGGER.info(
|
|
1290
|
+
"ocr structured parse fallback | provider=%s model=%s page=%s | parsing_error=%s",
|
|
1291
|
+
provider_settings.ocr.provider,
|
|
1292
|
+
provider_settings.ocr.model,
|
|
1293
|
+
page_number,
|
|
1294
|
+
parsing_error,
|
|
1295
|
+
)
|
|
1296
|
+
return _minimal_ocr_response_from_text(
|
|
1297
|
+
text=raw_text,
|
|
1298
|
+
page_number=page_number,
|
|
1299
|
+
image_path=image_path,
|
|
1300
|
+
)
|
|
1301
|
+
raise RuntimeError(
|
|
1302
|
+
f"ocr provider returned an unreadable response for page {page_number}: {exc}"
|
|
1303
|
+
) from exc
|
|
1304
|
+
|
|
1305
|
+
|
|
1306
|
+
def _legacy_folder_name(*, document_id: str, pdf_path: Path | None) -> str:
|
|
1307
|
+
if pdf_path is not None:
|
|
1308
|
+
return pdf_path.name
|
|
1309
|
+
return document_id
|
|
1310
|
+
|
|
1311
|
+
|
|
1312
|
+
def _write_summary(
|
|
1313
|
+
artifacts: OCRWorkflowArtifacts,
|
|
1314
|
+
*,
|
|
1315
|
+
document_id: str,
|
|
1316
|
+
provider_settings: WorkflowProviderSettings,
|
|
1317
|
+
state_store: OCRWorkflowStateStore,
|
|
1318
|
+
) -> None:
|
|
1319
|
+
payload: dict[str, Any] = {
|
|
1320
|
+
"document_id": document_id,
|
|
1321
|
+
"ocr_provider": provider_settings.ocr.provider,
|
|
1322
|
+
"ocr_model": provider_settings.ocr.model,
|
|
1323
|
+
"legacy_dir": str(artifacts.legacy_dir),
|
|
1324
|
+
"rendered_dir": str(artifacts.rendered_dir),
|
|
1325
|
+
"state_db_path": str(artifacts.state_db_path),
|
|
1326
|
+
"progress_path": str(artifacts.progress_path),
|
|
1327
|
+
"document_completed": state_store.read_document_completed(document_id=document_id),
|
|
1328
|
+
"completed_pages": artifacts.completed_pages,
|
|
1329
|
+
"reused_pages": artifacts.reused_pages,
|
|
1330
|
+
"model_attempts": {
|
|
1331
|
+
str(page_number): state_store.list_model_attempts(document_id=document_id, page_number=page_number)
|
|
1332
|
+
for page_number in artifacts.completed_pages
|
|
1333
|
+
},
|
|
1334
|
+
"page_count": len(artifacts.ocr_pages),
|
|
1335
|
+
}
|
|
1336
|
+
artifacts.summary_path.write_text(json.dumps(payload, indent=2), encoding="utf-8")
|
|
1337
|
+
|
|
1338
|
+
|
|
1339
|
+
def _emit_ocr_event(probe: WorkflowProbe | None, kind: str, /, **payload: Any) -> None:
|
|
1340
|
+
emit_probe_event(probe, kind, **payload)
|
|
1341
|
+
|
|
1342
|
+
|
|
1343
|
+
def prepare_ocr_workflow_input(
|
|
1344
|
+
*,
|
|
1345
|
+
document_id: str,
|
|
1346
|
+
title: str,
|
|
1347
|
+
output_dir: str | Path,
|
|
1348
|
+
image_payloads: Sequence[OCRImagePayload] | None = None,
|
|
1349
|
+
pdf_path: str | Path | None = None,
|
|
1350
|
+
provider_settings: WorkflowProviderSettings | None = None,
|
|
1351
|
+
ocr_runner: OCRRunner | None = None,
|
|
1352
|
+
pdf_rasterizer: PDFRasterizer | None = None,
|
|
1353
|
+
ocr_candidate_models: Sequence[str] | None = None,
|
|
1354
|
+
probe: WorkflowProbe | None = None,
|
|
1355
|
+
) -> OCRWorkflowArtifacts:
|
|
1356
|
+
"""Prepare OCR pages and normalize them into workflow ingest input.
|
|
1357
|
+
|
|
1358
|
+
The output directory is intentionally inspectable. It keeps:
|
|
1359
|
+
|
|
1360
|
+
- rendered page images
|
|
1361
|
+
- legacy-compatible ``page_N.json`` files
|
|
1362
|
+
- ``ocr-state.sqlite`` as the authoritative state store
|
|
1363
|
+
- a progress manifest mirrored from SQLite for quick inspection
|
|
1364
|
+
- a short summary file for manual debugging
|
|
1365
|
+
|
|
1366
|
+
Hook contracts:
|
|
1367
|
+
- ``ocr_runner`` must accept ``(image_path, page_number, provider_settings)``
|
|
1368
|
+
and return one structured OCR page result.
|
|
1369
|
+
- ``pdf_rasterizer`` must accept ``(pdf_path, rendered_dir)`` and return the
|
|
1370
|
+
list of rendered page image paths.
|
|
1371
|
+
"""
|
|
1372
|
+
|
|
1373
|
+
# Require exactly one input shape so the rest of the function can stay linear.
|
|
1374
|
+
if bool(image_payloads) == bool(pdf_path):
|
|
1375
|
+
raise ValueError("provide exactly one of image_payloads or pdf_path")
|
|
1376
|
+
|
|
1377
|
+
# Load provider defaults only when the caller did not pass them explicitly.
|
|
1378
|
+
provider_settings = provider_settings or WorkflowProviderSettings.from_env()
|
|
1379
|
+
|
|
1380
|
+
# Record the start of OCR preparation for probes and debug traces.
|
|
1381
|
+
_emit_ocr_event(
|
|
1382
|
+
probe,
|
|
1383
|
+
"ocr.prepare_started",
|
|
1384
|
+
document_id=document_id,
|
|
1385
|
+
title=title,
|
|
1386
|
+
output_dir=str(output_dir),
|
|
1387
|
+
source_kind="pdf" if pdf_path is not None else "image",
|
|
1388
|
+
)
|
|
1389
|
+
# Resolve the output folder and the inspectable files we will write.
|
|
1390
|
+
output_dir = Path(output_dir)
|
|
1391
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
1392
|
+
summary_path = output_dir / "ocr-summary.json"
|
|
1393
|
+
progress_path = output_dir / "ocr-progress.json"
|
|
1394
|
+
state_db_path = output_dir / "ocr-state.sqlite"
|
|
1395
|
+
|
|
1396
|
+
# Turn the source input into a concrete page-by-page plan.
|
|
1397
|
+
source_plan = _resolve_ocr_source_plan(
|
|
1398
|
+
document_id=document_id,
|
|
1399
|
+
output_dir=output_dir,
|
|
1400
|
+
image_payloads=image_payloads,
|
|
1401
|
+
pdf_path=Path(pdf_path) if pdf_path is not None else None,
|
|
1402
|
+
pdf_rasterizer=pdf_rasterizer,
|
|
1403
|
+
)
|
|
1404
|
+
|
|
1405
|
+
# Open or rebuild the SQLite resume store for this exact document input.
|
|
1406
|
+
state_store = OCRWorkflowStateStore.open_or_rebuild(
|
|
1407
|
+
db_path=state_db_path,
|
|
1408
|
+
document_id=document_id,
|
|
1409
|
+
title=title,
|
|
1410
|
+
source_kind=source_plan.source_kind,
|
|
1411
|
+
total_pages=len(source_plan.page_sources),
|
|
1412
|
+
input_fingerprint=source_plan.input_fingerprint,
|
|
1413
|
+
rendered_dir=source_plan.rendered_dir,
|
|
1414
|
+
legacy_dir=source_plan.legacy_dir,
|
|
1415
|
+
progress_path=progress_path,
|
|
1416
|
+
)
|
|
1417
|
+
|
|
1418
|
+
# Emit a trace that the state store and inspectable folders are ready.
|
|
1419
|
+
_emit_ocr_event(
|
|
1420
|
+
probe,
|
|
1421
|
+
"ocr.state_ready",
|
|
1422
|
+
document_id=document_id,
|
|
1423
|
+
state_db_path=str(state_db_path),
|
|
1424
|
+
progress_path=str(progress_path),
|
|
1425
|
+
rendered_dir=str(source_plan.rendered_dir),
|
|
1426
|
+
legacy_dir=str(source_plan.legacy_dir),
|
|
1427
|
+
)
|
|
1428
|
+
|
|
1429
|
+
# Accumulate the serialized OCR pages that will later be normalized into workflow input.
|
|
1430
|
+
raw_pages: list[OCRPageJSON] = []
|
|
1431
|
+
|
|
1432
|
+
# Track which pages completed, were reused, or still failed after retries.
|
|
1433
|
+
completed_pages: list[int] = []
|
|
1434
|
+
reused_pages: list[int] = []
|
|
1435
|
+
|
|
1436
|
+
# Select candidate OCR models, falling back to the configured default model.
|
|
1437
|
+
candidate_models: list[str] = list(ocr_candidate_models or [provider_settings.ocr.model])
|
|
1438
|
+
|
|
1439
|
+
# Collect pages that still fail after exhausting every candidate model.
|
|
1440
|
+
page_failures: list[int] = []
|
|
1441
|
+
|
|
1442
|
+
# Bundle the reusable per-run OCR context so the per-page loop stays small.
|
|
1443
|
+
page_context = OCRPageProcessingContext(
|
|
1444
|
+
document_id=document_id,
|
|
1445
|
+
title=title,
|
|
1446
|
+
source_kind=source_plan.source_kind,
|
|
1447
|
+
total_pages=len(source_plan.page_sources),
|
|
1448
|
+
input_fingerprint=source_plan.input_fingerprint,
|
|
1449
|
+
candidate_models=candidate_models,
|
|
1450
|
+
provider_settings=provider_settings,
|
|
1451
|
+
ocr_page_runner=ocr_runner or _run_live_ocr_page,
|
|
1452
|
+
state_store=state_store,
|
|
1453
|
+
probe=probe,
|
|
1454
|
+
)
|
|
1455
|
+
|
|
1456
|
+
# Process every rendered page one-by-one.
|
|
1457
|
+
for done_count, (page_number, image_path) in enumerate(source_plan.page_sources, start=1):
|
|
1458
|
+
image_path = Path(image_path)
|
|
1459
|
+
image_sha256 = _sha256_file(image_path)
|
|
1460
|
+
page_json_path = source_plan.legacy_dir / f"page_{page_number}.json"
|
|
1461
|
+
page_image_path = source_plan.legacy_dir / f"page_{page_number}{image_path.suffix or '.png'}"
|
|
1462
|
+
|
|
1463
|
+
# Run the page through the resume-aware OCR processor.
|
|
1464
|
+
_process_ocr_page(
|
|
1465
|
+
context=page_context,
|
|
1466
|
+
done_count=done_count,
|
|
1467
|
+
page_number=page_number,
|
|
1468
|
+
image_path=image_path,
|
|
1469
|
+
image_sha256=image_sha256,
|
|
1470
|
+
page_json_path=page_json_path,
|
|
1471
|
+
page_image_path=page_image_path,
|
|
1472
|
+
raw_pages=raw_pages,
|
|
1473
|
+
completed_pages=completed_pages,
|
|
1474
|
+
reused_pages=reused_pages,
|
|
1475
|
+
page_failures=page_failures,
|
|
1476
|
+
)
|
|
1477
|
+
|
|
1478
|
+
# Mirror the SQLite state into the human-readable progress file.
|
|
1479
|
+
_sync_progress_from_state(
|
|
1480
|
+
state_store=state_store,
|
|
1481
|
+
document_id=document_id,
|
|
1482
|
+
title=title,
|
|
1483
|
+
source_kind=source_plan.source_kind,
|
|
1484
|
+
total_pages=len(source_plan.page_sources),
|
|
1485
|
+
progress_path=progress_path,
|
|
1486
|
+
)
|
|
1487
|
+
|
|
1488
|
+
# Mark the whole document complete once every page has been processed.
|
|
1489
|
+
state_store.refresh_document_completion(document_id=document_id)
|
|
1490
|
+
|
|
1491
|
+
# Write one final mirrored progress snapshot after completion.
|
|
1492
|
+
_sync_progress_from_state(
|
|
1493
|
+
state_store=state_store,
|
|
1494
|
+
document_id=document_id,
|
|
1495
|
+
title=title,
|
|
1496
|
+
source_kind=source_plan.source_kind,
|
|
1497
|
+
total_pages=len(source_plan.page_sources),
|
|
1498
|
+
progress_path=progress_path,
|
|
1499
|
+
)
|
|
1500
|
+
|
|
1501
|
+
# Emit a final probe event for tooling and debug traces.
|
|
1502
|
+
_emit_ocr_event(
|
|
1503
|
+
probe,
|
|
1504
|
+
"ocr.prepare_finished",
|
|
1505
|
+
document_id=document_id,
|
|
1506
|
+
status="failed" if page_failures else "succeeded",
|
|
1507
|
+
completed_pages=completed_pages,
|
|
1508
|
+
reused_pages=reused_pages,
|
|
1509
|
+
failed_pages=page_failures,
|
|
1510
|
+
state_db_path=str(state_db_path),
|
|
1511
|
+
)
|
|
1512
|
+
|
|
1513
|
+
# Fail the run if any page still has no successful OCR artifact.
|
|
1514
|
+
if page_failures:
|
|
1515
|
+
raise RuntimeError(
|
|
1516
|
+
f"OCR preparation incomplete for pages {page_failures}; rerun will retry failed pages."
|
|
1517
|
+
)
|
|
1518
|
+
|
|
1519
|
+
# Convert the serialized OCR page dicts into the workflow ingest source model.
|
|
1520
|
+
return _finalize_ocr_workflow_artifacts(
|
|
1521
|
+
document_id=document_id,
|
|
1522
|
+
title=title,
|
|
1523
|
+
raw_pages=raw_pages,
|
|
1524
|
+
completed_pages=completed_pages,
|
|
1525
|
+
reused_pages=reused_pages,
|
|
1526
|
+
rendered_dir=source_plan.rendered_dir,
|
|
1527
|
+
legacy_dir=source_plan.legacy_dir,
|
|
1528
|
+
state_db_path=state_db_path,
|
|
1529
|
+
progress_path=progress_path,
|
|
1530
|
+
summary_path=summary_path,
|
|
1531
|
+
provider_settings=provider_settings,
|
|
1532
|
+
state_store=state_store,
|
|
1533
|
+
)
|
|
1534
|
+
|
|
1535
|
+
|
|
1536
|
+
def run_ocr_ingest_workflow(
|
|
1537
|
+
*,
|
|
1538
|
+
document_id: str,
|
|
1539
|
+
title: str,
|
|
1540
|
+
output_dir: str | Path,
|
|
1541
|
+
workflow_engine: Any,
|
|
1542
|
+
conversation_engine: Any,
|
|
1543
|
+
knowledge_engine: Any | None = None,
|
|
1544
|
+
image_payloads: Sequence[OCRImagePayload] | None = None,
|
|
1545
|
+
pdf_path: str | Path | None = None,
|
|
1546
|
+
provider_settings: WorkflowProviderSettings | None = None,
|
|
1547
|
+
ocr_runner: OCRRunner | None = None,
|
|
1548
|
+
pdf_rasterizer: PDFRasterizer | None = None,
|
|
1549
|
+
ocr_candidate_models: Sequence[str] | None = None,
|
|
1550
|
+
deps: dict[str, Any] | None = None,
|
|
1551
|
+
probe: WorkflowProbe | None = None,
|
|
1552
|
+
):
|
|
1553
|
+
"""Run OCR preparation and then feed the normalized result into workflow ingest.
|
|
1554
|
+
|
|
1555
|
+
This is the high-level two-stage orchestration:
|
|
1556
|
+
1. produce or reuse per-page OCR artifacts
|
|
1557
|
+
2. normalize those pages and hand them to the semantic workflow parser
|
|
1558
|
+
"""
|
|
1559
|
+
|
|
1560
|
+
from .parsing import parse_ocr_document
|
|
1561
|
+
|
|
1562
|
+
artifacts = parse_ocr_document(
|
|
1563
|
+
document_id=document_id,
|
|
1564
|
+
title=title,
|
|
1565
|
+
output_dir=output_dir,
|
|
1566
|
+
image_payloads=image_payloads,
|
|
1567
|
+
pdf_path=pdf_path,
|
|
1568
|
+
provider_settings=provider_settings,
|
|
1569
|
+
ocr_runner=ocr_runner,
|
|
1570
|
+
pdf_rasterizer=pdf_rasterizer,
|
|
1571
|
+
ocr_candidate_models=ocr_candidate_models,
|
|
1572
|
+
probe=probe,
|
|
1573
|
+
)
|
|
1574
|
+
run, bundle = run_ingest_workflow(
|
|
1575
|
+
inp=artifacts.workflow_input,
|
|
1576
|
+
workflow_engine=workflow_engine,
|
|
1577
|
+
conversation_engine=conversation_engine,
|
|
1578
|
+
knowledge_engine=knowledge_engine,
|
|
1579
|
+
deps=deps,
|
|
1580
|
+
)
|
|
1581
|
+
return run, bundle, artifacts
|