graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,546 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
"""Reusable workflow runners for CLI and higher-level orchestration.
|
|
4
|
+
|
|
5
|
+
These helpers compose the existing workflow ingest primitives without changing
|
|
6
|
+
their core behavior. The CLI calls into this module, but test code and other
|
|
7
|
+
workflow code can also reuse the same wrappers directly.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
from contextlib import contextmanager
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, Iterable, Literal, Sequence
|
|
16
|
+
|
|
17
|
+
from kogwistar.engine_core.models import Edge, Node
|
|
18
|
+
|
|
19
|
+
from .demo_harness import DemoHarnessConfig, run_demo_harness
|
|
20
|
+
from .ocr_pipeline import OCRImagePayload, OCRWorkflowArtifacts, prepare_ocr_workflow_input, run_ocr_ingest_workflow
|
|
21
|
+
from .page_index import PageIndexParseResult, PageIndexSourceFormat
|
|
22
|
+
from .parsing import parse_ocr_document, parse_page_index_document, parse_tree_document
|
|
23
|
+
from .probe import WorkflowProbe, emit_probe_event
|
|
24
|
+
from .providers import WorkflowProviderSettings
|
|
25
|
+
from .parser_core import default_parse_semantic_fn
|
|
26
|
+
from .service import build_default_engines, run_ingest_workflow
|
|
27
|
+
from .semantics import HydratedTextPointer, SemanticNode
|
|
28
|
+
|
|
29
|
+
SupportedOCRInput = Literal["image", "pdf"]
|
|
30
|
+
SupportedPageIndexInput = Literal["text", "markdown"]
|
|
31
|
+
|
|
32
|
+
OCR_FILE_SUFFIXES = {".png", ".jpg", ".jpeg", ".webp", ".bmp", ".tif", ".tiff", ".pdf"}
|
|
33
|
+
PAGE_INDEX_SUFFIXES = {".txt", ".md"}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(slots=True)
|
|
37
|
+
class WorkflowCommandResult:
|
|
38
|
+
kind: str
|
|
39
|
+
input_path: Path
|
|
40
|
+
output_dir: Path
|
|
41
|
+
status: str | None = None
|
|
42
|
+
probe_path: Path | None = None
|
|
43
|
+
summary_path: Path | None = None
|
|
44
|
+
extra: dict[str, Any] = field(default_factory=dict)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(slots=True)
|
|
48
|
+
class OcrWorkflowCommandResult(WorkflowCommandResult):
|
|
49
|
+
artifacts: OCRWorkflowArtifacts | None = None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(slots=True)
|
|
53
|
+
class PageIndexWorkflowCommandResult(WorkflowCommandResult):
|
|
54
|
+
result: PageIndexParseResult | None = None
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(slots=True)
|
|
58
|
+
class LayerwiseWorkflowCommandResult(WorkflowCommandResult):
|
|
59
|
+
tree: Any | None = None
|
|
60
|
+
source_map: dict[str, Any] | None = None
|
|
61
|
+
graph_payload: dict[str, Any] | None = None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _fallback_parse_semantic_fn(*, collection, parser_input_dict: dict[str, Any], parser_source_map: dict[str, dict[str, Any]]):
|
|
65
|
+
root = SemanticNode(
|
|
66
|
+
title=collection.title,
|
|
67
|
+
node_type="DOCUMENT_ROOT",
|
|
68
|
+
total_content_pointers=[],
|
|
69
|
+
child_nodes=[],
|
|
70
|
+
level_from_root=0,
|
|
71
|
+
)
|
|
72
|
+
for page in collection.pages:
|
|
73
|
+
page_text_parts: list[str] = []
|
|
74
|
+
page_node = SemanticNode(
|
|
75
|
+
title=f"Page {page.page_number}",
|
|
76
|
+
node_type="PAGE",
|
|
77
|
+
total_content_pointers=[],
|
|
78
|
+
child_nodes=[],
|
|
79
|
+
level_from_root=1,
|
|
80
|
+
parent_id=root.node_id,
|
|
81
|
+
)
|
|
82
|
+
for unit in page.units:
|
|
83
|
+
if not getattr(unit, "text", None):
|
|
84
|
+
continue
|
|
85
|
+
text = str(unit.text)
|
|
86
|
+
page_text_parts.append(text)
|
|
87
|
+
page_node.child_nodes.append(
|
|
88
|
+
SemanticNode(
|
|
89
|
+
title=(text.splitlines()[0].strip()[:80] or f"Unit {unit.cluster_number or 0}"),
|
|
90
|
+
node_type="TEXT_FLOW",
|
|
91
|
+
total_content_pointers=[
|
|
92
|
+
HydratedTextPointer(
|
|
93
|
+
source_cluster_id=unit.unit_id or f"{collection.collection_id}|p{page.page_number}",
|
|
94
|
+
start_char=0,
|
|
95
|
+
end_char=max(0, len(text) - 1),
|
|
96
|
+
verbatim_text=text,
|
|
97
|
+
)
|
|
98
|
+
],
|
|
99
|
+
child_nodes=[],
|
|
100
|
+
level_from_root=2,
|
|
101
|
+
parent_id=page_node.node_id,
|
|
102
|
+
)
|
|
103
|
+
)
|
|
104
|
+
if not page_node.child_nodes:
|
|
105
|
+
page_node.child_nodes.append(
|
|
106
|
+
SemanticNode(
|
|
107
|
+
title=f"Page {page.page_number} Empty",
|
|
108
|
+
node_type="TEXT_FLOW",
|
|
109
|
+
total_content_pointers=[],
|
|
110
|
+
child_nodes=[],
|
|
111
|
+
level_from_root=2,
|
|
112
|
+
parent_id=page_node.node_id,
|
|
113
|
+
)
|
|
114
|
+
)
|
|
115
|
+
if page_text_parts:
|
|
116
|
+
page_node.total_content_pointers = [
|
|
117
|
+
HydratedTextPointer(
|
|
118
|
+
source_cluster_id=page.units[0].unit_id or f"{collection.collection_id}|p{page.page_number}",
|
|
119
|
+
start_char=0,
|
|
120
|
+
end_char=max(0, len("\n".join(page_text_parts)) - 1),
|
|
121
|
+
verbatim_text="\n".join(page_text_parts),
|
|
122
|
+
)
|
|
123
|
+
]
|
|
124
|
+
root.child_nodes.append(page_node)
|
|
125
|
+
return root
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
@contextmanager
|
|
129
|
+
def _temporary_env(overrides: dict[str, str | None]):
|
|
130
|
+
previous: dict[str, str | None] = {}
|
|
131
|
+
try:
|
|
132
|
+
for key, value in overrides.items():
|
|
133
|
+
previous[key] = os.environ.get(key)
|
|
134
|
+
if value is None:
|
|
135
|
+
os.environ.pop(key, None)
|
|
136
|
+
else:
|
|
137
|
+
os.environ[key] = value
|
|
138
|
+
yield
|
|
139
|
+
finally:
|
|
140
|
+
for key, value in previous.items():
|
|
141
|
+
if value is None:
|
|
142
|
+
os.environ.pop(key, None)
|
|
143
|
+
else:
|
|
144
|
+
os.environ[key] = value
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def build_legacy_parse_semantic_fn(
|
|
148
|
+
*,
|
|
149
|
+
provider_settings: WorkflowProviderSettings,
|
|
150
|
+
model_names: Sequence[str] | None = None,
|
|
151
|
+
):
|
|
152
|
+
parser_spec = provider_settings.parser
|
|
153
|
+
parser_model_names = list(model_names) if model_names else [parser_spec.model]
|
|
154
|
+
|
|
155
|
+
def _parse_semantic_fn(*, collection, parser_input_dict: dict[str, Any], parser_source_map: dict[str, dict[str, Any]]):
|
|
156
|
+
env_overrides = {
|
|
157
|
+
"KG_DOC_PARSER_PROVIDER": parser_spec.provider,
|
|
158
|
+
"KG_DOC_PARSER_MODEL": parser_spec.model,
|
|
159
|
+
"KG_DOC_PARSER_TEMPERATURE": str(parser_spec.temperature),
|
|
160
|
+
"KG_DOC_PARSER_BASE_URL": parser_spec.base_url,
|
|
161
|
+
"KG_DOC_PARSER_API_KEY_ENV": parser_spec.api_key_env,
|
|
162
|
+
"KG_DOC_PARSER_PROJECT": parser_spec.project,
|
|
163
|
+
"KG_DOC_PARSER_LOCATION": parser_spec.location,
|
|
164
|
+
"KG_DOC_PARSER_MAX_RETRIES": str(parser_spec.max_retries),
|
|
165
|
+
}
|
|
166
|
+
with _temporary_env(env_overrides):
|
|
167
|
+
return default_parse_semantic_fn(
|
|
168
|
+
collection=collection,
|
|
169
|
+
parser_input_dict=parser_input_dict,
|
|
170
|
+
parser_source_map=parser_source_map,
|
|
171
|
+
model_names=parser_model_names,
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
return _parse_semantic_fn
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _ensure_probe(output_dir: Path, probe: WorkflowProbe | None = None) -> WorkflowProbe:
|
|
178
|
+
if probe is not None:
|
|
179
|
+
return probe
|
|
180
|
+
return WorkflowProbe(output_dir / "workflow-events.jsonl")
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _emit(probe: WorkflowProbe | None, kind: str, /, **payload: Any) -> None:
|
|
184
|
+
emit_probe_event(probe, kind, **payload)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def discover_input_files(
|
|
188
|
+
paths: Sequence[str | Path],
|
|
189
|
+
*,
|
|
190
|
+
allowed_suffixes: set[str],
|
|
191
|
+
recursive: bool = True,
|
|
192
|
+
) -> list[Path]:
|
|
193
|
+
files: list[Path] = []
|
|
194
|
+
for raw_path in paths:
|
|
195
|
+
path = Path(raw_path)
|
|
196
|
+
if path.is_dir():
|
|
197
|
+
iterator = path.rglob("*") if recursive else path.iterdir()
|
|
198
|
+
for candidate in iterator:
|
|
199
|
+
if candidate.is_file() and candidate.suffix.lower() in allowed_suffixes:
|
|
200
|
+
files.append(candidate)
|
|
201
|
+
continue
|
|
202
|
+
if path.suffix.lower() in allowed_suffixes:
|
|
203
|
+
files.append(path)
|
|
204
|
+
return sorted({p.resolve(): p for p in files}.values(), key=lambda p: str(p))
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def run_ocr_source_workflow(
|
|
208
|
+
source_path: str | Path,
|
|
209
|
+
*,
|
|
210
|
+
output_dir: str | Path,
|
|
211
|
+
provider_settings: WorkflowProviderSettings | None = None,
|
|
212
|
+
ocr_runner=None,
|
|
213
|
+
pdf_rasterizer=None,
|
|
214
|
+
ocr_candidate_models: Sequence[str] | None = None,
|
|
215
|
+
workflow_engine=None,
|
|
216
|
+
conversation_engine=None,
|
|
217
|
+
knowledge_engine=None,
|
|
218
|
+
probe: WorkflowProbe | None = None,
|
|
219
|
+
deps: dict[str, Any] | None = None,
|
|
220
|
+
document_id: str | None = None,
|
|
221
|
+
title: str | None = None,
|
|
222
|
+
) -> OcrWorkflowCommandResult:
|
|
223
|
+
source_path = Path(source_path)
|
|
224
|
+
output_dir = Path(output_dir)
|
|
225
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
226
|
+
probe = _ensure_probe(output_dir, probe)
|
|
227
|
+
provider_settings = provider_settings or WorkflowProviderSettings.from_env()
|
|
228
|
+
document_id = document_id or source_path.stem
|
|
229
|
+
title = title or source_path.stem
|
|
230
|
+
|
|
231
|
+
_emit(
|
|
232
|
+
probe,
|
|
233
|
+
"workflow.file_started",
|
|
234
|
+
workflow_kind="ocr",
|
|
235
|
+
source_path=str(source_path),
|
|
236
|
+
output_dir=str(output_dir),
|
|
237
|
+
document_id=document_id,
|
|
238
|
+
)
|
|
239
|
+
if source_path.suffix.lower() == ".pdf":
|
|
240
|
+
image_payloads = None
|
|
241
|
+
pdf_path = source_path
|
|
242
|
+
else:
|
|
243
|
+
pdf_path = None
|
|
244
|
+
image_payloads = [
|
|
245
|
+
OCRImagePayload(page_number=1, image_path=str(source_path)),
|
|
246
|
+
]
|
|
247
|
+
if workflow_engine is None or conversation_engine is None:
|
|
248
|
+
workflow_engine, conversation_engine, default_knowledge_engine = build_default_engines(
|
|
249
|
+
output_dir / "engines",
|
|
250
|
+
provider_settings=provider_settings,
|
|
251
|
+
)
|
|
252
|
+
if knowledge_engine is None:
|
|
253
|
+
knowledge_engine = default_knowledge_engine
|
|
254
|
+
run, bundle, artifacts = run_ocr_ingest_workflow(
|
|
255
|
+
document_id=document_id,
|
|
256
|
+
title=title,
|
|
257
|
+
output_dir=output_dir,
|
|
258
|
+
workflow_engine=workflow_engine,
|
|
259
|
+
conversation_engine=conversation_engine,
|
|
260
|
+
knowledge_engine=knowledge_engine,
|
|
261
|
+
image_payloads=image_payloads,
|
|
262
|
+
pdf_path=pdf_path,
|
|
263
|
+
provider_settings=provider_settings,
|
|
264
|
+
ocr_runner=ocr_runner,
|
|
265
|
+
pdf_rasterizer=pdf_rasterizer,
|
|
266
|
+
ocr_candidate_models=ocr_candidate_models,
|
|
267
|
+
deps={"probe": probe, **(deps or {})},
|
|
268
|
+
probe=probe,
|
|
269
|
+
)
|
|
270
|
+
_emit(
|
|
271
|
+
probe,
|
|
272
|
+
"workflow.file_finished",
|
|
273
|
+
workflow_kind="ocr",
|
|
274
|
+
source_path=str(source_path),
|
|
275
|
+
output_dir=str(output_dir),
|
|
276
|
+
document_id=document_id,
|
|
277
|
+
status=run.status,
|
|
278
|
+
summary_path=str(artifacts.summary_path),
|
|
279
|
+
)
|
|
280
|
+
return OcrWorkflowCommandResult(
|
|
281
|
+
kind="ocr",
|
|
282
|
+
input_path=source_path,
|
|
283
|
+
output_dir=output_dir,
|
|
284
|
+
status=run.status,
|
|
285
|
+
probe_path=probe.path,
|
|
286
|
+
summary_path=artifacts.summary_path,
|
|
287
|
+
extra={
|
|
288
|
+
"run_id": run.run_id,
|
|
289
|
+
"bundle": bundle.model_dump(field_mode="backend", dump_format="json") if bundle is not None else None,
|
|
290
|
+
"state_db_path": str(artifacts.state_db_path),
|
|
291
|
+
"legacy_dir": str(artifacts.legacy_dir),
|
|
292
|
+
"rendered_dir": str(artifacts.rendered_dir),
|
|
293
|
+
},
|
|
294
|
+
artifacts=artifacts,
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def run_ocr_batch_workflow(
|
|
299
|
+
source_paths: Sequence[str | Path],
|
|
300
|
+
*,
|
|
301
|
+
output_dir: str | Path,
|
|
302
|
+
provider_settings: WorkflowProviderSettings | None = None,
|
|
303
|
+
ocr_runner=None,
|
|
304
|
+
pdf_rasterizer=None,
|
|
305
|
+
ocr_candidate_models: Sequence[str] | None = None,
|
|
306
|
+
workflow_engine=None,
|
|
307
|
+
conversation_engine=None,
|
|
308
|
+
knowledge_engine=None,
|
|
309
|
+
probe: WorkflowProbe | None = None,
|
|
310
|
+
deps: dict[str, Any] | None = None,
|
|
311
|
+
) -> list[OcrWorkflowCommandResult]:
|
|
312
|
+
output_dir = Path(output_dir)
|
|
313
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
314
|
+
probe = _ensure_probe(output_dir, probe)
|
|
315
|
+
files = discover_input_files(source_paths, allowed_suffixes=OCR_FILE_SUFFIXES, recursive=True)
|
|
316
|
+
_emit(probe, "workflow.batch_started", workflow_kind="ocr", file_count=len(files), output_dir=str(output_dir))
|
|
317
|
+
results: list[OcrWorkflowCommandResult] = []
|
|
318
|
+
for source_path in files:
|
|
319
|
+
relative_output = output_dir / source_path.stem
|
|
320
|
+
result = run_ocr_source_workflow(
|
|
321
|
+
source_path,
|
|
322
|
+
output_dir=relative_output,
|
|
323
|
+
provider_settings=provider_settings,
|
|
324
|
+
ocr_runner=ocr_runner,
|
|
325
|
+
pdf_rasterizer=pdf_rasterizer,
|
|
326
|
+
ocr_candidate_models=ocr_candidate_models,
|
|
327
|
+
workflow_engine=workflow_engine,
|
|
328
|
+
conversation_engine=conversation_engine,
|
|
329
|
+
knowledge_engine=knowledge_engine,
|
|
330
|
+
probe=probe,
|
|
331
|
+
deps=deps,
|
|
332
|
+
document_id=source_path.stem,
|
|
333
|
+
title=source_path.stem,
|
|
334
|
+
)
|
|
335
|
+
results.append(result)
|
|
336
|
+
_emit(probe, "workflow.batch_finished", workflow_kind="ocr", file_count=len(files), output_dir=str(output_dir))
|
|
337
|
+
return results
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def run_page_index_source_workflow(
|
|
341
|
+
source_path: str | Path,
|
|
342
|
+
*,
|
|
343
|
+
output_dir: str | Path,
|
|
344
|
+
mode: str = "heuristic",
|
|
345
|
+
source_format: str = "auto",
|
|
346
|
+
provider_settings: WorkflowProviderSettings | None = None,
|
|
347
|
+
probe: WorkflowProbe | None = None,
|
|
348
|
+
) -> PageIndexWorkflowCommandResult:
|
|
349
|
+
source_path = Path(source_path)
|
|
350
|
+
output_dir = Path(output_dir)
|
|
351
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
352
|
+
probe = _ensure_probe(output_dir, probe)
|
|
353
|
+
raw_text = source_path.read_text(encoding="utf-8")
|
|
354
|
+
inferred_format: PageIndexSourceFormat = (
|
|
355
|
+
"markdown" if source_format == "auto" and source_path.suffix.lower() == ".md" else "text"
|
|
356
|
+
)
|
|
357
|
+
if source_format in {"text", "markdown"}:
|
|
358
|
+
inferred_format = source_format # type: ignore[assignment]
|
|
359
|
+
|
|
360
|
+
_emit(
|
|
361
|
+
probe,
|
|
362
|
+
"workflow.file_started",
|
|
363
|
+
workflow_kind="page_index",
|
|
364
|
+
source_path=str(source_path),
|
|
365
|
+
output_dir=str(output_dir),
|
|
366
|
+
mode=mode,
|
|
367
|
+
)
|
|
368
|
+
result = parse_page_index_document(
|
|
369
|
+
document_id=source_path.stem,
|
|
370
|
+
title=source_path.stem,
|
|
371
|
+
raw_text=raw_text,
|
|
372
|
+
source_format=inferred_format,
|
|
373
|
+
mode=mode, # type: ignore[arg-type]
|
|
374
|
+
provider_settings=provider_settings,
|
|
375
|
+
)
|
|
376
|
+
summary = {
|
|
377
|
+
"kind": "page_index",
|
|
378
|
+
"source_path": str(source_path),
|
|
379
|
+
"mode": mode,
|
|
380
|
+
"source_format": inferred_format,
|
|
381
|
+
"overall_coverage": result.coverage.get("overall"),
|
|
382
|
+
"page_count": len(result.workflow_input.collections[0].pages),
|
|
383
|
+
"max_depth": max((len(page.child_nodes) for page in result.semantic_tree.child_nodes), default=0),
|
|
384
|
+
}
|
|
385
|
+
summary_path = output_dir / "page-index-summary.json"
|
|
386
|
+
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
|
387
|
+
_emit(
|
|
388
|
+
probe,
|
|
389
|
+
"workflow.file_finished",
|
|
390
|
+
workflow_kind="page_index",
|
|
391
|
+
source_path=str(source_path),
|
|
392
|
+
output_dir=str(output_dir),
|
|
393
|
+
mode=mode,
|
|
394
|
+
summary_path=str(summary_path),
|
|
395
|
+
)
|
|
396
|
+
return PageIndexWorkflowCommandResult(
|
|
397
|
+
kind="page_index",
|
|
398
|
+
input_path=source_path,
|
|
399
|
+
output_dir=output_dir,
|
|
400
|
+
status="succeeded",
|
|
401
|
+
probe_path=probe.path,
|
|
402
|
+
summary_path=summary_path,
|
|
403
|
+
extra=summary,
|
|
404
|
+
result=result,
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def run_page_index_batch_workflow(
|
|
409
|
+
source_paths: Sequence[str | Path],
|
|
410
|
+
*,
|
|
411
|
+
output_dir: str | Path,
|
|
412
|
+
mode: str = "heuristic",
|
|
413
|
+
source_format: str = "auto",
|
|
414
|
+
provider_settings: WorkflowProviderSettings | None = None,
|
|
415
|
+
probe: WorkflowProbe | None = None,
|
|
416
|
+
) -> list[PageIndexWorkflowCommandResult]:
|
|
417
|
+
output_dir = Path(output_dir)
|
|
418
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
419
|
+
probe = _ensure_probe(output_dir, probe)
|
|
420
|
+
files = discover_input_files(source_paths, allowed_suffixes=PAGE_INDEX_SUFFIXES, recursive=True)
|
|
421
|
+
_emit(
|
|
422
|
+
probe,
|
|
423
|
+
"workflow.batch_started",
|
|
424
|
+
workflow_kind="page_index",
|
|
425
|
+
file_count=len(files),
|
|
426
|
+
output_dir=str(output_dir),
|
|
427
|
+
)
|
|
428
|
+
results: list[PageIndexWorkflowCommandResult] = []
|
|
429
|
+
for source_path in files:
|
|
430
|
+
result = run_page_index_source_workflow(
|
|
431
|
+
source_path,
|
|
432
|
+
output_dir=output_dir / source_path.stem,
|
|
433
|
+
mode=mode,
|
|
434
|
+
source_format=source_format,
|
|
435
|
+
provider_settings=provider_settings,
|
|
436
|
+
probe=probe,
|
|
437
|
+
)
|
|
438
|
+
results.append(result)
|
|
439
|
+
_emit(probe, "workflow.batch_finished", workflow_kind="page_index", file_count=len(files), output_dir=str(output_dir))
|
|
440
|
+
return results
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def run_layerwise_source_workflow(
|
|
444
|
+
source_path: str | Path,
|
|
445
|
+
*,
|
|
446
|
+
output_dir: str | Path,
|
|
447
|
+
parsing_mode: str = "snippet",
|
|
448
|
+
max_depth: int = 10,
|
|
449
|
+
probe: WorkflowProbe | None = None,
|
|
450
|
+
) -> LayerwiseWorkflowCommandResult:
|
|
451
|
+
source_path = Path(source_path)
|
|
452
|
+
output_dir = Path(output_dir)
|
|
453
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
454
|
+
probe = _ensure_probe(output_dir, probe)
|
|
455
|
+
_emit(
|
|
456
|
+
probe,
|
|
457
|
+
"workflow.file_started",
|
|
458
|
+
workflow_kind="layerwise",
|
|
459
|
+
source_path=str(source_path),
|
|
460
|
+
output_dir=str(output_dir),
|
|
461
|
+
parsing_mode=parsing_mode,
|
|
462
|
+
)
|
|
463
|
+
if not source_path.is_dir():
|
|
464
|
+
raise ValueError("layerwise workflow expects a directory of legacy OCR page artifacts")
|
|
465
|
+
from kg_doc_parser.ocr import regen_doc
|
|
466
|
+
from kg_doc_parser.semantic_document_splitting_layerwise_edits import (
|
|
467
|
+
semantic_tree_to_kge_payload as legacy_semantic_tree_to_kge_payload,
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
raw_doc = {source_path.name: regen_doc(str(source_path), use_raw=True)}
|
|
471
|
+
tree, source_map = parse_tree_document(
|
|
472
|
+
doc_id=source_path.name,
|
|
473
|
+
raw_doc_dict=raw_doc,
|
|
474
|
+
parsing_mode=parsing_mode, # type: ignore[arg-type]
|
|
475
|
+
max_depth=max_depth,
|
|
476
|
+
)
|
|
477
|
+
graph_payload = legacy_semantic_tree_to_kge_payload(tree, doc_id=source_path.name)
|
|
478
|
+
graph_path = output_dir / "layerwise-graph.json"
|
|
479
|
+
graph_path.write_text(json.dumps(graph_payload, indent=2), encoding="utf-8")
|
|
480
|
+
summary = {
|
|
481
|
+
"kind": "layerwise",
|
|
482
|
+
"source_path": str(source_path),
|
|
483
|
+
"node_count": len(graph_payload.get("nodes", [])),
|
|
484
|
+
"edge_count": len(graph_payload.get("edges", [])),
|
|
485
|
+
"source_count": len(source_map),
|
|
486
|
+
"graph_path": str(graph_path),
|
|
487
|
+
}
|
|
488
|
+
summary_path = output_dir / "layerwise-summary.json"
|
|
489
|
+
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
|
490
|
+
_emit(
|
|
491
|
+
probe,
|
|
492
|
+
"workflow.file_finished",
|
|
493
|
+
workflow_kind="layerwise",
|
|
494
|
+
source_path=str(source_path),
|
|
495
|
+
output_dir=str(output_dir),
|
|
496
|
+
summary_path=str(summary_path),
|
|
497
|
+
)
|
|
498
|
+
return LayerwiseWorkflowCommandResult(
|
|
499
|
+
kind="layerwise",
|
|
500
|
+
input_path=source_path,
|
|
501
|
+
output_dir=output_dir,
|
|
502
|
+
status="succeeded",
|
|
503
|
+
probe_path=probe.path,
|
|
504
|
+
summary_path=summary_path,
|
|
505
|
+
extra=summary,
|
|
506
|
+
tree=tree,
|
|
507
|
+
source_map=source_map,
|
|
508
|
+
graph_payload=graph_payload,
|
|
509
|
+
)
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def run_layerwise_batch_workflow(
|
|
513
|
+
source_paths: Sequence[str | Path],
|
|
514
|
+
*,
|
|
515
|
+
output_dir: str | Path,
|
|
516
|
+
parsing_mode: str = "snippet",
|
|
517
|
+
max_depth: int = 10,
|
|
518
|
+
probe: WorkflowProbe | None = None,
|
|
519
|
+
) -> list[LayerwiseWorkflowCommandResult]:
|
|
520
|
+
output_dir = Path(output_dir)
|
|
521
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
522
|
+
probe = _ensure_probe(output_dir, probe)
|
|
523
|
+
dirs = [Path(path) for path in source_paths if Path(path).is_dir()]
|
|
524
|
+
_emit(
|
|
525
|
+
probe,
|
|
526
|
+
"workflow.batch_started",
|
|
527
|
+
workflow_kind="layerwise",
|
|
528
|
+
file_count=len(dirs),
|
|
529
|
+
output_dir=str(output_dir),
|
|
530
|
+
)
|
|
531
|
+
results: list[LayerwiseWorkflowCommandResult] = []
|
|
532
|
+
for source_path in dirs:
|
|
533
|
+
result = run_layerwise_source_workflow(
|
|
534
|
+
source_path,
|
|
535
|
+
output_dir=output_dir / source_path.name,
|
|
536
|
+
parsing_mode=parsing_mode,
|
|
537
|
+
max_depth=max_depth,
|
|
538
|
+
probe=probe,
|
|
539
|
+
)
|
|
540
|
+
results.append(result)
|
|
541
|
+
_emit(probe, "workflow.batch_finished", workflow_kind="layerwise", file_count=len(dirs), output_dir=str(output_dir))
|
|
542
|
+
return results
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def run_demo_harness_workflow(config: DemoHarnessConfig):
|
|
546
|
+
return run_demo_harness(config)
|