graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,187 @@
1
+ from .adapters import (
2
+ build_authoritative_source_map,
3
+ build_parser_input_dict,
4
+ build_parser_source_map,
5
+ normalize_ocr_pages,
6
+ )
7
+ from .clients import (
8
+ CanonicalGraphPersistenceClient,
9
+ DocumentTreeApiPersistenceClient,
10
+ DirectRuntimeIngestClient,
11
+ IngestExecutionClient,
12
+ ServerCanonicalKgClient,
13
+ UnsupportedClientOperation,
14
+ )
15
+ from .cache import WorkflowLLMCallCache
16
+ from .design import DEFAULT_WORKFLOW_ID, build_ingest_workflow_design, ensure_ingest_workflow_design
17
+ from .demo_harness import DemoHarnessArtifacts, DemoHarnessConfig, run_demo_harness
18
+ from .handlers import build_ingest_step_resolver
19
+ from .parsing import (
20
+ OCRParseRequest,
21
+ PageIndexParseRequest,
22
+ ParseMode,
23
+ TreeParseRequest,
24
+ parse_document,
25
+ parse_ocr_document,
26
+ parse_page_index_document,
27
+ parse_tree_document,
28
+ )
29
+ from .ocr_pipeline import (
30
+ OCRImagePayload,
31
+ OCRWorkflowArtifacts,
32
+ OCRWorkflowStateStore,
33
+ prepare_ocr_workflow_input,
34
+ run_ocr_ingest_workflow,
35
+ )
36
+ from .runners import (
37
+ LayerwiseWorkflowCommandResult,
38
+ OcrWorkflowCommandResult,
39
+ PageIndexWorkflowCommandResult,
40
+ WorkflowCommandResult,
41
+ discover_input_files,
42
+ run_demo_harness_workflow,
43
+ run_layerwise_batch_workflow,
44
+ run_layerwise_source_workflow,
45
+ run_ocr_batch_workflow,
46
+ run_ocr_source_workflow,
47
+ run_page_index_batch_workflow,
48
+ run_page_index_source_workflow,
49
+ )
50
+ from .smoke_assets import generate_ocr_smoke_assets
51
+ from .page_index import PageIndexBlockSpec, PageIndexParseResult, build_page_index_workflow_input
52
+ from .models import (
53
+ BoundingBox,
54
+ CanonicalGraphWriteResult,
55
+ CurrentLayerContext,
56
+ CurrentLayerReview,
57
+ CurrentLayerResult,
58
+ LayerCoverageGap,
59
+ LayerDuplicateChildNote,
60
+ GroundedSourceRecord,
61
+ IngestRunHandle,
62
+ IngestRunResult,
63
+ LayerChildCandidate,
64
+ LayerFrontierItem,
65
+ LayerSpanConflict,
66
+ NormalizedPage,
67
+ NormalizedSourceCollection,
68
+ ParseSessionState,
69
+ SourceUnit,
70
+ ValidationReport,
71
+ WorkflowExportBundle,
72
+ WorkflowIngestInput,
73
+ )
74
+ from .providers import (
75
+ EmbeddingProviderConfig,
76
+ FakeChatModel,
77
+ ProviderEndpointConfig,
78
+ WorkflowProviderSettings,
79
+ build_chat_model,
80
+ build_chat_model_for_role,
81
+ build_embedding_function,
82
+ )
83
+ from .probe import WorkflowProbe, emit_probe_event
84
+ from .parser_core import (
85
+ apply_cud_update,
86
+ check_layer_coverage,
87
+ default_parse_semantic_fn,
88
+ finalize_semantic_tree,
89
+ initialize_parse_session,
90
+ prepare_layer_frontier,
91
+ propose_layer_breakdown,
92
+ review_layer,
93
+ )
94
+ from .service import build_default_engines, build_runtime, run_ingest_workflow
95
+ from .semantics import HydratedTextPointer, SemanticNode
96
+
97
+ __all__ = [
98
+ "BoundingBox",
99
+ "CanonicalGraphPersistenceClient",
100
+ "CanonicalGraphWriteResult",
101
+ "CurrentLayerContext",
102
+ "CurrentLayerReview",
103
+ "CurrentLayerResult",
104
+ "LayerCoverageGap",
105
+ "LayerDuplicateChildNote",
106
+ "DemoHarnessArtifacts",
107
+ "DemoHarnessConfig",
108
+ "DocumentTreeApiPersistenceClient",
109
+ "DirectRuntimeIngestClient",
110
+ "EmbeddingProviderConfig",
111
+ "GroundedSourceRecord",
112
+ "HydratedTextPointer",
113
+ "FakeChatModel",
114
+ "IngestExecutionClient",
115
+ "IngestRunHandle",
116
+ "IngestRunResult",
117
+ "LayerChildCandidate",
118
+ "LayerFrontierItem",
119
+ "LayerSpanConflict",
120
+ "NormalizedPage",
121
+ "NormalizedSourceCollection",
122
+ "OCRImagePayload",
123
+ "OCRWorkflowArtifacts",
124
+ "OCRWorkflowStateStore",
125
+ "OCRParseRequest",
126
+ "ParseSessionState",
127
+ "ProviderEndpointConfig",
128
+ "PageIndexParseRequest",
129
+ "PageIndexBlockSpec",
130
+ "PageIndexParseResult",
131
+ "ParseMode",
132
+ "TreeParseRequest",
133
+ "SemanticNode",
134
+ "ServerCanonicalKgClient",
135
+ "SourceUnit",
136
+ "UnsupportedClientOperation",
137
+ "ValidationReport",
138
+ "WorkflowLLMCallCache",
139
+ "WorkflowProbe",
140
+ "WorkflowProviderSettings",
141
+ "WorkflowExportBundle",
142
+ "WorkflowIngestInput",
143
+ "WorkflowCommandResult",
144
+ "OcrWorkflowCommandResult",
145
+ "PageIndexWorkflowCommandResult",
146
+ "LayerwiseWorkflowCommandResult",
147
+ "DEFAULT_WORKFLOW_ID",
148
+ "build_chat_model",
149
+ "build_chat_model_for_role",
150
+ "build_embedding_function",
151
+ "apply_cud_update",
152
+ "check_layer_coverage",
153
+ "default_parse_semantic_fn",
154
+ "finalize_semantic_tree",
155
+ "initialize_parse_session",
156
+ "prepare_layer_frontier",
157
+ "propose_layer_breakdown",
158
+ "review_layer",
159
+ "build_authoritative_source_map",
160
+ "build_default_engines",
161
+ "build_ingest_workflow_design",
162
+ "build_ingest_step_resolver",
163
+ "build_parser_input_dict",
164
+ "build_page_index_workflow_input",
165
+ "build_parser_source_map",
166
+ "build_runtime",
167
+ "ensure_ingest_workflow_design",
168
+ "emit_probe_event",
169
+ "normalize_ocr_pages",
170
+ "parse_document",
171
+ "parse_ocr_document",
172
+ "parse_page_index_document",
173
+ "parse_tree_document",
174
+ "prepare_ocr_workflow_input",
175
+ "run_demo_harness",
176
+ "run_demo_harness_workflow",
177
+ "run_ingest_workflow",
178
+ "discover_input_files",
179
+ "run_layerwise_batch_workflow",
180
+ "run_layerwise_source_workflow",
181
+ "run_ocr_batch_workflow",
182
+ "run_ocr_source_workflow",
183
+ "run_page_index_batch_workflow",
184
+ "run_page_index_source_workflow",
185
+ "generate_ocr_smoke_assets",
186
+ "run_ocr_ingest_workflow",
187
+ ]
@@ -0,0 +1,13 @@
1
+ from __future__ import annotations
2
+
3
+ from importlib import import_module
4
+
5
+
6
+ def ensure_kogwistar_on_path() -> None:
7
+ """Compatibility shim: require a normal kogwistar installation/import.
8
+
9
+ We intentionally do not mutate ``sys.path`` here. The repo may contain a sibling
10
+ checkout for inspection, but runtime imports should resolve through the installed
11
+ package environment, including editable installs.
12
+ """
13
+ import_module("kogwistar")
@@ -0,0 +1,212 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any, TypedDict
4
+
5
+ from .models import (
6
+ BoundingBox,
7
+ GroundedSourceRecord,
8
+ NormalizedPage,
9
+ NormalizedSourceCollection,
10
+ SourceUnit,
11
+ WorkflowIngestInput,
12
+ )
13
+
14
+
15
+ class OCRPageJSON(TypedDict, total=False):
16
+ """Serialized OCR page payload produced by the OCR preparation layer."""
17
+
18
+ pdf_page_num: int
19
+ printed_page_number: str
20
+ contains_table: bool
21
+ OCR_text_clusters: list[dict[str, Any]]
22
+ non_text_objects: list[dict[str, Any]]
23
+ text: str
24
+
25
+
26
+ def normalize_ocr_pages(
27
+ *,
28
+ document_id: str,
29
+ title: str,
30
+ pages: list[OCRPageJSON],
31
+ ) -> WorkflowIngestInput:
32
+ """Normalize raw OCR JSON into workflow ingest models.
33
+
34
+ OCR text clusters and non-text regions are both preserved with page and
35
+ cluster metadata. `embedding_space="image"` is an intent label here; it
36
+ does not imply a separate image embedder is already wired at runtime.
37
+ """
38
+ normalized_pages: list[NormalizedPage] = []
39
+ for raw_page in pages:
40
+ page_number = int(raw_page["pdf_page_num"])
41
+ units: list[SourceUnit] = []
42
+ for cluster in raw_page.get("OCR_text_clusters", []):
43
+ bbox = BoundingBox(
44
+ y_min=float(cluster.get("bb_y_min", 0.0)),
45
+ x_min=float(cluster.get("bb_x_min", 0.0)),
46
+ y_max=float(cluster.get("bb_y_max", 0.0)),
47
+ x_max=float(cluster.get("bb_x_max", 0.0)),
48
+ )
49
+ units.append(
50
+ SourceUnit(
51
+ modality="ocr_text",
52
+ page_number=page_number,
53
+ cluster_number=cluster.get("cluster_number"),
54
+ text=cluster.get("text"),
55
+ bbox=bbox,
56
+ metadata={},
57
+ )
58
+ )
59
+ for obj in raw_page.get("non_text_objects", []):
60
+ bbox = BoundingBox(
61
+ y_min=float(obj.get("bb_y_min", 0.0)),
62
+ x_min=float(obj.get("bb_x_min", 0.0)),
63
+ y_max=float(obj.get("bb_y_max", 0.0)),
64
+ x_max=float(obj.get("bb_x_max", 0.0)),
65
+ )
66
+ units.append(
67
+ SourceUnit(
68
+ modality="image_region",
69
+ page_number=page_number,
70
+ cluster_number=obj.get("cluster_number"),
71
+ description=obj.get("description"),
72
+ bbox=bbox,
73
+ # Keep the "image" space label for future routing; the
74
+ # current engine still embeds through a single function.
75
+ embedding_space="image",
76
+ metadata={"participates_in_semantic_text": False},
77
+ )
78
+ )
79
+ normalized_pages.append(
80
+ NormalizedPage(
81
+ page_number=page_number,
82
+ units=units,
83
+ metadata={
84
+ "printed_page_number": raw_page.get("printed_page_number"),
85
+ "contains_table": raw_page.get("contains_table"),
86
+ },
87
+ )
88
+ )
89
+ embedding_spaces = ["default_text"]
90
+ if any(
91
+ unit.embedding_space == "image"
92
+ for page in normalized_pages
93
+ for unit in page.units
94
+ ):
95
+ embedding_spaces.append("image")
96
+ return WorkflowIngestInput(
97
+ request_id=document_id,
98
+ collections=[
99
+ NormalizedSourceCollection(
100
+ collection_id=document_id,
101
+ title=title,
102
+ modality="ocr",
103
+ pages=normalized_pages,
104
+ embedding_spaces=embedding_spaces,
105
+ )
106
+ ],
107
+ )
108
+
109
+
110
+ def build_authoritative_source_map(
111
+ inp: WorkflowIngestInput,
112
+ ) -> dict[str, GroundedSourceRecord]:
113
+ source_map: dict[str, GroundedSourceRecord] = {}
114
+ for collection in inp.collections:
115
+ for page in collection.pages:
116
+ ordinal_by_modality: dict[str, int] = {}
117
+ for unit in page.units:
118
+ modality_prefix = "t" if unit.modality in {"text", "ocr_text"} else "i"
119
+ original_cluster = unit.cluster_number
120
+ if original_cluster is None:
121
+ original_cluster = ordinal_by_modality.get(modality_prefix, 0)
122
+ ordinal_by_modality[modality_prefix] = ordinal_by_modality.get(modality_prefix, 0) + 1
123
+ unit_id = unit.unit_id or f"{collection.collection_id}|p{page.page_number}_{modality_prefix}{original_cluster}"
124
+ record = GroundedSourceRecord(
125
+ unit_id=unit_id,
126
+ collection_id=collection.collection_id,
127
+ modality=unit.modality,
128
+ page_number=page.page_number,
129
+ cluster_number=unit.cluster_number,
130
+ text=unit.text or unit.description or "",
131
+ parser_text=unit.parser_text,
132
+ source_uri=unit.source_uri,
133
+ embedding_space=unit.embedding_space,
134
+ participates_in_semantic_text=unit.modality in {"text", "ocr_text"},
135
+ bbox=unit.bbox,
136
+ metadata={
137
+ **unit.metadata,
138
+ "original_cluster_number": unit.cluster_number,
139
+ },
140
+ )
141
+ if unit_id in source_map:
142
+ raise ValueError(f"source map collision detected for {unit_id}")
143
+ source_map[unit_id] = record
144
+ return source_map
145
+
146
+
147
+ def select_primary_collection(inp: WorkflowIngestInput) -> NormalizedSourceCollection:
148
+ return inp.collections[0]
149
+
150
+
151
+ def build_parser_input_dict(
152
+ collection: NormalizedSourceCollection,
153
+ ) -> dict[str, Any]:
154
+ pages: list[dict[str, Any]] = []
155
+ for page in collection.pages:
156
+ text_clusters = []
157
+ non_text_objects = []
158
+ next_cluster = 0
159
+ for unit in page.units:
160
+ bbox = unit.bbox
161
+ cluster_number = unit.cluster_number if unit.cluster_number is not None else next_cluster
162
+ if unit.modality in {"text", "ocr_text"}:
163
+ text_clusters.append(
164
+ {
165
+ "text": unit.text or "",
166
+ "bb_x_min": bbox.x_min if bbox else 0.0,
167
+ "bb_x_max": bbox.x_max if bbox else 0.0,
168
+ "bb_y_min": bbox.y_min if bbox else 0.0,
169
+ "bb_y_max": bbox.y_max if bbox else 0.0,
170
+ "cluster_number": cluster_number,
171
+ }
172
+ )
173
+ else:
174
+ non_text_objects.append(
175
+ {
176
+ "description": unit.description or unit.source_uri or "image-region",
177
+ "bb_x_min": bbox.x_min if bbox else 0.0,
178
+ "bb_x_max": bbox.x_max if bbox else 0.0,
179
+ "bb_y_min": bbox.y_min if bbox else 0.0,
180
+ "bb_y_max": bbox.y_max if bbox else 0.0,
181
+ "cluster_number": cluster_number,
182
+ }
183
+ )
184
+ next_cluster += 1
185
+ pages.append(
186
+ {
187
+ "pdf_page_num": page.page_number,
188
+ "printed_page_number": page.metadata.get("printed_page_number"),
189
+ "contains_table": page.metadata.get("contains_table", False),
190
+ "OCR_text_clusters": text_clusters,
191
+ "non_text_objects": non_text_objects,
192
+ }
193
+ )
194
+ return {"document_filename": collection.title, "pages": pages}
195
+
196
+
197
+ def build_parser_source_map(
198
+ source_map: dict[str, GroundedSourceRecord],
199
+ ) -> dict[str, dict[str, Any]]:
200
+ return {
201
+ unit_id: {
202
+ "id": record.unit_id,
203
+ "text": record.parser_text,
204
+ "modality": record.modality,
205
+ "page_number": record.page_number,
206
+ "cluster_number": record.cluster_number,
207
+ "embedding_space": record.embedding_space,
208
+ "participates_in_semantic_text": record.participates_in_semantic_text,
209
+ "metadata": record.metadata,
210
+ }
211
+ for unit_id, record in source_map.items()
212
+ }
@@ -0,0 +1,63 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Any, Callable, TypeVar
6
+
7
+ from kogwistar.id_provider import stable_id
8
+
9
+ from .probe import emit_probe_event
10
+
11
+ T = TypeVar("T")
12
+
13
+
14
+ def _jsonable(value: Any) -> Any:
15
+ if hasattr(value, "model_dump"):
16
+ try:
17
+ return value.model_dump(field_mode="backend", dump_format="json")
18
+ except TypeError:
19
+ return value.model_dump()
20
+ if isinstance(value, dict):
21
+ return {str(k): _jsonable(v) for k, v in value.items()}
22
+ if isinstance(value, (list, tuple)):
23
+ return [_jsonable(v) for v in value]
24
+ return value
25
+
26
+
27
+ class WorkflowLLMCallCache:
28
+ def __init__(self, cache_dir: str | Path, *, probe=None) -> None:
29
+ self.cache_dir = Path(cache_dir)
30
+ self.cache_dir.mkdir(parents=True, exist_ok=True)
31
+ self.probe = probe
32
+
33
+ def _cache_path(self, operation: str, fingerprint: dict[str, Any]) -> Path:
34
+ payload = json.dumps(_jsonable(fingerprint), sort_keys=True, ensure_ascii=False, separators=(",", ":"))
35
+ cache_id = stable_id("workflow_ingest.llm_call", operation, payload)
36
+ return self.cache_dir / f"{cache_id}.json"
37
+
38
+ def cached_call(
39
+ self,
40
+ *,
41
+ operation: str,
42
+ fingerprint: dict[str, Any],
43
+ fn: Callable[[], T],
44
+ ) -> Any:
45
+ path = self._cache_path(operation, fingerprint)
46
+ if path.exists():
47
+ emit_probe_event(
48
+ self.probe,
49
+ "workflow.llm_cache_hit",
50
+ operation=operation,
51
+ cache_path=str(path),
52
+ )
53
+ return json.loads(path.read_text(encoding="utf-8"))
54
+ result = fn()
55
+ payload = _jsonable(result)
56
+ path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
57
+ emit_probe_event(
58
+ self.probe,
59
+ "workflow.llm_cache_miss",
60
+ operation=operation,
61
+ cache_path=str(path),
62
+ )
63
+ return payload