graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
kg_doc_parser/models.py
ADDED
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Data Models & Schema Definitions
|
|
3
|
+
================================
|
|
4
|
+
|
|
5
|
+
This module defines the core data models for the Optical Character Recognition (OCR) layer
|
|
6
|
+
and documents the implicit schema of the Resultant Knowledge Graph.
|
|
7
|
+
|
|
8
|
+
Physical Layer (OCR)
|
|
9
|
+
--------------------
|
|
10
|
+
- SplitPage: Represents a single page of a document.
|
|
11
|
+
- OCRClusterResponse: Contains detected text clusters and non-text objects.
|
|
12
|
+
- TextCluster: A bounding box with text content.
|
|
13
|
+
- text: str (The recognized text content)
|
|
14
|
+
- cluster_number: int (Unique ID on the page)
|
|
15
|
+
- bb_x_min, bb_x_max, bb_y_min, bb_y_max: float (Bounding box coordinates)
|
|
16
|
+
|
|
17
|
+
[PDF/Image] -> [OCR Service] -> [SplitPage] -> [TextCluster]
|
|
18
|
+
|
|
19
|
+
Resultant Knowledge Graph Structure
|
|
20
|
+
-----------------------------------
|
|
21
|
+
The semantic parsing pipeline (`kg_doc_parser.semantic_document_splitting_layerwise_edits`) transforms
|
|
22
|
+
these physical models into a Knowledge Graph.
|
|
23
|
+
|
|
24
|
+
[SemanticNode] (Entity)
|
|
25
|
+
|
|
|
26
|
+
+--- mentions ---+
|
|
27
|
+
| |
|
|
28
|
+
| [TextCluster] (Grounding)
|
|
29
|
+
|
|
|
30
|
+
+--- HAS_CHILD --> [SemanticNode] (Sub-entity)
|
|
31
|
+
|
|
32
|
+
Graph Schema (Nodes & Edges)
|
|
33
|
+
----------------------------
|
|
34
|
+
Nodes (Entities):
|
|
35
|
+
- id: UUID
|
|
36
|
+
- label: Title/Heading
|
|
37
|
+
- type: "entity"
|
|
38
|
+
- subtype: "TEXT_FLOW" | "KEY_VALUE_PAIR" | "TABLE"
|
|
39
|
+
- mentions: Link to source `TextCluster`s (provenance)
|
|
40
|
+
|
|
41
|
+
Edges (Relationships):
|
|
42
|
+
- id: UUID
|
|
43
|
+
- source_id: Parent Node UUID
|
|
44
|
+
- target_id: Child Node UUID
|
|
45
|
+
- relation: "HAS_CHILD"
|
|
46
|
+
- type: "relationship"
|
|
47
|
+
"""
|
|
48
|
+
if True:
|
|
49
|
+
import logging
|
|
50
|
+
import os
|
|
51
|
+
logger = logging.getLogger(__name__)
|
|
52
|
+
logger.addHandler(logging.NullHandler())
|
|
53
|
+
logger.debug("loading models")
|
|
54
|
+
from typing import List, Literal, Optional, Dict, Any, Type, Union, Annotated, ClassVar
|
|
55
|
+
try:
|
|
56
|
+
from typing import TypeAlias
|
|
57
|
+
except ImportError: # pragma: no cover
|
|
58
|
+
from typing_extensions import TypeAlias
|
|
59
|
+
from pydantic import BaseModel, Field, model_validator, field_validator, ValidationInfo
|
|
60
|
+
|
|
61
|
+
from pydantic_extension.model_slicing import (ModeSlicingMixin, NotMode, FrontendField, BackendField, LLMField,
|
|
62
|
+
DtoType,
|
|
63
|
+
BackendType,
|
|
64
|
+
FrontendType,
|
|
65
|
+
LLMType,
|
|
66
|
+
use_mode)
|
|
67
|
+
from pydantic_extension.model_slicing.mixin import ExcludeMode, DtoField
|
|
68
|
+
JsonPrimitive = Union[str, int, float, bool, None]
|
|
69
|
+
#========================= OCR DOC
|
|
70
|
+
|
|
71
|
+
# pre-validation model
|
|
72
|
+
class box_2d(BaseModel): # text box for ocr, the id here is not a database unique id but a unique id for identified object
|
|
73
|
+
box_2d: list[int] = Field(description = 'box y min, x min, y max and x max')
|
|
74
|
+
label : str = Field(description = 'text in the box')
|
|
75
|
+
id: int = Field(description = 'id of the text box in the page, autoincrement from 0')
|
|
76
|
+
class NonText_box_2d(BaseModel): # text box for ocr, the id here is not a database unique id but a unique id for identified object
|
|
77
|
+
"""Recognised meaningful objects other than OCR characters, include image, figures. """
|
|
78
|
+
description: str = Field(description='the description or summary of the non-OCR object')
|
|
79
|
+
box_2d: list[int] = Field(description = 'box y min, x min, y max and x max')
|
|
80
|
+
id: int = Field(description="per page unique number of the cluster, starting from 0")
|
|
81
|
+
|
|
82
|
+
# post validation model
|
|
83
|
+
class TextCluster(ModeSlicingMixin, BaseModel):
|
|
84
|
+
"""a text cluster along with spatial information"""
|
|
85
|
+
text: DtoType[str] = Field(description='the text content of the text cluster')
|
|
86
|
+
bb_x_min: DtoType[float] = Field(description='the bounding box x min in pixel coordinate of the text_cluster. ')
|
|
87
|
+
bb_x_max: DtoType[float] = Field(description='the bounding box x max in pixel coordinate of the text_cluster. ')
|
|
88
|
+
bb_y_min: DtoType[float] = Field(description='the bounding box y min in pixel coordinate of the text_cluster. ')
|
|
89
|
+
bb_y_max: DtoType[float] = Field(description='the bounding box y max in pixel coordinate of the text_cluster. ')
|
|
90
|
+
cluster_number: int = Field(description="per page unique number of the cluster, starting from 0")
|
|
91
|
+
class NonTextCluster(ModeSlicingMixin, BaseModel):
|
|
92
|
+
"""Recognised meaningful objects other than OCR characters, include image, figures. """
|
|
93
|
+
description: DtoType[str] = Field(description='the description or summary of the non-OCR object')
|
|
94
|
+
bb_x_min: DtoType[float] = Field(description='the bounding box x min in pixel coordinate of the non-OCR object. ')
|
|
95
|
+
bb_x_max: DtoType[float] = Field(description='the bounding box x max in pixel coordinate of the non-OCR object. ')
|
|
96
|
+
bb_y_min: DtoType[float] = Field(description='the bounding box y min in pixel coordinate of the non-OCR object. ')
|
|
97
|
+
bb_y_max: DtoType[float] = Field(description='the bounding box y max in pixel coordinate of the non-OCR object. ')
|
|
98
|
+
cluster_number: int = Field(description="per page unique number of the cluster, starting from 0")
|
|
99
|
+
|
|
100
|
+
class OCRClusterResponse(ModeSlicingMixin, BaseModel):
|
|
101
|
+
"""id/ cluster number must be all unique, for example, one of the ocr boxes_2d used id='1',
|
|
102
|
+
the first image box id (cluster numebr) will be '2', the next signature will be '3' """
|
|
103
|
+
OCR_text_clusters: DtoType[list[TextCluster]] = Field(description="the OCR text results. Share cluster number uniqueness with non-OCR objects. Include emoji or unicode text")
|
|
104
|
+
non_text_objects: DtoType[list[NonTextCluster]] = Field(description="the non-OCR object results. Share cluster number uniqueness with OCR texts. ")
|
|
105
|
+
is_empty_page: DtoType[Optional[bool]] = Field(default = False, description="true if the whole page is empty without recognisable text.")
|
|
106
|
+
printed_page_number: DtoType[Optional[str]] = Field(description='the page number identified from OCR texts, can be in form of roman numerals such as "i", "ii", "iii", "iv"...; '
|
|
107
|
+
'Arabic numeral such as 1, 2, 3... or letter such as "a", "b", "c"...\n'
|
|
108
|
+
'Sometimes the are surrounded by symbols such as "- 1 -", "- 2 -"'
|
|
109
|
+
r"Can be null/none if there is no page order assigned and printed and found in the scanned texts. Do not assign page number. Only use page number found.")
|
|
110
|
+
meaningful_ordering : DtoType[list[int]] = Field(description="The correct meaningful ordering of the identified text clusters. Must cover all OCR_text_clusters once and only once. ")
|
|
111
|
+
page_x_min : DtoType[float]=Field(description='the page x min in pixel coordinate. ')
|
|
112
|
+
page_x_max : DtoType[float]=Field(description='the page x max in pixel coordinate. ')
|
|
113
|
+
page_y_min : DtoType[float]=Field(description='the page y min in pixel coordinate. ')
|
|
114
|
+
page_y_max : DtoType[float]=Field(description='the page y max in pixel coordinate. ')
|
|
115
|
+
estimated_rotation_degrees : DtoType[float]=Field(description='the page estimated rotation degree using right hand rule. ')
|
|
116
|
+
incomplete_words_on_edge: DtoType[bool] = Field(description='If there is any text being incomplete due to the scan does not scan the edges properly. ')
|
|
117
|
+
incomplete_text: DtoType[bool] = Field(description='Any incomplete text')
|
|
118
|
+
data_loss_likelihood: DtoType[float] = Field(description='The likelihood (range from 0.0 to 1.0 inclusive) that the page has lost information by missing the scan data on the edges of the page.' )
|
|
119
|
+
scan_quality: DtoType[Literal['low', 'medium', 'high']] = Field(description='The image quality of the scan. All qualities exclude signatures. '
|
|
120
|
+
'"low", "medium" or "high". '
|
|
121
|
+
'low: text barely legible. medium: Legible with non smooth due to pixelation. high: texts are easily and highly identifiable. ' )
|
|
122
|
+
contains_table: DtoType[bool] = Field(description='Whether this page contains table. ')
|
|
123
|
+
|
|
124
|
+
@model_validator(mode='after')
|
|
125
|
+
def check_cluster_meaningful_ordering_agreement(self):
|
|
126
|
+
assert bool(self.is_empty_page) ^ (len(self.OCR_text_clusters) > 0), f"is_empty_page value {self.is_empty_page} disagree with OCR_text_clusters len={len(self.OCR_text_clusters)}"
|
|
127
|
+
overlap_id = set(i.cluster_number for i in self.OCR_text_clusters).intersection(set(i.cluster_number for i in self.non_text_objects))
|
|
128
|
+
if overlap_id:
|
|
129
|
+
raise ValueError(f"cluster number from non_text_objects block and ocr text blocks must be ALL distinct. Overlapped ids: {list(overlap_id)}")
|
|
130
|
+
try:
|
|
131
|
+
if not (len(self.meaningful_ordering) == len(set(self.meaningful_ordering))): # <= len(self.OCR_text_clusters)):
|
|
132
|
+
raise ValueError("meaningful_order must cover each text cluster at most once")
|
|
133
|
+
except Exception as e:
|
|
134
|
+
raise e
|
|
135
|
+
return self
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class OCRClusterResponseBc(OCRClusterResponse):
|
|
140
|
+
# backward compability layer, can use by overriding fields with union of previous versions
|
|
141
|
+
|
|
142
|
+
pass
|
|
143
|
+
# OCR_text_clusters: list[TextCluster | TextCluster_yolo_bb] = Field(description="the OCR text results, " # type: ignore
|
|
144
|
+
# "prefer min max bounding box to centre width height style")
|
|
145
|
+
|
|
146
|
+
class SplitPageMeta(BaseModel):
|
|
147
|
+
ocr_model_name: str = Field(description="model that perform OCR")
|
|
148
|
+
ocr_datetime: float = Field(description="unix timestamp when ocr is performed")
|
|
149
|
+
ocr_json_version: str = Field(description = "the model does the OCR")
|
|
150
|
+
@field_validator('ocr_json_version', mode = "before")
|
|
151
|
+
def version_to_str(cls, v):
|
|
152
|
+
return str(v)
|
|
153
|
+
class SplitPage(OCRClusterResponseBc):
|
|
154
|
+
# model not for LLM response
|
|
155
|
+
pdf_page_num: int
|
|
156
|
+
metadata: SplitPageMeta
|
|
157
|
+
refined_version: Optional[OCRClusterResponse[DtoField]] = Field(default = None, description = "refined processed/ grouped/ merged version of ocr text clusters. ")
|
|
158
|
+
def model_dump(self, *arg, **kwarg):
|
|
159
|
+
return self.to_doc()
|
|
160
|
+
def dump_raw(self, *arg, **kwarg):
|
|
161
|
+
return super(SplitPage, self).model_dump(exclude = ["refined_version"], *arg, **kwarg)
|
|
162
|
+
def dump_supercede_parse(self, *arg, **kwarg):
|
|
163
|
+
return super(SplitPage, self).model_dump(exclude = ["refined_version", "metadata"], *arg, **kwarg)
|
|
164
|
+
@model_validator(mode="after")
|
|
165
|
+
def roundtrip_invariant(self, info: ValidationInfo) -> "SplitPage":
|
|
166
|
+
# Context may be None if caller didn't pass it
|
|
167
|
+
ctx = info.context or {}
|
|
168
|
+
# Re-entrancy guard
|
|
169
|
+
if ctx.get("_roundtrip_active", False):
|
|
170
|
+
return self
|
|
171
|
+
# First time from here
|
|
172
|
+
self.to_doc()
|
|
173
|
+
# Mark active for the nested validation
|
|
174
|
+
nested_ctx = dict(ctx)
|
|
175
|
+
nested_ctx["_roundtrip_active"] = True
|
|
176
|
+
|
|
177
|
+
dumped = self.dump_raw()
|
|
178
|
+
|
|
179
|
+
# IMPORTANT: pass context so nested validation sees the flag
|
|
180
|
+
again = self.__class__.model_validate(dumped, context=nested_ctx)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
# Optional: assert equivalence (pick your definition)
|
|
184
|
+
if again != self:
|
|
185
|
+
raise ValueError("Roundtrip invariant failed: dump->validate changed the model")
|
|
186
|
+
|
|
187
|
+
return self
|
|
188
|
+
def to_doc(self):
|
|
189
|
+
"""Model to llm one-way serializer with manual slicing logic, can refactor using sliced view
|
|
190
|
+
with some token saving logic.
|
|
191
|
+
"""
|
|
192
|
+
target = self.refined_version or self
|
|
193
|
+
ocr_cluster = target.OCR_text_clusters
|
|
194
|
+
non_ocr_cluster = target.non_text_objects
|
|
195
|
+
if target.contains_table:
|
|
196
|
+
id_sorted_text_cluster = []
|
|
197
|
+
cluster_numbers = (i.cluster_number for i in ocr_cluster)
|
|
198
|
+
assert len(set(cluster_numbers)) == len(ocr_cluster)
|
|
199
|
+
cluster_lookup_by_number = {i.cluster_number : i for i in (ocr_cluster + non_ocr_cluster#+ target.signature_blocks
|
|
200
|
+
)}
|
|
201
|
+
for i in target.meaningful_ordering:
|
|
202
|
+
c_p = cluster_lookup_by_number.get(i)
|
|
203
|
+
if c_p is None:
|
|
204
|
+
raise KeyError(f"{i} does not exist")
|
|
205
|
+
cluster_dump: dict = c_p.model_dump()
|
|
206
|
+
cluster_dump.pop("cluster_number")
|
|
207
|
+
id_sorted_text_cluster.append(cluster_dump)
|
|
208
|
+
others = (set(cluster_numbers) - set(target.meaningful_ordering))
|
|
209
|
+
for i in others:
|
|
210
|
+
cluster_dump: dict
|
|
211
|
+
cluster_dump = cluster_lookup_by_number[i].model_dump()
|
|
212
|
+
cluster_dump.pop("cluster_number")
|
|
213
|
+
id_sorted_text_cluster.append(cluster_dump)
|
|
214
|
+
c_return = {}
|
|
215
|
+
c_return['pdf_page_num'] = self.pdf_page_num
|
|
216
|
+
c_return['printed_page_number'] = self.printed_page_number
|
|
217
|
+
c_return['OCR_text_clusters'] = id_sorted_text_cluster
|
|
218
|
+
c_return['contains_table'] = self.contains_table
|
|
219
|
+
return c_return
|
|
220
|
+
else:
|
|
221
|
+
# isSorted = True
|
|
222
|
+
expected_next = 0
|
|
223
|
+
i_ocr = 0
|
|
224
|
+
# i_sig = 0
|
|
225
|
+
i_non_ocr = 0
|
|
226
|
+
ocr_clus_nums = sorted([i.cluster_number for i in ocr_cluster])
|
|
227
|
+
non_clus_nums = sorted([i.cluster_number for i in non_ocr_cluster])
|
|
228
|
+
# sig_clus_nums = sorted([i.cluster_number for i in target.signature_blocks])
|
|
229
|
+
# expected_next = min(ocr_clus_nums + sig_clus_nums)
|
|
230
|
+
if ocr_clus_nums:
|
|
231
|
+
start_num = min(ocr_clus_nums + non_clus_nums)
|
|
232
|
+
assert start_num in [0, 1], "only allow 0-indexed based or 1-indexed based cluster numbers"
|
|
233
|
+
expected_next = start_num
|
|
234
|
+
while expected_next < start_num + (len(ocr_clus_nums) + len(non_clus_nums)):
|
|
235
|
+
if (i_ocr <len(ocr_cluster)) and expected_next == ocr_cluster[i_ocr].cluster_number:
|
|
236
|
+
i_ocr += 1
|
|
237
|
+
# elif (i_sig < len(target.signature_blocks)) and expected_next == target.signature_blocks[i_sig].cluster_number:
|
|
238
|
+
# i_sig += 1
|
|
239
|
+
elif (i_non_ocr < len(non_ocr_cluster)) and expected_next == non_ocr_cluster[i_non_ocr].cluster_number:
|
|
240
|
+
i_non_ocr += 1
|
|
241
|
+
else:
|
|
242
|
+
raise Exception(f'expected index {expected_next} not found in both ocr_cluster nor signature_blocks')
|
|
243
|
+
expected_next += 1
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
id_sorted_text_cluster = sorted(ocr_cluster, key=lambda x : x.cluster_number )
|
|
247
|
+
assert len(id_sorted_text_cluster) == len(set(i.cluster_number for i in id_sorted_text_cluster))
|
|
248
|
+
is_normal = True
|
|
249
|
+
shift = 0 # try zero indexing sanity
|
|
250
|
+
c : TextCluster #| TextCluster_yolo_bb
|
|
251
|
+
for i, c in enumerate(id_sorted_text_cluster):
|
|
252
|
+
|
|
253
|
+
if c.cluster_number != i:
|
|
254
|
+
is_normal = False
|
|
255
|
+
break
|
|
256
|
+
c2: TextCluster #| TextCluster_yolo_bb
|
|
257
|
+
if not is_normal:
|
|
258
|
+
is_normal = True
|
|
259
|
+
shift = -1 # try one-indexing sanity
|
|
260
|
+
for i, c2 in enumerate(id_sorted_text_cluster):
|
|
261
|
+
if c2.cluster_number != i+1:
|
|
262
|
+
is_normal = False
|
|
263
|
+
break
|
|
264
|
+
# sig_num = set([i.cluster_number for i in self.signature_blocks])
|
|
265
|
+
assert set(self.meaningful_ordering) <= set(i.cluster_number for i in (self.OCR_text_clusters))
|
|
266
|
+
if is_normal:
|
|
267
|
+
# is sorted list where index is order + shift
|
|
268
|
+
texts = '\n'.join(id_sorted_text_cluster[i+shift].text for i in self.meaningful_ordering)
|
|
269
|
+
else:
|
|
270
|
+
# general case
|
|
271
|
+
tcd = {x.cluster_number:x for x in id_sorted_text_cluster}
|
|
272
|
+
texts = '\n'.join(tcd[i].text for i in self.meaningful_ordering)
|
|
273
|
+
c_return = {}
|
|
274
|
+
c_return['pdf_page_num'] = self.pdf_page_num
|
|
275
|
+
c_return['printed_page_number'] = self.printed_page_number
|
|
276
|
+
c_return['text'] = texts
|
|
277
|
+
return c_return
|