graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,277 @@
1
+ """
2
+ Data Models & Schema Definitions
3
+ ================================
4
+
5
+ This module defines the core data models for the Optical Character Recognition (OCR) layer
6
+ and documents the implicit schema of the Resultant Knowledge Graph.
7
+
8
+ Physical Layer (OCR)
9
+ --------------------
10
+ - SplitPage: Represents a single page of a document.
11
+ - OCRClusterResponse: Contains detected text clusters and non-text objects.
12
+ - TextCluster: A bounding box with text content.
13
+ - text: str (The recognized text content)
14
+ - cluster_number: int (Unique ID on the page)
15
+ - bb_x_min, bb_x_max, bb_y_min, bb_y_max: float (Bounding box coordinates)
16
+
17
+ [PDF/Image] -> [OCR Service] -> [SplitPage] -> [TextCluster]
18
+
19
+ Resultant Knowledge Graph Structure
20
+ -----------------------------------
21
+ The semantic parsing pipeline (`kg_doc_parser.semantic_document_splitting_layerwise_edits`) transforms
22
+ these physical models into a Knowledge Graph.
23
+
24
+ [SemanticNode] (Entity)
25
+ |
26
+ +--- mentions ---+
27
+ | |
28
+ | [TextCluster] (Grounding)
29
+ |
30
+ +--- HAS_CHILD --> [SemanticNode] (Sub-entity)
31
+
32
+ Graph Schema (Nodes & Edges)
33
+ ----------------------------
34
+ Nodes (Entities):
35
+ - id: UUID
36
+ - label: Title/Heading
37
+ - type: "entity"
38
+ - subtype: "TEXT_FLOW" | "KEY_VALUE_PAIR" | "TABLE"
39
+ - mentions: Link to source `TextCluster`s (provenance)
40
+
41
+ Edges (Relationships):
42
+ - id: UUID
43
+ - source_id: Parent Node UUID
44
+ - target_id: Child Node UUID
45
+ - relation: "HAS_CHILD"
46
+ - type: "relationship"
47
+ """
48
+ if True:
49
+ import logging
50
+ import os
51
+ logger = logging.getLogger(__name__)
52
+ logger.addHandler(logging.NullHandler())
53
+ logger.debug("loading models")
54
+ from typing import List, Literal, Optional, Dict, Any, Type, Union, Annotated, ClassVar
55
+ try:
56
+ from typing import TypeAlias
57
+ except ImportError: # pragma: no cover
58
+ from typing_extensions import TypeAlias
59
+ from pydantic import BaseModel, Field, model_validator, field_validator, ValidationInfo
60
+
61
+ from pydantic_extension.model_slicing import (ModeSlicingMixin, NotMode, FrontendField, BackendField, LLMField,
62
+ DtoType,
63
+ BackendType,
64
+ FrontendType,
65
+ LLMType,
66
+ use_mode)
67
+ from pydantic_extension.model_slicing.mixin import ExcludeMode, DtoField
68
+ JsonPrimitive = Union[str, int, float, bool, None]
69
+ #========================= OCR DOC
70
+
71
+ # pre-validation model
72
+ class box_2d(BaseModel): # text box for ocr, the id here is not a database unique id but a unique id for identified object
73
+ box_2d: list[int] = Field(description = 'box y min, x min, y max and x max')
74
+ label : str = Field(description = 'text in the box')
75
+ id: int = Field(description = 'id of the text box in the page, autoincrement from 0')
76
+ class NonText_box_2d(BaseModel): # text box for ocr, the id here is not a database unique id but a unique id for identified object
77
+ """Recognised meaningful objects other than OCR characters, include image, figures. """
78
+ description: str = Field(description='the description or summary of the non-OCR object')
79
+ box_2d: list[int] = Field(description = 'box y min, x min, y max and x max')
80
+ id: int = Field(description="per page unique number of the cluster, starting from 0")
81
+
82
+ # post validation model
83
+ class TextCluster(ModeSlicingMixin, BaseModel):
84
+ """a text cluster along with spatial information"""
85
+ text: DtoType[str] = Field(description='the text content of the text cluster')
86
+ bb_x_min: DtoType[float] = Field(description='the bounding box x min in pixel coordinate of the text_cluster. ')
87
+ bb_x_max: DtoType[float] = Field(description='the bounding box x max in pixel coordinate of the text_cluster. ')
88
+ bb_y_min: DtoType[float] = Field(description='the bounding box y min in pixel coordinate of the text_cluster. ')
89
+ bb_y_max: DtoType[float] = Field(description='the bounding box y max in pixel coordinate of the text_cluster. ')
90
+ cluster_number: int = Field(description="per page unique number of the cluster, starting from 0")
91
+ class NonTextCluster(ModeSlicingMixin, BaseModel):
92
+ """Recognised meaningful objects other than OCR characters, include image, figures. """
93
+ description: DtoType[str] = Field(description='the description or summary of the non-OCR object')
94
+ bb_x_min: DtoType[float] = Field(description='the bounding box x min in pixel coordinate of the non-OCR object. ')
95
+ bb_x_max: DtoType[float] = Field(description='the bounding box x max in pixel coordinate of the non-OCR object. ')
96
+ bb_y_min: DtoType[float] = Field(description='the bounding box y min in pixel coordinate of the non-OCR object. ')
97
+ bb_y_max: DtoType[float] = Field(description='the bounding box y max in pixel coordinate of the non-OCR object. ')
98
+ cluster_number: int = Field(description="per page unique number of the cluster, starting from 0")
99
+
100
+ class OCRClusterResponse(ModeSlicingMixin, BaseModel):
101
+ """id/ cluster number must be all unique, for example, one of the ocr boxes_2d used id='1',
102
+ the first image box id (cluster numebr) will be '2', the next signature will be '3' """
103
+ OCR_text_clusters: DtoType[list[TextCluster]] = Field(description="the OCR text results. Share cluster number uniqueness with non-OCR objects. Include emoji or unicode text")
104
+ non_text_objects: DtoType[list[NonTextCluster]] = Field(description="the non-OCR object results. Share cluster number uniqueness with OCR texts. ")
105
+ is_empty_page: DtoType[Optional[bool]] = Field(default = False, description="true if the whole page is empty without recognisable text.")
106
+ printed_page_number: DtoType[Optional[str]] = Field(description='the page number identified from OCR texts, can be in form of roman numerals such as "i", "ii", "iii", "iv"...; '
107
+ 'Arabic numeral such as 1, 2, 3... or letter such as "a", "b", "c"...\n'
108
+ 'Sometimes the are surrounded by symbols such as "- 1 -", "- 2 -"'
109
+ r"Can be null/none if there is no page order assigned and printed and found in the scanned texts. Do not assign page number. Only use page number found.")
110
+ meaningful_ordering : DtoType[list[int]] = Field(description="The correct meaningful ordering of the identified text clusters. Must cover all OCR_text_clusters once and only once. ")
111
+ page_x_min : DtoType[float]=Field(description='the page x min in pixel coordinate. ')
112
+ page_x_max : DtoType[float]=Field(description='the page x max in pixel coordinate. ')
113
+ page_y_min : DtoType[float]=Field(description='the page y min in pixel coordinate. ')
114
+ page_y_max : DtoType[float]=Field(description='the page y max in pixel coordinate. ')
115
+ estimated_rotation_degrees : DtoType[float]=Field(description='the page estimated rotation degree using right hand rule. ')
116
+ incomplete_words_on_edge: DtoType[bool] = Field(description='If there is any text being incomplete due to the scan does not scan the edges properly. ')
117
+ incomplete_text: DtoType[bool] = Field(description='Any incomplete text')
118
+ data_loss_likelihood: DtoType[float] = Field(description='The likelihood (range from 0.0 to 1.0 inclusive) that the page has lost information by missing the scan data on the edges of the page.' )
119
+ scan_quality: DtoType[Literal['low', 'medium', 'high']] = Field(description='The image quality of the scan. All qualities exclude signatures. '
120
+ '"low", "medium" or "high". '
121
+ 'low: text barely legible. medium: Legible with non smooth due to pixelation. high: texts are easily and highly identifiable. ' )
122
+ contains_table: DtoType[bool] = Field(description='Whether this page contains table. ')
123
+
124
+ @model_validator(mode='after')
125
+ def check_cluster_meaningful_ordering_agreement(self):
126
+ assert bool(self.is_empty_page) ^ (len(self.OCR_text_clusters) > 0), f"is_empty_page value {self.is_empty_page} disagree with OCR_text_clusters len={len(self.OCR_text_clusters)}"
127
+ overlap_id = set(i.cluster_number for i in self.OCR_text_clusters).intersection(set(i.cluster_number for i in self.non_text_objects))
128
+ if overlap_id:
129
+ raise ValueError(f"cluster number from non_text_objects block and ocr text blocks must be ALL distinct. Overlapped ids: {list(overlap_id)}")
130
+ try:
131
+ if not (len(self.meaningful_ordering) == len(set(self.meaningful_ordering))): # <= len(self.OCR_text_clusters)):
132
+ raise ValueError("meaningful_order must cover each text cluster at most once")
133
+ except Exception as e:
134
+ raise e
135
+ return self
136
+
137
+
138
+
139
+ class OCRClusterResponseBc(OCRClusterResponse):
140
+ # backward compability layer, can use by overriding fields with union of previous versions
141
+
142
+ pass
143
+ # OCR_text_clusters: list[TextCluster | TextCluster_yolo_bb] = Field(description="the OCR text results, " # type: ignore
144
+ # "prefer min max bounding box to centre width height style")
145
+
146
+ class SplitPageMeta(BaseModel):
147
+ ocr_model_name: str = Field(description="model that perform OCR")
148
+ ocr_datetime: float = Field(description="unix timestamp when ocr is performed")
149
+ ocr_json_version: str = Field(description = "the model does the OCR")
150
+ @field_validator('ocr_json_version', mode = "before")
151
+ def version_to_str(cls, v):
152
+ return str(v)
153
+ class SplitPage(OCRClusterResponseBc):
154
+ # model not for LLM response
155
+ pdf_page_num: int
156
+ metadata: SplitPageMeta
157
+ refined_version: Optional[OCRClusterResponse[DtoField]] = Field(default = None, description = "refined processed/ grouped/ merged version of ocr text clusters. ")
158
+ def model_dump(self, *arg, **kwarg):
159
+ return self.to_doc()
160
+ def dump_raw(self, *arg, **kwarg):
161
+ return super(SplitPage, self).model_dump(exclude = ["refined_version"], *arg, **kwarg)
162
+ def dump_supercede_parse(self, *arg, **kwarg):
163
+ return super(SplitPage, self).model_dump(exclude = ["refined_version", "metadata"], *arg, **kwarg)
164
+ @model_validator(mode="after")
165
+ def roundtrip_invariant(self, info: ValidationInfo) -> "SplitPage":
166
+ # Context may be None if caller didn't pass it
167
+ ctx = info.context or {}
168
+ # Re-entrancy guard
169
+ if ctx.get("_roundtrip_active", False):
170
+ return self
171
+ # First time from here
172
+ self.to_doc()
173
+ # Mark active for the nested validation
174
+ nested_ctx = dict(ctx)
175
+ nested_ctx["_roundtrip_active"] = True
176
+
177
+ dumped = self.dump_raw()
178
+
179
+ # IMPORTANT: pass context so nested validation sees the flag
180
+ again = self.__class__.model_validate(dumped, context=nested_ctx)
181
+
182
+
183
+ # Optional: assert equivalence (pick your definition)
184
+ if again != self:
185
+ raise ValueError("Roundtrip invariant failed: dump->validate changed the model")
186
+
187
+ return self
188
+ def to_doc(self):
189
+ """Model to llm one-way serializer with manual slicing logic, can refactor using sliced view
190
+ with some token saving logic.
191
+ """
192
+ target = self.refined_version or self
193
+ ocr_cluster = target.OCR_text_clusters
194
+ non_ocr_cluster = target.non_text_objects
195
+ if target.contains_table:
196
+ id_sorted_text_cluster = []
197
+ cluster_numbers = (i.cluster_number for i in ocr_cluster)
198
+ assert len(set(cluster_numbers)) == len(ocr_cluster)
199
+ cluster_lookup_by_number = {i.cluster_number : i for i in (ocr_cluster + non_ocr_cluster#+ target.signature_blocks
200
+ )}
201
+ for i in target.meaningful_ordering:
202
+ c_p = cluster_lookup_by_number.get(i)
203
+ if c_p is None:
204
+ raise KeyError(f"{i} does not exist")
205
+ cluster_dump: dict = c_p.model_dump()
206
+ cluster_dump.pop("cluster_number")
207
+ id_sorted_text_cluster.append(cluster_dump)
208
+ others = (set(cluster_numbers) - set(target.meaningful_ordering))
209
+ for i in others:
210
+ cluster_dump: dict
211
+ cluster_dump = cluster_lookup_by_number[i].model_dump()
212
+ cluster_dump.pop("cluster_number")
213
+ id_sorted_text_cluster.append(cluster_dump)
214
+ c_return = {}
215
+ c_return['pdf_page_num'] = self.pdf_page_num
216
+ c_return['printed_page_number'] = self.printed_page_number
217
+ c_return['OCR_text_clusters'] = id_sorted_text_cluster
218
+ c_return['contains_table'] = self.contains_table
219
+ return c_return
220
+ else:
221
+ # isSorted = True
222
+ expected_next = 0
223
+ i_ocr = 0
224
+ # i_sig = 0
225
+ i_non_ocr = 0
226
+ ocr_clus_nums = sorted([i.cluster_number for i in ocr_cluster])
227
+ non_clus_nums = sorted([i.cluster_number for i in non_ocr_cluster])
228
+ # sig_clus_nums = sorted([i.cluster_number for i in target.signature_blocks])
229
+ # expected_next = min(ocr_clus_nums + sig_clus_nums)
230
+ if ocr_clus_nums:
231
+ start_num = min(ocr_clus_nums + non_clus_nums)
232
+ assert start_num in [0, 1], "only allow 0-indexed based or 1-indexed based cluster numbers"
233
+ expected_next = start_num
234
+ while expected_next < start_num + (len(ocr_clus_nums) + len(non_clus_nums)):
235
+ if (i_ocr <len(ocr_cluster)) and expected_next == ocr_cluster[i_ocr].cluster_number:
236
+ i_ocr += 1
237
+ # elif (i_sig < len(target.signature_blocks)) and expected_next == target.signature_blocks[i_sig].cluster_number:
238
+ # i_sig += 1
239
+ elif (i_non_ocr < len(non_ocr_cluster)) and expected_next == non_ocr_cluster[i_non_ocr].cluster_number:
240
+ i_non_ocr += 1
241
+ else:
242
+ raise Exception(f'expected index {expected_next} not found in both ocr_cluster nor signature_blocks')
243
+ expected_next += 1
244
+
245
+
246
+ id_sorted_text_cluster = sorted(ocr_cluster, key=lambda x : x.cluster_number )
247
+ assert len(id_sorted_text_cluster) == len(set(i.cluster_number for i in id_sorted_text_cluster))
248
+ is_normal = True
249
+ shift = 0 # try zero indexing sanity
250
+ c : TextCluster #| TextCluster_yolo_bb
251
+ for i, c in enumerate(id_sorted_text_cluster):
252
+
253
+ if c.cluster_number != i:
254
+ is_normal = False
255
+ break
256
+ c2: TextCluster #| TextCluster_yolo_bb
257
+ if not is_normal:
258
+ is_normal = True
259
+ shift = -1 # try one-indexing sanity
260
+ for i, c2 in enumerate(id_sorted_text_cluster):
261
+ if c2.cluster_number != i+1:
262
+ is_normal = False
263
+ break
264
+ # sig_num = set([i.cluster_number for i in self.signature_blocks])
265
+ assert set(self.meaningful_ordering) <= set(i.cluster_number for i in (self.OCR_text_clusters))
266
+ if is_normal:
267
+ # is sorted list where index is order + shift
268
+ texts = '\n'.join(id_sorted_text_cluster[i+shift].text for i in self.meaningful_ordering)
269
+ else:
270
+ # general case
271
+ tcd = {x.cluster_number:x for x in id_sorted_text_cluster}
272
+ texts = '\n'.join(tcd[i].text for i in self.meaningful_ordering)
273
+ c_return = {}
274
+ c_return['pdf_page_num'] = self.pdf_page_num
275
+ c_return['printed_page_number'] = self.printed_page_number
276
+ c_return['text'] = texts
277
+ return c_return