graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,30 @@
1
+
2
+ from typing import Dict, Literal, Any
3
+ from .semantic_document_splitting_layerwise_edits import parse_doc
4
+
5
+ def text_to_ocr_format(text: str, filename: str = "input_text") -> Dict:
6
+ """
7
+ Wraps a raw string into the expected OCR dictionary format with a single dummy cluster.
8
+ """
9
+ return {
10
+ filename: [
11
+ {
12
+ "pdf_page_num": 1,
13
+ "OCR_text_clusters": [
14
+ {
15
+ "text": text,
16
+ "bb_x_min": 0, "bb_y_min": 0, "bb_x_max": 1000, "bb_y_max": 1000,
17
+ "cluster_number": 0
18
+ }
19
+ ],
20
+ "non_text_objects": []
21
+ }
22
+ ]
23
+ }
24
+
25
+ def parse_doc_text(text: str, doc_id: str = "text_doc", parsing_mode: Literal["snippet", "delimiter"] = "snippet", max_depth: int = 10):
26
+ """
27
+ Convenience function to parse a raw text string.
28
+ """
29
+ raw_doc = text_to_ocr_format(text)
30
+ return parse_doc(doc_id, raw_doc, parsing_mode=parsing_mode, max_depth=max_depth)
File without changes
@@ -0,0 +1,37 @@
1
+ from concurrent.futures import ThreadPoolExecutor
2
+ from threading import Semaphore, Lock, Condition
3
+
4
+
5
+ class BoundedExecutor:
6
+ def __init__(self, max_workers=5, max_pending=10):
7
+ self._executor = ThreadPoolExecutor(max_workers=max_workers)
8
+ self._semaphore = Semaphore(max_pending)
9
+ self._lock = Lock()
10
+ self._condition = Condition(self._lock)
11
+ self._active_tasks = 0
12
+
13
+ def submit(self, fn, *args, **kwargs):
14
+ self._semaphore.acquire()
15
+
16
+ def wrapped_fn(*args, **kwargs):
17
+ with self._condition:
18
+ self._active_tasks += 1
19
+ try:
20
+ return fn(*args, **kwargs)
21
+ finally:
22
+ with self._condition:
23
+ self._active_tasks -= 1
24
+ self._condition.notify_all()
25
+ self._semaphore.release()
26
+
27
+ return self._executor.submit(wrapped_fn, *args, **kwargs)
28
+
29
+ def wait_for_all(self):
30
+ """Block until all submitted tasks have completed."""
31
+ with self._condition:
32
+ while self._active_tasks > 0:
33
+ self._condition.wait()
34
+
35
+ def shutdown(self, wait=True):
36
+ """Shut down the underlying ThreadPoolExecutor."""
37
+ self._executor.shutdown(wait=wait)
@@ -0,0 +1,405 @@
1
+
2
+ import logging
3
+ logger = logging.getLogger(__name__)
4
+ logger.addHandler(logging.NullHandler())
5
+ logger.debug("library loading")
6
+ from json import JSONDecodeError
7
+ import os
8
+ import pathlib
9
+ import pathspec
10
+ from typing import Iterable, Tuple, Optional, Generator, Any
11
+ from typing import Callable
12
+ from functools import partial
13
+
14
+ def bool2yn(maybe_bool: bool):
15
+ if type(maybe_bool) is bool:
16
+ return "Yes" if maybe_bool else "No"
17
+ else:
18
+ return maybe_bool
19
+
20
+ def recur_apply_json_inplace(json : dict | list | bool |str | int | float, fn: Callable):
21
+ if isinstance(json, dict):
22
+ for k, v in json.items():
23
+ json[k] = recur_apply_json_inplace(v, fn)
24
+
25
+ elif isinstance(json, list):
26
+ for i,v in enumerate(json):
27
+ json[i] = recur_apply_json_inplace(json[i], fn)
28
+ else:
29
+ return fn(json)
30
+ return json
31
+
32
+ # convert any boolean to yes no
33
+ json_bool_to_yn: Callable = partial(recur_apply_json_inplace, fn = bool2yn)
34
+
35
+ def nullable_concat(a: list| None, b: list| None) -> list | None:
36
+ if a is None and b is None:
37
+ return None
38
+ return (a or []) + (b or [])
39
+
40
+ def fast_walk(path, prune_condition=None):
41
+ stack = [path]
42
+ while stack:
43
+ current = stack.pop() # DFS, or use `.pop(0)` for BFS
44
+ try:
45
+ with os.scandir(current) as it:
46
+ dirs = []
47
+ files = []
48
+ for entry in it:
49
+ if entry.is_dir(follow_symlinks=False):
50
+ if prune_condition and prune_condition(entry.path):
51
+ continue
52
+ dirs.append(entry.path)
53
+ elif entry.is_file():
54
+ files.append(entry.path)
55
+ yield current, dirs, files
56
+ stack.extend(reversed(dirs)) # DFS
57
+ except PermissionError:
58
+ continue
59
+ def find_folders_two_levels_from_leaves_mem_optimized(root_path: str, required_level: int = 2) -> Generator[Tuple[str , list[str],list[str]], Any, None]:
60
+ """
61
+ Yields folders that are exactly two levels away from leaf nodes using a
62
+ highly memory-efficient iterator.
63
+
64
+ The memoization cache is actively pruned to keep memory usage proportional
65
+ to the filesystem's width, not its total size.
66
+
67
+ Args:
68
+ root_path: The absolute or relative path to the directory to start searching from.
69
+
70
+ Yields:
71
+ A path to a qualifying folder as soon as it is identified.
72
+ """
73
+ if not os.path.isdir(root_path):
74
+ print(f"Error: Provided path '{root_path}' is not a valid directory.")
75
+ return
76
+
77
+ memo = {}
78
+
79
+ # Using realpath normalizes paths (e.g., C:/Users vs C:\Users) and handles
80
+ # symlinks, making the cache keys consistent.
81
+ # real_root = os.path.realpath(root_path)
82
+
83
+ for dirpath, dirnames, filenames in os.walk(root_path, topdown=False):
84
+ no_solution = False
85
+ # 1. Get the depths of all children of the current `dirpath`.
86
+ child_depths = [0] * len(filenames) # Files are leaves, depth 0.
87
+
88
+ for dirname in dirnames:
89
+ child_path = os.path.join(dirpath, dirname)
90
+ child_depth = memo.get(child_path, -1)
91
+ if child_depth is None:
92
+ no_solution = True
93
+ child_depths = [None]
94
+ break
95
+ child_depths.append(memo.get(child_path, -1))
96
+
97
+ # 2. THE CORE LOGIC: Check if this directory qualifies.
98
+ # if child_depths and all(depth == required_level for depth in child_depths):
99
+ # # Yield the original path if it's the root, otherwise the current dirpath
100
+ # if dirpath == real_root:
101
+ # return # yield root_path
102
+ # else:
103
+ # yield dirpath, dirnames, filenames
104
+
105
+ # 3. Calculate and cache the depth for the current path.
106
+ if no_solution:
107
+ current_depth = None
108
+ else:
109
+ if len(set(child_depths)) > 1:
110
+ current_depth = None
111
+ else:
112
+
113
+ if not child_depths:
114
+ current_depth = 0
115
+ elif -1 in child_depths:
116
+ current_depth = -1
117
+ else:
118
+ # child_depths must have no None here
119
+ current_depth = 1 + child_depths[0] # type: ignore
120
+ if current_depth == required_level:
121
+ yield dirpath, dirnames, filenames
122
+ memo[dirpath] = current_depth
123
+
124
+ # 4. PRUNING STEP: Release memory by removing child depths.
125
+ # This is safe because we are done with this level and the parent
126
+ # only needs the depth of `dirpath`, which we just cached.
127
+ for dirname in dirnames:
128
+ child_path = os.path.join(dirpath, dirname)
129
+ memo.pop(child_path, None) # Safely remove the key
130
+ class RawFileLoader():
131
+ def __init__(self, env_flist_path: Optional[str] = None,
132
+ allow_file_list: None | list[str] = None,
133
+ allow_file_list_ref: Optional[str] = None,
134
+ max_num_file = float('inf'), oldest_datetime = None, newest_datetime = None, root_folder_name : str | None= None,
135
+ # in_folder_name :Optional[str] = None,
136
+ walk_root:Optional[str] = None, compare_root:Optional[str] = None,
137
+ include = None,
138
+ bucket_blob_connection_str = None,
139
+ file_walker_callback: Callable[[str, int], Generator[Tuple[str , list[str],list[str]], Any, None]] | None = None,
140
+ pattern = None,
141
+ allow_startwith_relative_paths = False,
142
+ filtering_callbacks : Optional[list[Callable]] = None):
143
+ """file loader to either backward support for local folder loading behaviour, or cloud bucket/ blob stoages
144
+
145
+ Args:
146
+ env_var_name (_type_): environment variable that contains the path to the file list
147
+ allow_file_list (None | list[str] | str): file path that contains the file information or a list of filenames directly passed into
148
+ pattern re.compile returned pattern to apply on the path
149
+ filtering_callbacks: that apply to each path, when True, it allows continue, when False, it continue to next file
150
+ """
151
+
152
+ assert not ((bucket_blob_connection_str is not None) and (env_flist_path is not None))
153
+ self.allow_startwith_relative_paths = allow_startwith_relative_paths
154
+ self.pattern = pattern
155
+ self.include = include or []
156
+ self.oldest_datetime = oldest_datetime
157
+ self.newest_datetime = newest_datetime
158
+ self.bucket_blob_connection_str = bucket_blob_connection_str
159
+ self.file_walker_callback: Optional[Callable] = file_walker_callback
160
+ if root_folder_name is None:
161
+ root_folder_name = os.getcwd()
162
+ if walk_root is None:
163
+ walk_root = os.getcwd()
164
+ self.walk_root = walk_root
165
+ if compare_root is None:
166
+ compare_root = walk_root
167
+ self.compare_root = compare_root
168
+ self.max_num_file = max_num_file
169
+ allowed_relative_paths: Optional[list[str]] = None
170
+
171
+ if allow_file_list_ref is not None:
172
+ if allow_file_list is None:
173
+ allow_file_list = []
174
+ if isinstance(allow_file_list_ref, str):
175
+ need_attempt_readlines = False
176
+ try:
177
+ if allow_file_list_ref.endswith('.json'):
178
+ import json
179
+ with open(allow_file_list_ref, 'r') as f:
180
+ flist = json.load(f)
181
+ if isinstance(flist, list):
182
+ pass
183
+ else:
184
+ raise Exception('only list-like json is accepted')
185
+ allow_file_list = flist # type: ignore
186
+ else:
187
+ need_attempt_readlines = True
188
+ except JSONDecodeError as e:
189
+ need_attempt_readlines = True
190
+
191
+ except Exception as e:
192
+ raise
193
+ if need_attempt_readlines:
194
+ with open(allow_file_list_ref, 'r') as f:
195
+ flist = f.readlines()
196
+
197
+ allow_file_list = flist # type: ignore
198
+ else:
199
+ raise(ValueError("allow_file_list does not either provide a path to file containing file list "
200
+ "or directly passing file list"))
201
+
202
+ # allow combine flist from argument with env specifed paths list concatenated
203
+ if env_flist_path is not None:
204
+ if allowed_relative_paths is None:
205
+ allowed_relative_paths = []
206
+ if flist:=os.environ.get(env_flist_path, None):
207
+ if flist is not None:
208
+ if os.path.exists(flist):
209
+ if flist.endswith('.txt'):
210
+ with open(flist, 'r', encoding='utf-8') as f:
211
+ allowed_relative_paths = [i.strip() for i in f.readlines()]
212
+
213
+ if flist.endswith('.csv'):
214
+ with open(flist, 'r') as f:
215
+ for ln in f.readlines():
216
+ allowed_relative_paths.append(os.path.join(*(i.strip() for i in ln.split(','))))
217
+ elif flist.endswith('.xls') or flist.endswith('.xlsx'):
218
+ from pandas import read_excel
219
+ df = read_excel(flist)
220
+ allowed_relative_paths = []
221
+ for i, row in df.iterrows():
222
+ new_path = os.path.join(*(row[:3]))
223
+ allowed_relative_paths.append(new_path)
224
+ if allowed_relative_paths is not None:
225
+ if allow_file_list is None:
226
+ allow_file_list = []
227
+
228
+ allow_file_list = allow_file_list + allowed_relative_paths
229
+ self.allow_file_list = allow_file_list
230
+ self.filtering_callbacks = filtering_callbacks or []
231
+ self.resolve_paths = False
232
+ if self.resolve_paths:
233
+ self.check_allowed_relative_path()
234
+ pass
235
+ def check_allowed_relative_path(self, paths= None, ):
236
+ # ensure allowed
237
+ allow_file_list = nullable_concat(self.allow_file_list, paths)
238
+ if allow_file_list is None:
239
+ return
240
+ out_file_list = []
241
+ try:
242
+ for p in allow_file_list:
243
+ out_path = pathlib.Path(p).relative_to(self.compare_root)
244
+ out_file_list.append(str(out_path))
245
+ except ValueError as e:
246
+ if "is not in the subpath of" in str(e):
247
+ print("allow_file_list contains path outside of compare root")
248
+ raise
249
+ self.allow_file_list = out_file_list
250
+
251
+ def __iter__(self, leaf_only = False, file_non_exist_ok = False, include = None,
252
+ allowed_files: Optional[list[str]] = None,
253
+ # allowed_prefixes : Optional[list[str | int]] = None,
254
+ allowed_relative_paths: Optional[list[str]]= None,
255
+ ):
256
+ """Iterate through availble files
257
+
258
+ Args:
259
+ leaf_only (bool, optional): _description_. Defaults to True.
260
+ file_non_exist_ok (bool, optional): _description_. Defaults to False.
261
+ include (_type_, optional): _description_. Defaults to None.
262
+ allowed_files (Optional[list[str]], optional): _description_. Defaults to None.
263
+ allowed_relative_paths (Optional[list[str]], optional): _description_. Defaults to None.
264
+ allow_startwith_relative_paths also check if the file start with any of the allowed relative paths
265
+ Yields:
266
+ _type_: _description_
267
+ """
268
+ from collections import Counter
269
+ filter_stat = Counter()
270
+ if include is None:
271
+ include = include or self.include or set(['files'])#, ['files', "dirs"]
272
+ else:
273
+ include = set(include)
274
+ include.update(self.include)
275
+ count = 0
276
+ from datetime import datetime
277
+ if self.oldest_datetime is not None:
278
+ if type(self.oldest_datetime) is str:
279
+ time_threshold_dt = datetime.strptime(self.oldest_datetime, '%Y-%m-%d %H:%M')
280
+ else:
281
+ time_threshold_dt = self.oldest_datetime
282
+ else:
283
+ time_threshold_dt = None
284
+ if self.newest_datetime is not None:
285
+ if type(self.newest_datetime) is str:
286
+ time_threshold_upper_dt = datetime.strptime(self.newest_datetime, '%Y-%m-%d %H:%M')
287
+ else:
288
+ time_threshold_upper_dt = self.newest_datetime
289
+ else:
290
+ time_threshold_upper_dt = None
291
+ nullable_allowed_relative_paths_set = []
292
+ nullable_allowed_relative_paths_set_t = nullable_concat(allowed_relative_paths, self.allow_file_list)
293
+ if nullable_allowed_relative_paths_set_t is not None:
294
+ nullable_allowed_relative_paths_set_t = set(nullable_allowed_relative_paths_set_t)
295
+ if len(nullable_allowed_relative_paths_set_t) > 0:
296
+ nullable_allowed_relative_paths_set = list(nullable_allowed_relative_paths_set_t)
297
+ else:
298
+ nullable_allowed_relative_paths_set = None
299
+ # debug_list = []
300
+ if self.file_walker_callback is None:
301
+ if self.bucket_blob_connection_str is None:
302
+ file_walker = os.walk(self.walk_root)
303
+ else:
304
+ raise NotImplementedError('blob or bucket stage file walker not implemented')
305
+ else:
306
+ file_walker = self.file_walker_callback(self.walk_root)
307
+ spec = None
308
+ if nullable_allowed_relative_paths_set is not None:
309
+ if self.allow_startwith_relative_paths:
310
+
311
+ prefixes = [pathlib.Path(i).as_posix() for i in nullable_allowed_relative_paths_set]
312
+ # normalized_prefixes = [p.rstrip("/\\") + "/" for p in raw_prefixes]
313
+ spec = pathspec.PathSpec.from_lines("gitwildmatch", prefixes)
314
+ for root, dirs, files in file_walker:
315
+ # print(root)
316
+ if leaf_only:
317
+ if dirs != []:
318
+ filter_stat.update(['dir != []'])
319
+ continue
320
+ if allowed_files:
321
+ if pathlib.Path(root).parts[-1] in allowed_files:
322
+ pass
323
+ else:
324
+ filter_stat.update(['pathlib.Path(root).parts[-1] in allowed_files'])
325
+ continue
326
+ to_iter = []
327
+ if "files" in include:
328
+ to_iter += files
329
+ if "dirs" in include:
330
+ to_iter += dirs
331
+ if "root" in include:
332
+ to_iter += ['.']
333
+ for f in to_iter:
334
+ if self.pattern:
335
+ if self.pattern.match(os.path.join(root, f)):
336
+ pass
337
+ else:
338
+ filter_stat.update(['self.pattern.match(os.path.join(root, f))'])
339
+ continue
340
+ f_rel = str(pathlib.Path(os.path.join(root, f)).relative_to(self.compare_root))
341
+ if spec is not None:
342
+ matched = True
343
+ matched = spec.match_file(pathlib.Path(f_rel).as_posix())
344
+
345
+ if not matched:
346
+ filter_stat.update(['not spec.match_file(pathlib.Path(f_rel).as_posix())'])
347
+ continue
348
+ else:
349
+ if nullable_allowed_relative_paths_set is None:
350
+ pass
351
+ elif f_rel in nullable_allowed_relative_paths_set:
352
+ pass
353
+ else:
354
+ filter_stat.update(['f_rel in allowed_relative_paths'])
355
+ continue
356
+ filtered = False
357
+ for cb in self.filtering_callbacks:
358
+
359
+ if cb(os.path.join(self.compare_root, f_rel)):
360
+ pass
361
+ else:
362
+ filter_stat.update([f'callback filtered by {cb}'])
363
+ filtered = True
364
+ break
365
+ if filtered:
366
+ continue
367
+ creation_timestamp = os.path.getctime(os.path.join(self.compare_root, f_rel))
368
+ creation_time = datetime.fromtimestamp(creation_timestamp)
369
+ if time_threshold_dt is not None:
370
+ if not (creation_time > time_threshold_dt):
371
+ filter_stat.update(['not (creation_time > time_threshold_dt)'])
372
+ continue
373
+ if time_threshold_upper_dt is not None:
374
+ if not (creation_time < time_threshold_upper_dt):
375
+ filter_stat.update(['not (creation_time < time_threshold_upper_dt)'])
376
+ continue
377
+ count+= 1
378
+ # debug_list.append(f_rel)
379
+ yield f_rel
380
+ if count >= self.max_num_file:
381
+ print(filter_stat.most_common())
382
+ return
383
+ pass
384
+ print(filter_stat.most_common())
385
+ pass
386
+
387
+ def filter_folder(folder_root = os.path.join(
388
+ "..", "odc_data", "split_pages"
389
+
390
+ ), min_page = 45, max_page: int | float = 55, first = 10, verbose = True):
391
+ folders = []
392
+ for root, dirs, files in os.walk(folder_root):
393
+ if dirs == []:
394
+ n_page = len([i for i in files if i.lower().endswith('.pdf')])
395
+
396
+ if n_page < max_page and min_page < n_page:
397
+ rel_path = pathlib.Path(root).relative_to(folder_root)
398
+ folders.append(str(rel_path))
399
+
400
+ if verbose:
401
+ print(folders)
402
+ if first is not None:
403
+ return (sorted(folders)[:first])
404
+ else:
405
+ return (sorted(folders))