graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,286 @@
1
+ if True:
2
+ import logging
3
+ import logging.handlers
4
+ import os
5
+ logger = logging.getLogger(__name__)
6
+ logger.addHandler(logging.NullHandler())
7
+
8
+ import shutil
9
+ import tempfile
10
+ from pdf2image import convert_from_path
11
+ import platform
12
+ import pathlib
13
+ from pypdf import PdfReader, PdfWriter
14
+ import threading
15
+ import pikepdf
16
+
17
+ from .utils.file_loaders import RawFileLoader
18
+
19
+ def batch_split_pdf(document_folder: pathlib.Path | str | None = None, outfolder_path: str | pathlib.Path = "split_pages", exists_ok = 'skip', allowed_relative_paths: list[str] | None= None,
20
+ file_loader : RawFileLoader | None = None):
21
+ cnt = 0
22
+ assert not ((document_folder is None) and (file_loader is None))
23
+ class old_walker_inplace(RawFileLoader):
24
+ def __init__(self, walk_root = None, compare_root = None):
25
+ self.walk_root: str | pathlib.Path
26
+ if walk_root is None:
27
+ if document_folder:
28
+ self.walk_root = document_folder
29
+ else:
30
+ raise Exception("unreachable")
31
+ else:
32
+ self.walk_root = walk_root
33
+ if compare_root is None:
34
+ self.compare_root = self.walk_root
35
+ def __iter__(self):
36
+
37
+ for root, dirs, files in os.walk(self.walk_root):
38
+ for f in files:
39
+ if dirs == []:
40
+ pass
41
+ else:
42
+ continue
43
+ input_pdf: pathlib.Path = pathlib.Path(root)/f
44
+ rel_path = input_pdf.relative_to(self.compare_root)
45
+ if allowed_relative_paths is not None:
46
+ if str(rel_path) in allowed_relative_paths:
47
+ pass
48
+ else:
49
+ continue
50
+ yield rel_path
51
+ if file_loader is None: # document_folder must not be None
52
+ if document_folder is None:
53
+ raise Exception("unreacheable")
54
+ else:
55
+ file_loader = old_walker_inplace(document_folder)
56
+ for rel_path in file_loader:
57
+
58
+ out_path = pathlib.Path(outfolder_path)/ pathlib.Path(rel_path)
59
+ os.makedirs(out_path.parent, exist_ok = True)
60
+ input_pdf = pathlib.Path(file_loader.compare_root) / rel_path
61
+ if str(input_pdf).lower().endswith('.pdf'):
62
+
63
+ try:
64
+ cnt += 1
65
+ print(f"{cnt} {str(input_pdf)}")
66
+ split_pdf(input_pdf, out_path.parent, exists_ok = exists_ok)
67
+ except Exception as e:
68
+ try:
69
+ logger.exception(e)
70
+ split_pdf_with_pikepdf(input_pdf, out_path.parent, exists_ok = exists_ok)
71
+ print(f"error for file {input_pdf}")
72
+ except Exception as e:
73
+ raise Exception("exhausted all pdf splitter")
74
+ elif str(input_pdf).endswith('.docx'):
75
+ continue
76
+
77
+
78
+ def split_pdf_with_pikepdf(input_pdf_path, output_folder, exists_ok='skip'):
79
+ # Ensure output folder exists
80
+ os.makedirs(output_folder, exist_ok=True)
81
+
82
+ # Open the PDF with pikepdf (automatically handles decryption if no password needed)
83
+ try:
84
+ pdf = pikepdf.open(input_pdf_path) # Add `password=""` if needed explicitly
85
+ except pikepdf.PasswordError:
86
+ raise ValueError("This PDF requires a password and could not be opened.")
87
+
88
+ num_pages = len(pdf.pages)
89
+ print(f"Total pages found: {num_pages}")
90
+ fname = pathlib.Path(input_pdf_path).parts[-1] # No extension
91
+ output_subfolder = os.path.join(output_folder, fname)
92
+ os.makedirs(output_subfolder, exist_ok=True)
93
+ existing_pages = [i for i in os.listdir(os.path.join(output_folder, fname)) if i.endswith('.pdf')]
94
+ if num_pages == len(existing_pages):
95
+ return
96
+
97
+
98
+
99
+
100
+ for i, page in enumerate(pdf.pages):
101
+ output_pdf_path = os.path.join(output_subfolder, f"page_{i + 1}.pdf")
102
+ if os.path.exists(output_pdf_path) and exists_ok == 'skip':
103
+ print(f"Skipped (already exists): {output_pdf_path}")
104
+ continue
105
+
106
+ # Create a new PDF with just one page
107
+ new_pdf = pikepdf.Pdf.new()
108
+ new_pdf.pages.append(page)
109
+
110
+ new_pdf.save(output_pdf_path)
111
+ print(f"Created: {output_pdf_path}")
112
+ return True
113
+ def split_pdf(input_pdf_path, output_folder, exists_ok = 'skip'):
114
+ # Ensure output folder exists
115
+
116
+
117
+ # Open the PDF file for reading
118
+ reader = PdfReader(input_pdf_path)
119
+ if reader.is_encrypted:
120
+ reader.decrypt("")
121
+ num_pages = len(reader.pages)
122
+ print(f"Total pages found: {num_pages}")
123
+ fname = pathlib.Path(input_pdf_path).name
124
+ os.makedirs(os.path.join(output_folder, fname), exist_ok=True)
125
+ existing_pages = [i for i in os.listdir(os.path.join(output_folder, fname)) if i.endswith('.pdf')]
126
+ if num_pages == len(existing_pages):
127
+ return # skip as all num pages exported
128
+ # Loop through each page and write it to a new PDF file
129
+ for i in range(num_pages):
130
+ writer = PdfWriter()
131
+ page = reader.pages[i]
132
+ writer.add_page(page)
133
+
134
+ output_pdf_path = os.path.join(output_folder, fname, f"page_{i + 1}.pdf")
135
+ os.makedirs(str(pathlib.Path(output_pdf_path).parent), exist_ok=True)
136
+ with open(output_pdf_path, "wb") as out_file:
137
+ writer.write(out_file)
138
+
139
+ print(f"Created: {output_pdf_path}")
140
+
141
+
142
+ def batch_pdf2png(document_folder, outfolder_path = None, exists_ok = 'skip', allowed_relative_paths = None,
143
+ loader = None):
144
+
145
+ """_summary_
146
+
147
+ Args:
148
+ document_folder (_type_): folder containing splitted document folder, each folder same name as pdf file
149
+ outfolder_path (str, optional): folder containing the folder F that contain png files. F is the name of the unsplitted pdf Defaults output same folder as splitted pdf folder.
150
+ exists_ok (str, optional): _description_. Defaults to 'skip'.
151
+ """
152
+ if outfolder_path is None:
153
+ outfolder_path = document_folder
154
+ from kg_doc_parser.utils.bounded_threadpool_executor import BoundedExecutor
155
+ bounded_executor = BoundedExecutor(max_workers= 2, max_pending= 5) # num of pdf
156
+ # for pdf_file in os.listdir(document_folder):
157
+ # pdf_file : str
158
+ if loader:
159
+ for rel_path in loader:
160
+ single_pdf2png(os.path.join(loader.compare_root, rel_path), os.path.join(outfolder_path, rel_path), exists_ok = exists_ok)
161
+ else:
162
+ for root, dirs, files in os.walk(document_folder):
163
+ if allowed_relative_paths is not None :
164
+ if dirs == []:
165
+ rel_path = str((pathlib.Path(root)).relative_to(document_folder))
166
+ if rel_path not in allowed_relative_paths:
167
+ continue
168
+ else:
169
+ continue
170
+
171
+ single_pdf2png(os.path.join(root), str(pathlib.Path(root)), exists_ok = exists_ok)
172
+
173
+
174
+ # OS-specific file locking
175
+ if platform.system() == "Windows":
176
+ import msvcrt
177
+ def lock_file(file_handle):
178
+ msvcrt.locking(file_handle.fileno(), msvcrt.LK_NBLCK, 1)
179
+
180
+ def unlock_file(file_handle):
181
+ try:
182
+ msvcrt.locking(file_handle.fileno(), msvcrt.LK_UNLCK, 1)
183
+ except Exception:
184
+ pass
185
+
186
+ else:
187
+ import fcntl
188
+ def lock_file(file_handle):
189
+ fcntl.flock(file_handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
190
+
191
+ def unlock_file(file_handle):
192
+ try:
193
+ fcntl.flock(file_handle, fcntl.LOCK_UN)
194
+ except Exception:
195
+ pass
196
+
197
+
198
+ def get_thread_safe_tempfile(suffix=".pdf"):
199
+ pid = os.getpid()
200
+ thread_name = threading.current_thread().name.replace(" ", "_")
201
+ prefix = f"{thread_name}_{pid}_"
202
+
203
+ fd, path = tempfile.mkstemp(suffix=suffix, prefix=prefix)
204
+ return fd, path
205
+ def process_pdf_page(pdf_path, output_path):
206
+ if os.path.exists(output_path):
207
+ print(f"Skipped {output_path} (already exists)")
208
+ return
209
+
210
+ tmp_fd = None
211
+ tmp_path = None
212
+ file_handle = None
213
+
214
+ try:
215
+ # Create temporary file path
216
+ tmp_fd, tmp_path = get_thread_safe_tempfile(suffix=".pdf")
217
+ os.close(tmp_fd) # Close the low-level fd to avoid conflicts
218
+
219
+ # Copy the PDF to the temporary file
220
+ shutil.copy2(pdf_path, tmp_path)
221
+
222
+ # Open the temp file to lock
223
+ file_handle = open(tmp_path, 'rb')
224
+ lock_file(file_handle)
225
+
226
+ # Convert using the temp file
227
+ images = convert_from_path(tmp_path, dpi=300, fmt='png')
228
+
229
+ # Save each page as image
230
+ for i, image in enumerate(images):
231
+ image.save(output_path, "PNG")
232
+ print(f"Saved {output_path}")
233
+
234
+ except Exception as e:
235
+ print(f"Error processing {pdf_path}: {e}")
236
+
237
+ finally:
238
+ # Always release lock and delete temp file
239
+ if file_handle:
240
+ unlock_file(file_handle)
241
+ file_handle.close()
242
+ if tmp_path and os.path.exists(tmp_path):
243
+ os.remove(tmp_path)
244
+
245
+ def single_pdf2png(fname, folder_path, exists_ok = 'skip'):
246
+ """_summary_
247
+
248
+ Args:
249
+ fname (_type_): the raw pdf file name before splitting
250
+ folder_path (_type_): the folder path containing the splitted pdf pages
251
+ exists_ok (str, optional): _description_. Defaults to 'skip'.
252
+
253
+ Raises:
254
+ PermissionError: _description_
255
+ """
256
+
257
+ from concurrent.futures import ThreadPoolExecutor
258
+
259
+
260
+ def convert(folder_path, fname):
261
+ out_pdf_folder = fname # output folder
262
+ page_pdf_files = [f for f in os.listdir(out_pdf_folder) if f.endswith('.pdf')]
263
+
264
+ # Define the number of worker threads
265
+ max_workers = 3
266
+
267
+ # Use ThreadPoolExecutor to process PDF files concurrently
268
+ with ThreadPoolExecutor(max_workers=max_workers) as executor:
269
+ futures = []
270
+ for page_fname in page_pdf_files:
271
+ pdf_path = os.path.join(folder_path, page_fname)
272
+ output_path = os.path.join(out_pdf_folder, f"{os.path.splitext(page_fname)[0]}.png")
273
+ if os.path.exists(output_path):
274
+ if exists_ok == "skip":
275
+ continue
276
+ elif exists_ok == "raise":
277
+ raise PermissionError(f"File {output_path} already exists.")
278
+ else: # exist_ok == "ok"
279
+ pass
280
+ futures.append(executor.submit(process_pdf_page, pdf_path, output_path))
281
+
282
+ # Wait for all futures to complete
283
+ # it starts running only when start to be iterated.
284
+ for future in futures:
285
+ future.result()
286
+ convert(folder_path, fname)