graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
kg_doc_parser/pdf2png.py
ADDED
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
if True:
|
|
2
|
+
import logging
|
|
3
|
+
import logging.handlers
|
|
4
|
+
import os
|
|
5
|
+
logger = logging.getLogger(__name__)
|
|
6
|
+
logger.addHandler(logging.NullHandler())
|
|
7
|
+
|
|
8
|
+
import shutil
|
|
9
|
+
import tempfile
|
|
10
|
+
from pdf2image import convert_from_path
|
|
11
|
+
import platform
|
|
12
|
+
import pathlib
|
|
13
|
+
from pypdf import PdfReader, PdfWriter
|
|
14
|
+
import threading
|
|
15
|
+
import pikepdf
|
|
16
|
+
|
|
17
|
+
from .utils.file_loaders import RawFileLoader
|
|
18
|
+
|
|
19
|
+
def batch_split_pdf(document_folder: pathlib.Path | str | None = None, outfolder_path: str | pathlib.Path = "split_pages", exists_ok = 'skip', allowed_relative_paths: list[str] | None= None,
|
|
20
|
+
file_loader : RawFileLoader | None = None):
|
|
21
|
+
cnt = 0
|
|
22
|
+
assert not ((document_folder is None) and (file_loader is None))
|
|
23
|
+
class old_walker_inplace(RawFileLoader):
|
|
24
|
+
def __init__(self, walk_root = None, compare_root = None):
|
|
25
|
+
self.walk_root: str | pathlib.Path
|
|
26
|
+
if walk_root is None:
|
|
27
|
+
if document_folder:
|
|
28
|
+
self.walk_root = document_folder
|
|
29
|
+
else:
|
|
30
|
+
raise Exception("unreachable")
|
|
31
|
+
else:
|
|
32
|
+
self.walk_root = walk_root
|
|
33
|
+
if compare_root is None:
|
|
34
|
+
self.compare_root = self.walk_root
|
|
35
|
+
def __iter__(self):
|
|
36
|
+
|
|
37
|
+
for root, dirs, files in os.walk(self.walk_root):
|
|
38
|
+
for f in files:
|
|
39
|
+
if dirs == []:
|
|
40
|
+
pass
|
|
41
|
+
else:
|
|
42
|
+
continue
|
|
43
|
+
input_pdf: pathlib.Path = pathlib.Path(root)/f
|
|
44
|
+
rel_path = input_pdf.relative_to(self.compare_root)
|
|
45
|
+
if allowed_relative_paths is not None:
|
|
46
|
+
if str(rel_path) in allowed_relative_paths:
|
|
47
|
+
pass
|
|
48
|
+
else:
|
|
49
|
+
continue
|
|
50
|
+
yield rel_path
|
|
51
|
+
if file_loader is None: # document_folder must not be None
|
|
52
|
+
if document_folder is None:
|
|
53
|
+
raise Exception("unreacheable")
|
|
54
|
+
else:
|
|
55
|
+
file_loader = old_walker_inplace(document_folder)
|
|
56
|
+
for rel_path in file_loader:
|
|
57
|
+
|
|
58
|
+
out_path = pathlib.Path(outfolder_path)/ pathlib.Path(rel_path)
|
|
59
|
+
os.makedirs(out_path.parent, exist_ok = True)
|
|
60
|
+
input_pdf = pathlib.Path(file_loader.compare_root) / rel_path
|
|
61
|
+
if str(input_pdf).lower().endswith('.pdf'):
|
|
62
|
+
|
|
63
|
+
try:
|
|
64
|
+
cnt += 1
|
|
65
|
+
print(f"{cnt} {str(input_pdf)}")
|
|
66
|
+
split_pdf(input_pdf, out_path.parent, exists_ok = exists_ok)
|
|
67
|
+
except Exception as e:
|
|
68
|
+
try:
|
|
69
|
+
logger.exception(e)
|
|
70
|
+
split_pdf_with_pikepdf(input_pdf, out_path.parent, exists_ok = exists_ok)
|
|
71
|
+
print(f"error for file {input_pdf}")
|
|
72
|
+
except Exception as e:
|
|
73
|
+
raise Exception("exhausted all pdf splitter")
|
|
74
|
+
elif str(input_pdf).endswith('.docx'):
|
|
75
|
+
continue
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def split_pdf_with_pikepdf(input_pdf_path, output_folder, exists_ok='skip'):
|
|
79
|
+
# Ensure output folder exists
|
|
80
|
+
os.makedirs(output_folder, exist_ok=True)
|
|
81
|
+
|
|
82
|
+
# Open the PDF with pikepdf (automatically handles decryption if no password needed)
|
|
83
|
+
try:
|
|
84
|
+
pdf = pikepdf.open(input_pdf_path) # Add `password=""` if needed explicitly
|
|
85
|
+
except pikepdf.PasswordError:
|
|
86
|
+
raise ValueError("This PDF requires a password and could not be opened.")
|
|
87
|
+
|
|
88
|
+
num_pages = len(pdf.pages)
|
|
89
|
+
print(f"Total pages found: {num_pages}")
|
|
90
|
+
fname = pathlib.Path(input_pdf_path).parts[-1] # No extension
|
|
91
|
+
output_subfolder = os.path.join(output_folder, fname)
|
|
92
|
+
os.makedirs(output_subfolder, exist_ok=True)
|
|
93
|
+
existing_pages = [i for i in os.listdir(os.path.join(output_folder, fname)) if i.endswith('.pdf')]
|
|
94
|
+
if num_pages == len(existing_pages):
|
|
95
|
+
return
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
for i, page in enumerate(pdf.pages):
|
|
101
|
+
output_pdf_path = os.path.join(output_subfolder, f"page_{i + 1}.pdf")
|
|
102
|
+
if os.path.exists(output_pdf_path) and exists_ok == 'skip':
|
|
103
|
+
print(f"Skipped (already exists): {output_pdf_path}")
|
|
104
|
+
continue
|
|
105
|
+
|
|
106
|
+
# Create a new PDF with just one page
|
|
107
|
+
new_pdf = pikepdf.Pdf.new()
|
|
108
|
+
new_pdf.pages.append(page)
|
|
109
|
+
|
|
110
|
+
new_pdf.save(output_pdf_path)
|
|
111
|
+
print(f"Created: {output_pdf_path}")
|
|
112
|
+
return True
|
|
113
|
+
def split_pdf(input_pdf_path, output_folder, exists_ok = 'skip'):
|
|
114
|
+
# Ensure output folder exists
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# Open the PDF file for reading
|
|
118
|
+
reader = PdfReader(input_pdf_path)
|
|
119
|
+
if reader.is_encrypted:
|
|
120
|
+
reader.decrypt("")
|
|
121
|
+
num_pages = len(reader.pages)
|
|
122
|
+
print(f"Total pages found: {num_pages}")
|
|
123
|
+
fname = pathlib.Path(input_pdf_path).name
|
|
124
|
+
os.makedirs(os.path.join(output_folder, fname), exist_ok=True)
|
|
125
|
+
existing_pages = [i for i in os.listdir(os.path.join(output_folder, fname)) if i.endswith('.pdf')]
|
|
126
|
+
if num_pages == len(existing_pages):
|
|
127
|
+
return # skip as all num pages exported
|
|
128
|
+
# Loop through each page and write it to a new PDF file
|
|
129
|
+
for i in range(num_pages):
|
|
130
|
+
writer = PdfWriter()
|
|
131
|
+
page = reader.pages[i]
|
|
132
|
+
writer.add_page(page)
|
|
133
|
+
|
|
134
|
+
output_pdf_path = os.path.join(output_folder, fname, f"page_{i + 1}.pdf")
|
|
135
|
+
os.makedirs(str(pathlib.Path(output_pdf_path).parent), exist_ok=True)
|
|
136
|
+
with open(output_pdf_path, "wb") as out_file:
|
|
137
|
+
writer.write(out_file)
|
|
138
|
+
|
|
139
|
+
print(f"Created: {output_pdf_path}")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def batch_pdf2png(document_folder, outfolder_path = None, exists_ok = 'skip', allowed_relative_paths = None,
|
|
143
|
+
loader = None):
|
|
144
|
+
|
|
145
|
+
"""_summary_
|
|
146
|
+
|
|
147
|
+
Args:
|
|
148
|
+
document_folder (_type_): folder containing splitted document folder, each folder same name as pdf file
|
|
149
|
+
outfolder_path (str, optional): folder containing the folder F that contain png files. F is the name of the unsplitted pdf Defaults output same folder as splitted pdf folder.
|
|
150
|
+
exists_ok (str, optional): _description_. Defaults to 'skip'.
|
|
151
|
+
"""
|
|
152
|
+
if outfolder_path is None:
|
|
153
|
+
outfolder_path = document_folder
|
|
154
|
+
from kg_doc_parser.utils.bounded_threadpool_executor import BoundedExecutor
|
|
155
|
+
bounded_executor = BoundedExecutor(max_workers= 2, max_pending= 5) # num of pdf
|
|
156
|
+
# for pdf_file in os.listdir(document_folder):
|
|
157
|
+
# pdf_file : str
|
|
158
|
+
if loader:
|
|
159
|
+
for rel_path in loader:
|
|
160
|
+
single_pdf2png(os.path.join(loader.compare_root, rel_path), os.path.join(outfolder_path, rel_path), exists_ok = exists_ok)
|
|
161
|
+
else:
|
|
162
|
+
for root, dirs, files in os.walk(document_folder):
|
|
163
|
+
if allowed_relative_paths is not None :
|
|
164
|
+
if dirs == []:
|
|
165
|
+
rel_path = str((pathlib.Path(root)).relative_to(document_folder))
|
|
166
|
+
if rel_path not in allowed_relative_paths:
|
|
167
|
+
continue
|
|
168
|
+
else:
|
|
169
|
+
continue
|
|
170
|
+
|
|
171
|
+
single_pdf2png(os.path.join(root), str(pathlib.Path(root)), exists_ok = exists_ok)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# OS-specific file locking
|
|
175
|
+
if platform.system() == "Windows":
|
|
176
|
+
import msvcrt
|
|
177
|
+
def lock_file(file_handle):
|
|
178
|
+
msvcrt.locking(file_handle.fileno(), msvcrt.LK_NBLCK, 1)
|
|
179
|
+
|
|
180
|
+
def unlock_file(file_handle):
|
|
181
|
+
try:
|
|
182
|
+
msvcrt.locking(file_handle.fileno(), msvcrt.LK_UNLCK, 1)
|
|
183
|
+
except Exception:
|
|
184
|
+
pass
|
|
185
|
+
|
|
186
|
+
else:
|
|
187
|
+
import fcntl
|
|
188
|
+
def lock_file(file_handle):
|
|
189
|
+
fcntl.flock(file_handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
190
|
+
|
|
191
|
+
def unlock_file(file_handle):
|
|
192
|
+
try:
|
|
193
|
+
fcntl.flock(file_handle, fcntl.LOCK_UN)
|
|
194
|
+
except Exception:
|
|
195
|
+
pass
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def get_thread_safe_tempfile(suffix=".pdf"):
|
|
199
|
+
pid = os.getpid()
|
|
200
|
+
thread_name = threading.current_thread().name.replace(" ", "_")
|
|
201
|
+
prefix = f"{thread_name}_{pid}_"
|
|
202
|
+
|
|
203
|
+
fd, path = tempfile.mkstemp(suffix=suffix, prefix=prefix)
|
|
204
|
+
return fd, path
|
|
205
|
+
def process_pdf_page(pdf_path, output_path):
|
|
206
|
+
if os.path.exists(output_path):
|
|
207
|
+
print(f"Skipped {output_path} (already exists)")
|
|
208
|
+
return
|
|
209
|
+
|
|
210
|
+
tmp_fd = None
|
|
211
|
+
tmp_path = None
|
|
212
|
+
file_handle = None
|
|
213
|
+
|
|
214
|
+
try:
|
|
215
|
+
# Create temporary file path
|
|
216
|
+
tmp_fd, tmp_path = get_thread_safe_tempfile(suffix=".pdf")
|
|
217
|
+
os.close(tmp_fd) # Close the low-level fd to avoid conflicts
|
|
218
|
+
|
|
219
|
+
# Copy the PDF to the temporary file
|
|
220
|
+
shutil.copy2(pdf_path, tmp_path)
|
|
221
|
+
|
|
222
|
+
# Open the temp file to lock
|
|
223
|
+
file_handle = open(tmp_path, 'rb')
|
|
224
|
+
lock_file(file_handle)
|
|
225
|
+
|
|
226
|
+
# Convert using the temp file
|
|
227
|
+
images = convert_from_path(tmp_path, dpi=300, fmt='png')
|
|
228
|
+
|
|
229
|
+
# Save each page as image
|
|
230
|
+
for i, image in enumerate(images):
|
|
231
|
+
image.save(output_path, "PNG")
|
|
232
|
+
print(f"Saved {output_path}")
|
|
233
|
+
|
|
234
|
+
except Exception as e:
|
|
235
|
+
print(f"Error processing {pdf_path}: {e}")
|
|
236
|
+
|
|
237
|
+
finally:
|
|
238
|
+
# Always release lock and delete temp file
|
|
239
|
+
if file_handle:
|
|
240
|
+
unlock_file(file_handle)
|
|
241
|
+
file_handle.close()
|
|
242
|
+
if tmp_path and os.path.exists(tmp_path):
|
|
243
|
+
os.remove(tmp_path)
|
|
244
|
+
|
|
245
|
+
def single_pdf2png(fname, folder_path, exists_ok = 'skip'):
|
|
246
|
+
"""_summary_
|
|
247
|
+
|
|
248
|
+
Args:
|
|
249
|
+
fname (_type_): the raw pdf file name before splitting
|
|
250
|
+
folder_path (_type_): the folder path containing the splitted pdf pages
|
|
251
|
+
exists_ok (str, optional): _description_. Defaults to 'skip'.
|
|
252
|
+
|
|
253
|
+
Raises:
|
|
254
|
+
PermissionError: _description_
|
|
255
|
+
"""
|
|
256
|
+
|
|
257
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def convert(folder_path, fname):
|
|
261
|
+
out_pdf_folder = fname # output folder
|
|
262
|
+
page_pdf_files = [f for f in os.listdir(out_pdf_folder) if f.endswith('.pdf')]
|
|
263
|
+
|
|
264
|
+
# Define the number of worker threads
|
|
265
|
+
max_workers = 3
|
|
266
|
+
|
|
267
|
+
# Use ThreadPoolExecutor to process PDF files concurrently
|
|
268
|
+
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
269
|
+
futures = []
|
|
270
|
+
for page_fname in page_pdf_files:
|
|
271
|
+
pdf_path = os.path.join(folder_path, page_fname)
|
|
272
|
+
output_path = os.path.join(out_pdf_folder, f"{os.path.splitext(page_fname)[0]}.png")
|
|
273
|
+
if os.path.exists(output_path):
|
|
274
|
+
if exists_ok == "skip":
|
|
275
|
+
continue
|
|
276
|
+
elif exists_ok == "raise":
|
|
277
|
+
raise PermissionError(f"File {output_path} already exists.")
|
|
278
|
+
else: # exist_ok == "ok"
|
|
279
|
+
pass
|
|
280
|
+
futures.append(executor.submit(process_pdf_page, pdf_path, output_path))
|
|
281
|
+
|
|
282
|
+
# Wait for all futures to complete
|
|
283
|
+
# it starts running only when start to be iterated.
|
|
284
|
+
for future in futures:
|
|
285
|
+
future.result()
|
|
286
|
+
convert(folder_path, fname)
|