graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
|
|
2
|
+
from typing import Dict, Literal, Any
|
|
3
|
+
from .semantic_document_splitting_layerwise_edits import parse_doc
|
|
4
|
+
|
|
5
|
+
def text_to_ocr_format(text: str, filename: str = "input_text") -> Dict:
|
|
6
|
+
"""
|
|
7
|
+
Wraps a raw string into the expected OCR dictionary format with a single dummy cluster.
|
|
8
|
+
"""
|
|
9
|
+
return {
|
|
10
|
+
filename: [
|
|
11
|
+
{
|
|
12
|
+
"pdf_page_num": 1,
|
|
13
|
+
"OCR_text_clusters": [
|
|
14
|
+
{
|
|
15
|
+
"text": text,
|
|
16
|
+
"bb_x_min": 0, "bb_y_min": 0, "bb_x_max": 1000, "bb_y_max": 1000,
|
|
17
|
+
"cluster_number": 0
|
|
18
|
+
}
|
|
19
|
+
],
|
|
20
|
+
"non_text_objects": []
|
|
21
|
+
}
|
|
22
|
+
]
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
def parse_doc_text(text: str, doc_id: str = "text_doc", parsing_mode: Literal["snippet", "delimiter"] = "snippet", max_depth: int = 10):
|
|
26
|
+
"""
|
|
27
|
+
Convenience function to parse a raw text string.
|
|
28
|
+
"""
|
|
29
|
+
raw_doc = text_to_ocr_format(text)
|
|
30
|
+
return parse_doc(doc_id, raw_doc, parsing_mode=parsing_mode, max_depth=max_depth)
|
|
File without changes
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
2
|
+
from threading import Semaphore, Lock, Condition
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class BoundedExecutor:
|
|
6
|
+
def __init__(self, max_workers=5, max_pending=10):
|
|
7
|
+
self._executor = ThreadPoolExecutor(max_workers=max_workers)
|
|
8
|
+
self._semaphore = Semaphore(max_pending)
|
|
9
|
+
self._lock = Lock()
|
|
10
|
+
self._condition = Condition(self._lock)
|
|
11
|
+
self._active_tasks = 0
|
|
12
|
+
|
|
13
|
+
def submit(self, fn, *args, **kwargs):
|
|
14
|
+
self._semaphore.acquire()
|
|
15
|
+
|
|
16
|
+
def wrapped_fn(*args, **kwargs):
|
|
17
|
+
with self._condition:
|
|
18
|
+
self._active_tasks += 1
|
|
19
|
+
try:
|
|
20
|
+
return fn(*args, **kwargs)
|
|
21
|
+
finally:
|
|
22
|
+
with self._condition:
|
|
23
|
+
self._active_tasks -= 1
|
|
24
|
+
self._condition.notify_all()
|
|
25
|
+
self._semaphore.release()
|
|
26
|
+
|
|
27
|
+
return self._executor.submit(wrapped_fn, *args, **kwargs)
|
|
28
|
+
|
|
29
|
+
def wait_for_all(self):
|
|
30
|
+
"""Block until all submitted tasks have completed."""
|
|
31
|
+
with self._condition:
|
|
32
|
+
while self._active_tasks > 0:
|
|
33
|
+
self._condition.wait()
|
|
34
|
+
|
|
35
|
+
def shutdown(self, wait=True):
|
|
36
|
+
"""Shut down the underlying ThreadPoolExecutor."""
|
|
37
|
+
self._executor.shutdown(wait=wait)
|
|
@@ -0,0 +1,405 @@
|
|
|
1
|
+
|
|
2
|
+
import logging
|
|
3
|
+
logger = logging.getLogger(__name__)
|
|
4
|
+
logger.addHandler(logging.NullHandler())
|
|
5
|
+
logger.debug("library loading")
|
|
6
|
+
from json import JSONDecodeError
|
|
7
|
+
import os
|
|
8
|
+
import pathlib
|
|
9
|
+
import pathspec
|
|
10
|
+
from typing import Iterable, Tuple, Optional, Generator, Any
|
|
11
|
+
from typing import Callable
|
|
12
|
+
from functools import partial
|
|
13
|
+
|
|
14
|
+
def bool2yn(maybe_bool: bool):
|
|
15
|
+
if type(maybe_bool) is bool:
|
|
16
|
+
return "Yes" if maybe_bool else "No"
|
|
17
|
+
else:
|
|
18
|
+
return maybe_bool
|
|
19
|
+
|
|
20
|
+
def recur_apply_json_inplace(json : dict | list | bool |str | int | float, fn: Callable):
|
|
21
|
+
if isinstance(json, dict):
|
|
22
|
+
for k, v in json.items():
|
|
23
|
+
json[k] = recur_apply_json_inplace(v, fn)
|
|
24
|
+
|
|
25
|
+
elif isinstance(json, list):
|
|
26
|
+
for i,v in enumerate(json):
|
|
27
|
+
json[i] = recur_apply_json_inplace(json[i], fn)
|
|
28
|
+
else:
|
|
29
|
+
return fn(json)
|
|
30
|
+
return json
|
|
31
|
+
|
|
32
|
+
# convert any boolean to yes no
|
|
33
|
+
json_bool_to_yn: Callable = partial(recur_apply_json_inplace, fn = bool2yn)
|
|
34
|
+
|
|
35
|
+
def nullable_concat(a: list| None, b: list| None) -> list | None:
|
|
36
|
+
if a is None and b is None:
|
|
37
|
+
return None
|
|
38
|
+
return (a or []) + (b or [])
|
|
39
|
+
|
|
40
|
+
def fast_walk(path, prune_condition=None):
|
|
41
|
+
stack = [path]
|
|
42
|
+
while stack:
|
|
43
|
+
current = stack.pop() # DFS, or use `.pop(0)` for BFS
|
|
44
|
+
try:
|
|
45
|
+
with os.scandir(current) as it:
|
|
46
|
+
dirs = []
|
|
47
|
+
files = []
|
|
48
|
+
for entry in it:
|
|
49
|
+
if entry.is_dir(follow_symlinks=False):
|
|
50
|
+
if prune_condition and prune_condition(entry.path):
|
|
51
|
+
continue
|
|
52
|
+
dirs.append(entry.path)
|
|
53
|
+
elif entry.is_file():
|
|
54
|
+
files.append(entry.path)
|
|
55
|
+
yield current, dirs, files
|
|
56
|
+
stack.extend(reversed(dirs)) # DFS
|
|
57
|
+
except PermissionError:
|
|
58
|
+
continue
|
|
59
|
+
def find_folders_two_levels_from_leaves_mem_optimized(root_path: str, required_level: int = 2) -> Generator[Tuple[str , list[str],list[str]], Any, None]:
|
|
60
|
+
"""
|
|
61
|
+
Yields folders that are exactly two levels away from leaf nodes using a
|
|
62
|
+
highly memory-efficient iterator.
|
|
63
|
+
|
|
64
|
+
The memoization cache is actively pruned to keep memory usage proportional
|
|
65
|
+
to the filesystem's width, not its total size.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
root_path: The absolute or relative path to the directory to start searching from.
|
|
69
|
+
|
|
70
|
+
Yields:
|
|
71
|
+
A path to a qualifying folder as soon as it is identified.
|
|
72
|
+
"""
|
|
73
|
+
if not os.path.isdir(root_path):
|
|
74
|
+
print(f"Error: Provided path '{root_path}' is not a valid directory.")
|
|
75
|
+
return
|
|
76
|
+
|
|
77
|
+
memo = {}
|
|
78
|
+
|
|
79
|
+
# Using realpath normalizes paths (e.g., C:/Users vs C:\Users) and handles
|
|
80
|
+
# symlinks, making the cache keys consistent.
|
|
81
|
+
# real_root = os.path.realpath(root_path)
|
|
82
|
+
|
|
83
|
+
for dirpath, dirnames, filenames in os.walk(root_path, topdown=False):
|
|
84
|
+
no_solution = False
|
|
85
|
+
# 1. Get the depths of all children of the current `dirpath`.
|
|
86
|
+
child_depths = [0] * len(filenames) # Files are leaves, depth 0.
|
|
87
|
+
|
|
88
|
+
for dirname in dirnames:
|
|
89
|
+
child_path = os.path.join(dirpath, dirname)
|
|
90
|
+
child_depth = memo.get(child_path, -1)
|
|
91
|
+
if child_depth is None:
|
|
92
|
+
no_solution = True
|
|
93
|
+
child_depths = [None]
|
|
94
|
+
break
|
|
95
|
+
child_depths.append(memo.get(child_path, -1))
|
|
96
|
+
|
|
97
|
+
# 2. THE CORE LOGIC: Check if this directory qualifies.
|
|
98
|
+
# if child_depths and all(depth == required_level for depth in child_depths):
|
|
99
|
+
# # Yield the original path if it's the root, otherwise the current dirpath
|
|
100
|
+
# if dirpath == real_root:
|
|
101
|
+
# return # yield root_path
|
|
102
|
+
# else:
|
|
103
|
+
# yield dirpath, dirnames, filenames
|
|
104
|
+
|
|
105
|
+
# 3. Calculate and cache the depth for the current path.
|
|
106
|
+
if no_solution:
|
|
107
|
+
current_depth = None
|
|
108
|
+
else:
|
|
109
|
+
if len(set(child_depths)) > 1:
|
|
110
|
+
current_depth = None
|
|
111
|
+
else:
|
|
112
|
+
|
|
113
|
+
if not child_depths:
|
|
114
|
+
current_depth = 0
|
|
115
|
+
elif -1 in child_depths:
|
|
116
|
+
current_depth = -1
|
|
117
|
+
else:
|
|
118
|
+
# child_depths must have no None here
|
|
119
|
+
current_depth = 1 + child_depths[0] # type: ignore
|
|
120
|
+
if current_depth == required_level:
|
|
121
|
+
yield dirpath, dirnames, filenames
|
|
122
|
+
memo[dirpath] = current_depth
|
|
123
|
+
|
|
124
|
+
# 4. PRUNING STEP: Release memory by removing child depths.
|
|
125
|
+
# This is safe because we are done with this level and the parent
|
|
126
|
+
# only needs the depth of `dirpath`, which we just cached.
|
|
127
|
+
for dirname in dirnames:
|
|
128
|
+
child_path = os.path.join(dirpath, dirname)
|
|
129
|
+
memo.pop(child_path, None) # Safely remove the key
|
|
130
|
+
class RawFileLoader():
|
|
131
|
+
def __init__(self, env_flist_path: Optional[str] = None,
|
|
132
|
+
allow_file_list: None | list[str] = None,
|
|
133
|
+
allow_file_list_ref: Optional[str] = None,
|
|
134
|
+
max_num_file = float('inf'), oldest_datetime = None, newest_datetime = None, root_folder_name : str | None= None,
|
|
135
|
+
# in_folder_name :Optional[str] = None,
|
|
136
|
+
walk_root:Optional[str] = None, compare_root:Optional[str] = None,
|
|
137
|
+
include = None,
|
|
138
|
+
bucket_blob_connection_str = None,
|
|
139
|
+
file_walker_callback: Callable[[str, int], Generator[Tuple[str , list[str],list[str]], Any, None]] | None = None,
|
|
140
|
+
pattern = None,
|
|
141
|
+
allow_startwith_relative_paths = False,
|
|
142
|
+
filtering_callbacks : Optional[list[Callable]] = None):
|
|
143
|
+
"""file loader to either backward support for local folder loading behaviour, or cloud bucket/ blob stoages
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
env_var_name (_type_): environment variable that contains the path to the file list
|
|
147
|
+
allow_file_list (None | list[str] | str): file path that contains the file information or a list of filenames directly passed into
|
|
148
|
+
pattern re.compile returned pattern to apply on the path
|
|
149
|
+
filtering_callbacks: that apply to each path, when True, it allows continue, when False, it continue to next file
|
|
150
|
+
"""
|
|
151
|
+
|
|
152
|
+
assert not ((bucket_blob_connection_str is not None) and (env_flist_path is not None))
|
|
153
|
+
self.allow_startwith_relative_paths = allow_startwith_relative_paths
|
|
154
|
+
self.pattern = pattern
|
|
155
|
+
self.include = include or []
|
|
156
|
+
self.oldest_datetime = oldest_datetime
|
|
157
|
+
self.newest_datetime = newest_datetime
|
|
158
|
+
self.bucket_blob_connection_str = bucket_blob_connection_str
|
|
159
|
+
self.file_walker_callback: Optional[Callable] = file_walker_callback
|
|
160
|
+
if root_folder_name is None:
|
|
161
|
+
root_folder_name = os.getcwd()
|
|
162
|
+
if walk_root is None:
|
|
163
|
+
walk_root = os.getcwd()
|
|
164
|
+
self.walk_root = walk_root
|
|
165
|
+
if compare_root is None:
|
|
166
|
+
compare_root = walk_root
|
|
167
|
+
self.compare_root = compare_root
|
|
168
|
+
self.max_num_file = max_num_file
|
|
169
|
+
allowed_relative_paths: Optional[list[str]] = None
|
|
170
|
+
|
|
171
|
+
if allow_file_list_ref is not None:
|
|
172
|
+
if allow_file_list is None:
|
|
173
|
+
allow_file_list = []
|
|
174
|
+
if isinstance(allow_file_list_ref, str):
|
|
175
|
+
need_attempt_readlines = False
|
|
176
|
+
try:
|
|
177
|
+
if allow_file_list_ref.endswith('.json'):
|
|
178
|
+
import json
|
|
179
|
+
with open(allow_file_list_ref, 'r') as f:
|
|
180
|
+
flist = json.load(f)
|
|
181
|
+
if isinstance(flist, list):
|
|
182
|
+
pass
|
|
183
|
+
else:
|
|
184
|
+
raise Exception('only list-like json is accepted')
|
|
185
|
+
allow_file_list = flist # type: ignore
|
|
186
|
+
else:
|
|
187
|
+
need_attempt_readlines = True
|
|
188
|
+
except JSONDecodeError as e:
|
|
189
|
+
need_attempt_readlines = True
|
|
190
|
+
|
|
191
|
+
except Exception as e:
|
|
192
|
+
raise
|
|
193
|
+
if need_attempt_readlines:
|
|
194
|
+
with open(allow_file_list_ref, 'r') as f:
|
|
195
|
+
flist = f.readlines()
|
|
196
|
+
|
|
197
|
+
allow_file_list = flist # type: ignore
|
|
198
|
+
else:
|
|
199
|
+
raise(ValueError("allow_file_list does not either provide a path to file containing file list "
|
|
200
|
+
"or directly passing file list"))
|
|
201
|
+
|
|
202
|
+
# allow combine flist from argument with env specifed paths list concatenated
|
|
203
|
+
if env_flist_path is not None:
|
|
204
|
+
if allowed_relative_paths is None:
|
|
205
|
+
allowed_relative_paths = []
|
|
206
|
+
if flist:=os.environ.get(env_flist_path, None):
|
|
207
|
+
if flist is not None:
|
|
208
|
+
if os.path.exists(flist):
|
|
209
|
+
if flist.endswith('.txt'):
|
|
210
|
+
with open(flist, 'r', encoding='utf-8') as f:
|
|
211
|
+
allowed_relative_paths = [i.strip() for i in f.readlines()]
|
|
212
|
+
|
|
213
|
+
if flist.endswith('.csv'):
|
|
214
|
+
with open(flist, 'r') as f:
|
|
215
|
+
for ln in f.readlines():
|
|
216
|
+
allowed_relative_paths.append(os.path.join(*(i.strip() for i in ln.split(','))))
|
|
217
|
+
elif flist.endswith('.xls') or flist.endswith('.xlsx'):
|
|
218
|
+
from pandas import read_excel
|
|
219
|
+
df = read_excel(flist)
|
|
220
|
+
allowed_relative_paths = []
|
|
221
|
+
for i, row in df.iterrows():
|
|
222
|
+
new_path = os.path.join(*(row[:3]))
|
|
223
|
+
allowed_relative_paths.append(new_path)
|
|
224
|
+
if allowed_relative_paths is not None:
|
|
225
|
+
if allow_file_list is None:
|
|
226
|
+
allow_file_list = []
|
|
227
|
+
|
|
228
|
+
allow_file_list = allow_file_list + allowed_relative_paths
|
|
229
|
+
self.allow_file_list = allow_file_list
|
|
230
|
+
self.filtering_callbacks = filtering_callbacks or []
|
|
231
|
+
self.resolve_paths = False
|
|
232
|
+
if self.resolve_paths:
|
|
233
|
+
self.check_allowed_relative_path()
|
|
234
|
+
pass
|
|
235
|
+
def check_allowed_relative_path(self, paths= None, ):
|
|
236
|
+
# ensure allowed
|
|
237
|
+
allow_file_list = nullable_concat(self.allow_file_list, paths)
|
|
238
|
+
if allow_file_list is None:
|
|
239
|
+
return
|
|
240
|
+
out_file_list = []
|
|
241
|
+
try:
|
|
242
|
+
for p in allow_file_list:
|
|
243
|
+
out_path = pathlib.Path(p).relative_to(self.compare_root)
|
|
244
|
+
out_file_list.append(str(out_path))
|
|
245
|
+
except ValueError as e:
|
|
246
|
+
if "is not in the subpath of" in str(e):
|
|
247
|
+
print("allow_file_list contains path outside of compare root")
|
|
248
|
+
raise
|
|
249
|
+
self.allow_file_list = out_file_list
|
|
250
|
+
|
|
251
|
+
def __iter__(self, leaf_only = False, file_non_exist_ok = False, include = None,
|
|
252
|
+
allowed_files: Optional[list[str]] = None,
|
|
253
|
+
# allowed_prefixes : Optional[list[str | int]] = None,
|
|
254
|
+
allowed_relative_paths: Optional[list[str]]= None,
|
|
255
|
+
):
|
|
256
|
+
"""Iterate through availble files
|
|
257
|
+
|
|
258
|
+
Args:
|
|
259
|
+
leaf_only (bool, optional): _description_. Defaults to True.
|
|
260
|
+
file_non_exist_ok (bool, optional): _description_. Defaults to False.
|
|
261
|
+
include (_type_, optional): _description_. Defaults to None.
|
|
262
|
+
allowed_files (Optional[list[str]], optional): _description_. Defaults to None.
|
|
263
|
+
allowed_relative_paths (Optional[list[str]], optional): _description_. Defaults to None.
|
|
264
|
+
allow_startwith_relative_paths also check if the file start with any of the allowed relative paths
|
|
265
|
+
Yields:
|
|
266
|
+
_type_: _description_
|
|
267
|
+
"""
|
|
268
|
+
from collections import Counter
|
|
269
|
+
filter_stat = Counter()
|
|
270
|
+
if include is None:
|
|
271
|
+
include = include or self.include or set(['files'])#, ['files', "dirs"]
|
|
272
|
+
else:
|
|
273
|
+
include = set(include)
|
|
274
|
+
include.update(self.include)
|
|
275
|
+
count = 0
|
|
276
|
+
from datetime import datetime
|
|
277
|
+
if self.oldest_datetime is not None:
|
|
278
|
+
if type(self.oldest_datetime) is str:
|
|
279
|
+
time_threshold_dt = datetime.strptime(self.oldest_datetime, '%Y-%m-%d %H:%M')
|
|
280
|
+
else:
|
|
281
|
+
time_threshold_dt = self.oldest_datetime
|
|
282
|
+
else:
|
|
283
|
+
time_threshold_dt = None
|
|
284
|
+
if self.newest_datetime is not None:
|
|
285
|
+
if type(self.newest_datetime) is str:
|
|
286
|
+
time_threshold_upper_dt = datetime.strptime(self.newest_datetime, '%Y-%m-%d %H:%M')
|
|
287
|
+
else:
|
|
288
|
+
time_threshold_upper_dt = self.newest_datetime
|
|
289
|
+
else:
|
|
290
|
+
time_threshold_upper_dt = None
|
|
291
|
+
nullable_allowed_relative_paths_set = []
|
|
292
|
+
nullable_allowed_relative_paths_set_t = nullable_concat(allowed_relative_paths, self.allow_file_list)
|
|
293
|
+
if nullable_allowed_relative_paths_set_t is not None:
|
|
294
|
+
nullable_allowed_relative_paths_set_t = set(nullable_allowed_relative_paths_set_t)
|
|
295
|
+
if len(nullable_allowed_relative_paths_set_t) > 0:
|
|
296
|
+
nullable_allowed_relative_paths_set = list(nullable_allowed_relative_paths_set_t)
|
|
297
|
+
else:
|
|
298
|
+
nullable_allowed_relative_paths_set = None
|
|
299
|
+
# debug_list = []
|
|
300
|
+
if self.file_walker_callback is None:
|
|
301
|
+
if self.bucket_blob_connection_str is None:
|
|
302
|
+
file_walker = os.walk(self.walk_root)
|
|
303
|
+
else:
|
|
304
|
+
raise NotImplementedError('blob or bucket stage file walker not implemented')
|
|
305
|
+
else:
|
|
306
|
+
file_walker = self.file_walker_callback(self.walk_root)
|
|
307
|
+
spec = None
|
|
308
|
+
if nullable_allowed_relative_paths_set is not None:
|
|
309
|
+
if self.allow_startwith_relative_paths:
|
|
310
|
+
|
|
311
|
+
prefixes = [pathlib.Path(i).as_posix() for i in nullable_allowed_relative_paths_set]
|
|
312
|
+
# normalized_prefixes = [p.rstrip("/\\") + "/" for p in raw_prefixes]
|
|
313
|
+
spec = pathspec.PathSpec.from_lines("gitwildmatch", prefixes)
|
|
314
|
+
for root, dirs, files in file_walker:
|
|
315
|
+
# print(root)
|
|
316
|
+
if leaf_only:
|
|
317
|
+
if dirs != []:
|
|
318
|
+
filter_stat.update(['dir != []'])
|
|
319
|
+
continue
|
|
320
|
+
if allowed_files:
|
|
321
|
+
if pathlib.Path(root).parts[-1] in allowed_files:
|
|
322
|
+
pass
|
|
323
|
+
else:
|
|
324
|
+
filter_stat.update(['pathlib.Path(root).parts[-1] in allowed_files'])
|
|
325
|
+
continue
|
|
326
|
+
to_iter = []
|
|
327
|
+
if "files" in include:
|
|
328
|
+
to_iter += files
|
|
329
|
+
if "dirs" in include:
|
|
330
|
+
to_iter += dirs
|
|
331
|
+
if "root" in include:
|
|
332
|
+
to_iter += ['.']
|
|
333
|
+
for f in to_iter:
|
|
334
|
+
if self.pattern:
|
|
335
|
+
if self.pattern.match(os.path.join(root, f)):
|
|
336
|
+
pass
|
|
337
|
+
else:
|
|
338
|
+
filter_stat.update(['self.pattern.match(os.path.join(root, f))'])
|
|
339
|
+
continue
|
|
340
|
+
f_rel = str(pathlib.Path(os.path.join(root, f)).relative_to(self.compare_root))
|
|
341
|
+
if spec is not None:
|
|
342
|
+
matched = True
|
|
343
|
+
matched = spec.match_file(pathlib.Path(f_rel).as_posix())
|
|
344
|
+
|
|
345
|
+
if not matched:
|
|
346
|
+
filter_stat.update(['not spec.match_file(pathlib.Path(f_rel).as_posix())'])
|
|
347
|
+
continue
|
|
348
|
+
else:
|
|
349
|
+
if nullable_allowed_relative_paths_set is None:
|
|
350
|
+
pass
|
|
351
|
+
elif f_rel in nullable_allowed_relative_paths_set:
|
|
352
|
+
pass
|
|
353
|
+
else:
|
|
354
|
+
filter_stat.update(['f_rel in allowed_relative_paths'])
|
|
355
|
+
continue
|
|
356
|
+
filtered = False
|
|
357
|
+
for cb in self.filtering_callbacks:
|
|
358
|
+
|
|
359
|
+
if cb(os.path.join(self.compare_root, f_rel)):
|
|
360
|
+
pass
|
|
361
|
+
else:
|
|
362
|
+
filter_stat.update([f'callback filtered by {cb}'])
|
|
363
|
+
filtered = True
|
|
364
|
+
break
|
|
365
|
+
if filtered:
|
|
366
|
+
continue
|
|
367
|
+
creation_timestamp = os.path.getctime(os.path.join(self.compare_root, f_rel))
|
|
368
|
+
creation_time = datetime.fromtimestamp(creation_timestamp)
|
|
369
|
+
if time_threshold_dt is not None:
|
|
370
|
+
if not (creation_time > time_threshold_dt):
|
|
371
|
+
filter_stat.update(['not (creation_time > time_threshold_dt)'])
|
|
372
|
+
continue
|
|
373
|
+
if time_threshold_upper_dt is not None:
|
|
374
|
+
if not (creation_time < time_threshold_upper_dt):
|
|
375
|
+
filter_stat.update(['not (creation_time < time_threshold_upper_dt)'])
|
|
376
|
+
continue
|
|
377
|
+
count+= 1
|
|
378
|
+
# debug_list.append(f_rel)
|
|
379
|
+
yield f_rel
|
|
380
|
+
if count >= self.max_num_file:
|
|
381
|
+
print(filter_stat.most_common())
|
|
382
|
+
return
|
|
383
|
+
pass
|
|
384
|
+
print(filter_stat.most_common())
|
|
385
|
+
pass
|
|
386
|
+
|
|
387
|
+
def filter_folder(folder_root = os.path.join(
|
|
388
|
+
"..", "odc_data", "split_pages"
|
|
389
|
+
|
|
390
|
+
), min_page = 45, max_page: int | float = 55, first = 10, verbose = True):
|
|
391
|
+
folders = []
|
|
392
|
+
for root, dirs, files in os.walk(folder_root):
|
|
393
|
+
if dirs == []:
|
|
394
|
+
n_page = len([i for i in files if i.lower().endswith('.pdf')])
|
|
395
|
+
|
|
396
|
+
if n_page < max_page and min_page < n_page:
|
|
397
|
+
rel_path = pathlib.Path(root).relative_to(folder_root)
|
|
398
|
+
folders.append(str(rel_path))
|
|
399
|
+
|
|
400
|
+
if verbose:
|
|
401
|
+
print(folders)
|
|
402
|
+
if first is not None:
|
|
403
|
+
return (sorted(folders)[:first])
|
|
404
|
+
else:
|
|
405
|
+
return (sorted(folders))
|