text2rel 0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- text2rel/__init__.py +3 -0
- text2rel/cleaner.py +909 -0
- text2rel/io.py +12 -0
- text2rel/llm_processor.py +442 -0
- text2rel/reliability_assessor.py +364 -0
- text2rel/utils.py +39 -0
- text2rel-0.1.dist-info/METADATA +42 -0
- text2rel-0.1.dist-info/RECORD +9 -0
- text2rel-0.1.dist-info/WHEEL +4 -0
text2rel/cleaner.py
ADDED
|
@@ -0,0 +1,909 @@
|
|
|
1
|
+
|
|
2
|
+
import os
|
|
3
|
+
import sys
|
|
4
|
+
import json
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
import re
|
|
8
|
+
from bs4 import BeautifulSoup
|
|
9
|
+
import ast
|
|
10
|
+
import string
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from text2rel.utils import insert_jumpjump
|
|
13
|
+
from text2rel.io import _load_inventory, _save_inventory
|
|
14
|
+
from typing import Dict
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class HTMLCleaner:
|
|
19
|
+
def __init__(self, json_path: str, html_directory: str = None, extension: str = ".htm"):
|
|
20
|
+
"""
|
|
21
|
+
Initializes the cleaner. If the JSON doesn't exist but a directory is given, it builds the inventory.
|
|
22
|
+
"""
|
|
23
|
+
if not json_path:
|
|
24
|
+
raise ValueError("json_path parameter is required. Please provide a path to your JSON inventory file. It can be simply and empty JSON file")
|
|
25
|
+
if not html_directory:
|
|
26
|
+
raise ValueError("html_directory parameter is required. Please provide a path to your HTML files directory.")
|
|
27
|
+
self.json_path = Path(json_path)
|
|
28
|
+
|
|
29
|
+
if not self.json_path.exists():
|
|
30
|
+
if html_directory is None:
|
|
31
|
+
raise FileNotFoundError(f"JSON not found at {json_path} and no directory given to create one.")
|
|
32
|
+
self._create_inventory(html_directory, extension)
|
|
33
|
+
print(f"Created new inventory at {json_path}")
|
|
34
|
+
|
|
35
|
+
self.inventory = _load_inventory(self.json_path)
|
|
36
|
+
|
|
37
|
+
#Creates the inventory from a given directory of HTML files.
|
|
38
|
+
def _create_inventory(self, directory: str, extension: str = ".htm"):
|
|
39
|
+
directory = Path(directory)
|
|
40
|
+
if not directory.exists():
|
|
41
|
+
raise FileNotFoundError(f"Directory {directory} does not exist.")
|
|
42
|
+
|
|
43
|
+
data = []
|
|
44
|
+
for root, _, files in os.walk(directory):
|
|
45
|
+
for file in files:
|
|
46
|
+
if file.endswith(extension):
|
|
47
|
+
data.append({
|
|
48
|
+
"file_path": str(Path(root) / file),
|
|
49
|
+
"relevant_beginning": None, #Needs to be manually set by the user.
|
|
50
|
+
"relevant_end": None, #Needs to be manually set by the user.
|
|
51
|
+
"font_usage": [] #Gets updated during cleaning.
|
|
52
|
+
})
|
|
53
|
+
|
|
54
|
+
inventory = {"files": data}
|
|
55
|
+
with open(self.json_path, 'w', encoding='utf-8') as f:
|
|
56
|
+
json.dump(inventory, f, indent=4)
|
|
57
|
+
|
|
58
|
+
#Lists all files in the inventory with their index for easy reference.
|
|
59
|
+
def list_files(self):
|
|
60
|
+
"""
|
|
61
|
+
Prints all file paths with their corresponding index for easy reference.
|
|
62
|
+
"""
|
|
63
|
+
print("Inventory File Index:")
|
|
64
|
+
for idx, file_info in enumerate(self.inventory['files']):
|
|
65
|
+
print(f"{idx}: {Path(file_info['file_path']).name}")
|
|
66
|
+
|
|
67
|
+
#Allowing the user to set the relevant beginning and end for each file, the files can be accessed by filename
|
|
68
|
+
def set_relevant_bounds(self, filename: str, beginning, end):
|
|
69
|
+
for file in self.inventory['files']:
|
|
70
|
+
if file['file_path'].endswith(filename):
|
|
71
|
+
file['relevant_beginning'] = beginning
|
|
72
|
+
file['relevant_end'] = end
|
|
73
|
+
_save_inventory(self.json_path,self.inventory)
|
|
74
|
+
return
|
|
75
|
+
raise ValueError(f"File {filename} not found in inventory.")
|
|
76
|
+
|
|
77
|
+
def analyze_html_fonts(self, file_index: int):
|
|
78
|
+
"""
|
|
79
|
+
Analyzes the HTML content of a file's relevant section to extract font usage statistics.
|
|
80
|
+
Parameters:
|
|
81
|
+
- file_index (int): Index of the file in the inventory
|
|
82
|
+
Updates:
|
|
83
|
+
- Updates the `font_usage` field in the JSON for the selected file.
|
|
84
|
+
"""
|
|
85
|
+
file_info = self.inventory['files'][file_index]
|
|
86
|
+
html_path = Path(file_info['file_path'])
|
|
87
|
+
|
|
88
|
+
if not html_path.exists():
|
|
89
|
+
print(f"File not found: {html_path}")
|
|
90
|
+
return
|
|
91
|
+
|
|
92
|
+
if file_info['relevant_beginning'] is None or file_info['relevant_end'] is None:
|
|
93
|
+
print(f"relevant_beginning and relevant_end must be set first.")
|
|
94
|
+
return
|
|
95
|
+
|
|
96
|
+
with open(html_path, 'r', encoding='utf-8') as f:
|
|
97
|
+
html_content = f.read()
|
|
98
|
+
|
|
99
|
+
relevant_text = html_content[file_info['relevant_beginning']:file_info['relevant_end']-1]
|
|
100
|
+
soup = BeautifulSoup(relevant_text, 'html.parser')
|
|
101
|
+
|
|
102
|
+
# Match font classes like font1, font12, etc.
|
|
103
|
+
pattern = re.compile(r'font\d{1,2}')
|
|
104
|
+
class_stats = {}
|
|
105
|
+
|
|
106
|
+
for tag in soup.find_all(class_=pattern):
|
|
107
|
+
class_list = tag.get('class', [])
|
|
108
|
+
for cls_name in class_list:
|
|
109
|
+
if not pattern.match(cls_name):
|
|
110
|
+
continue
|
|
111
|
+
text = tag.get_text(strip=True)
|
|
112
|
+
if cls_name in class_stats:
|
|
113
|
+
class_stats[cls_name]['total_length'] += len(text)
|
|
114
|
+
class_stats[cls_name]['count'] += 1
|
|
115
|
+
else:
|
|
116
|
+
class_stats[cls_name] = {'total_length': len(text), 'count': 1}
|
|
117
|
+
|
|
118
|
+
# Format and save to inventory
|
|
119
|
+
font_usage = []
|
|
120
|
+
for cls, stats in class_stats.items():
|
|
121
|
+
font_usage.append({
|
|
122
|
+
'class': cls,
|
|
123
|
+
'instances': stats['count'],
|
|
124
|
+
'average_length': stats['total_length'] / stats['count'],
|
|
125
|
+
'usage': "" # Still to be set by user
|
|
126
|
+
})
|
|
127
|
+
|
|
128
|
+
file_info['font_usage'] = font_usage
|
|
129
|
+
_save_inventory(self.json_path, self.inventory)
|
|
130
|
+
#Display the fonts to the user
|
|
131
|
+
df = pd.DataFrame(font_usage)
|
|
132
|
+
print(df)
|
|
133
|
+
print(f"Font usage statistics for {Path(file_info['file_path']).name} updated.")
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
#Display the results as a DataFrame
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def classify_fonts_usage(self, index: int, usages: dict):
|
|
140
|
+
"""
|
|
141
|
+
Classifies font usage for a specific file in the inventory.
|
|
142
|
+
|
|
143
|
+
Parameters:
|
|
144
|
+
- index (int): The index of the file in the inventory.
|
|
145
|
+
- usages (dict): Dictionary of font classes to usage labels (e.g., {'font3': 'MainText'}).
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
if 0 <= index < len(self.inventory['files']):
|
|
149
|
+
file_info = self.inventory['files'][index]
|
|
150
|
+
for font, usage_type in usages.items():
|
|
151
|
+
for font_entry in file_info.get('font_usage', []):
|
|
152
|
+
if font_entry['class'] == font:
|
|
153
|
+
font_entry['usage'] = usage_type
|
|
154
|
+
break
|
|
155
|
+
|
|
156
|
+
_save_inventory(self.json_path,self.inventory)
|
|
157
|
+
print("Updated font usage for file index", index)
|
|
158
|
+
else:
|
|
159
|
+
print(f"Invalid index: {index}")
|
|
160
|
+
|
|
161
|
+
#Remove all the
|
|
162
|
+
|
|
163
|
+
#Converts the inventory into a pandas DataFrame for easy analysis if the user desires it.
|
|
164
|
+
|
|
165
|
+
def to_dataframe(self):
|
|
166
|
+
"""
|
|
167
|
+
Converts the inventory into a pandas DataFrame.
|
|
168
|
+
|
|
169
|
+
Returns:
|
|
170
|
+
- pd.DataFrame: Columns include File, RelevantBeginning, RelevantEnd, class, instances, average_length, Usage.
|
|
171
|
+
"""
|
|
172
|
+
records = []
|
|
173
|
+
|
|
174
|
+
for file_info in self.inventory['files']:
|
|
175
|
+
for font_entry in file_info.get('font_usage', []):
|
|
176
|
+
records.append({
|
|
177
|
+
'File': file_info['file_path'],
|
|
178
|
+
'RelevantBeginning': file_info['relevant_beginning'],
|
|
179
|
+
'RelevantEnd': file_info['relevant_end'],
|
|
180
|
+
'class': font_entry['class'],
|
|
181
|
+
'instances': font_entry['instances'],
|
|
182
|
+
'average_length': font_entry['average_length'],
|
|
183
|
+
'Usage': font_entry['usage']
|
|
184
|
+
})
|
|
185
|
+
|
|
186
|
+
def selecting_text_chunks(self):
|
|
187
|
+
json_path = self.json_path
|
|
188
|
+
|
|
189
|
+
for file_info in self.inventory['files']:
|
|
190
|
+
html_file_path = file_info['file_path']
|
|
191
|
+
|
|
192
|
+
# Skip files without bounds or font usage set
|
|
193
|
+
if file_info.get('relevant_beginning') is None or file_info.get('relevant_end') is None:
|
|
194
|
+
continue
|
|
195
|
+
if not any(entry.get('usage') == 'MainText' for entry in file_info.get('font_usage', [])):
|
|
196
|
+
continue
|
|
197
|
+
|
|
198
|
+
try:
|
|
199
|
+
with open(html_file_path, 'r', encoding='utf-8') as f:
|
|
200
|
+
html_content = f.read()
|
|
201
|
+
except FileNotFoundError:
|
|
202
|
+
print(f"File not found: {html_file_path}")
|
|
203
|
+
continue
|
|
204
|
+
|
|
205
|
+
print(f"Processing: {Path(html_file_path).name}")
|
|
206
|
+
|
|
207
|
+
# Slice relevant section, based on the markings in relevant beginning and relevant end
|
|
208
|
+
relevant_text = html_content[file_info['relevant_beginning']:file_info['relevant_end']]
|
|
209
|
+
soup = BeautifulSoup(relevant_text, 'html.parser')
|
|
210
|
+
|
|
211
|
+
# Step 1: Remove (decompose) elements with 'Rest' usage - they are not needed
|
|
212
|
+
for font_entry in file_info.get('font_usage', []):
|
|
213
|
+
if font_entry.get('usage') == 'Rest':
|
|
214
|
+
class_name = font_entry['class']
|
|
215
|
+
for el in soup.find_all(class_=class_name):
|
|
216
|
+
el.decompose()
|
|
217
|
+
|
|
218
|
+
# Step 2: Insert JumPJumP before Title classes
|
|
219
|
+
title_count = 0
|
|
220
|
+
for font_entry in file_info.get('font_usage', []):
|
|
221
|
+
if font_entry.get('usage') == 'Title':
|
|
222
|
+
class_name = font_entry['class']
|
|
223
|
+
for el in soup.find_all(class_=class_name):
|
|
224
|
+
el.insert_before(soup.new_string("JumPJumP"))
|
|
225
|
+
title_count += 1
|
|
226
|
+
|
|
227
|
+
# Step 3: Extract ALL remaining text (includes JumPJumP markers and titles)
|
|
228
|
+
text = soup.get_text()
|
|
229
|
+
|
|
230
|
+
print(f" → {title_count} title elements found (JumPJumP inserted before each)")
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# Clean up the text
|
|
234
|
+
text = re.sub(r'\n+', '\n', text)
|
|
235
|
+
text = re.sub(r'[ ]+', ' ', text)
|
|
236
|
+
|
|
237
|
+
excluded_punctuation = re.escape(string.punctuation.replace('.', ''))
|
|
238
|
+
text = re.sub(
|
|
239
|
+
fr"([a-záéíóúü]) ?([{excluded_punctuation}])?\n([{excluded_punctuation}])? ?([a-záéíóúü])",
|
|
240
|
+
r'\1 \4',
|
|
241
|
+
text,
|
|
242
|
+
flags=re.IGNORECASE
|
|
243
|
+
)
|
|
244
|
+
|
|
245
|
+
# Count JumPJumP from titles only
|
|
246
|
+
jumps_from_titles = text.count("JumPJumP")
|
|
247
|
+
|
|
248
|
+
# Note: Length-based chopping is handled in apply_file_specific_rules()
|
|
249
|
+
# based on the LengthChop flag in the protocol CSV
|
|
250
|
+
|
|
251
|
+
# Merge short sections unnecessarily split by jumps
|
|
252
|
+
while True:
|
|
253
|
+
new_text = re.sub(r'JumPJumP(.{1,500})JumPJumP', r'JumPJumP\1', text, flags=re.DOTALL)
|
|
254
|
+
if new_text == text:
|
|
255
|
+
break
|
|
256
|
+
|
|
257
|
+
text = new_text
|
|
258
|
+
|
|
259
|
+
# Consolidate multiple consecutive JumPJumP markers into one
|
|
260
|
+
text = re.sub(r'(\n*JumPJumP\n*)+', '\nJumPJumP\n', text)
|
|
261
|
+
|
|
262
|
+
# Split by marker and embed directly in JSON
|
|
263
|
+
text_sections = [t.strip() for t in re.split(r'\n*JumPJumP\n*', text) if t.strip()]
|
|
264
|
+
file_info['cleaned_subsections'] = text_sections
|
|
265
|
+
|
|
266
|
+
print(f" → JumPJumP from titles: {jumps_from_titles}")
|
|
267
|
+
print(f" → Initial subsections: {len(text_sections)}")
|
|
268
|
+
|
|
269
|
+
# Save updated JSON
|
|
270
|
+
_save_inventory(json_path, self.inventory)
|
|
271
|
+
|
|
272
|
+
print("Cleaned text added to each file entry in the JSON.")
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
###_____________REMOVE WHEN EVERYTHING REGARDING FILE-SPECIFIC RULES IS FULLY TESTED AND WORKING________________####
|
|
276
|
+
# def apply_cleaning_rules(self, file_index, rule_keys: list = None, multiline: bool = False):
|
|
277
|
+
# """
|
|
278
|
+
# Applies selected regex cleaning rules from cleaning_rules.py to each subsection.
|
|
279
|
+
|
|
280
|
+
# Parameters:
|
|
281
|
+
# - file_index (int): The index of the file in the inventory
|
|
282
|
+
# - rule_keys (list[str]): List of rule category keys to apply.
|
|
283
|
+
# If None or empty, no rules will be applied.
|
|
284
|
+
# - multiline (bool): Whether to apply rules across multiple lines using re.DOTALL
|
|
285
|
+
|
|
286
|
+
# Returns:
|
|
287
|
+
# - list[str]: List of cleaned text sections
|
|
288
|
+
# """
|
|
289
|
+
# # Get the file info from inventory
|
|
290
|
+
# file_info = self.inventory['files'][file_index]
|
|
291
|
+
|
|
292
|
+
# # Skip if no rules specified
|
|
293
|
+
# if not rule_keys:
|
|
294
|
+
# return file_info["cleaned_subsections"]
|
|
295
|
+
|
|
296
|
+
# cleaned_sections = []
|
|
297
|
+
|
|
298
|
+
# # Process each subsection
|
|
299
|
+
# for section in file_info["cleaned_subsections"]:
|
|
300
|
+
# cleaned_text = section
|
|
301
|
+
|
|
302
|
+
# # Apply each rule category
|
|
303
|
+
# for key in rule_keys:
|
|
304
|
+
# rules = CLEANING_RULES.get(key, [])
|
|
305
|
+
# print(f"Applying rules for {key}")
|
|
306
|
+
|
|
307
|
+
# # Apply each pattern/replacement pair in the rule category
|
|
308
|
+
# for pattern, repl in rules:
|
|
309
|
+
# if multiline:
|
|
310
|
+
# print(f"Applying multiline rule: {pattern} -> {repl}")
|
|
311
|
+
# # Use re.DOTALL to match across multiple lines
|
|
312
|
+
# cleaned_text = re.sub(pattern, repl, cleaned_text, flags=re.DOTALL)
|
|
313
|
+
# else:
|
|
314
|
+
# # Use default behavior (single line)
|
|
315
|
+
# cleaned_text = re.sub(pattern, repl, cleaned_text)
|
|
316
|
+
|
|
317
|
+
# cleaned_sections.append(cleaned_text)
|
|
318
|
+
|
|
319
|
+
# # Update the cleaned subsections in the inventory
|
|
320
|
+
# file_info["cleaned_subsections"] = cleaned_sections
|
|
321
|
+
|
|
322
|
+
# # Save changes to JSON
|
|
323
|
+
# with open(self.json_path, 'w', encoding='utf-8') as f:
|
|
324
|
+
# json.dump(self.inventory, f, indent=4, ensure_ascii=False)
|
|
325
|
+
|
|
326
|
+
# return cleaned_sections
|
|
327
|
+
|
|
328
|
+
def apply_file_specific_rules(self, file_index: int):
|
|
329
|
+
"""
|
|
330
|
+
Applies file-specific regex rules from the attached protocol CSV.
|
|
331
|
+
"""
|
|
332
|
+
if not hasattr(self, "protocol_df"):
|
|
333
|
+
print("No protocol attached. Use attach_protocol() first.")
|
|
334
|
+
return
|
|
335
|
+
|
|
336
|
+
file_info = self.inventory["files"][file_index]
|
|
337
|
+
filename = Path(file_info["file_path"]).name
|
|
338
|
+
|
|
339
|
+
# Find matching protocol entry by filename
|
|
340
|
+
matches = self.protocol_df[
|
|
341
|
+
self.protocol_df['HTML_Location'].apply(lambda p: Path(p).name == filename)
|
|
342
|
+
]
|
|
343
|
+
|
|
344
|
+
if len(matches) == 0:
|
|
345
|
+
print(f"No protocol entry found for file: {filename}")
|
|
346
|
+
return
|
|
347
|
+
elif len(matches) > 1:
|
|
348
|
+
print(f"Multiple entries found for {filename}. Using the first match.")
|
|
349
|
+
|
|
350
|
+
match = matches.iloc[0]
|
|
351
|
+
|
|
352
|
+
# Parse the rules dictionary from the CSV
|
|
353
|
+
try:
|
|
354
|
+
rules = ast.literal_eval(match['patterns_replacements'])
|
|
355
|
+
if not isinstance(rules, dict):
|
|
356
|
+
raise ValueError("Parsed rules are not a dictionary")
|
|
357
|
+
except Exception as e:
|
|
358
|
+
print(f"Error parsing rules for {filename}: {e}")
|
|
359
|
+
return
|
|
360
|
+
|
|
361
|
+
# Skip if no rules or empty rules
|
|
362
|
+
if not rules or rules == {'': ''}:
|
|
363
|
+
return
|
|
364
|
+
|
|
365
|
+
# Get flags from protocol
|
|
366
|
+
length_chop = bool(int(match.get('LengthChop', 0)))
|
|
367
|
+
use_dotall = bool(int(match.get('LongReplacement', 0)))
|
|
368
|
+
|
|
369
|
+
# Get current subsections
|
|
370
|
+
text_sections = file_info.get("cleaned_subsections", [])
|
|
371
|
+
if not text_sections:
|
|
372
|
+
print(f"⚠ No cleaned_subsections found for: {filename} (skipping)")
|
|
373
|
+
return
|
|
374
|
+
|
|
375
|
+
print(f"Processing: {filename}")
|
|
376
|
+
print(f" → {len(text_sections)} subsection(s) to process")
|
|
377
|
+
print(f" → Rules to apply:")
|
|
378
|
+
for p, r in rules.items():
|
|
379
|
+
print(f" '{p[:50]}...' → '{r[:30]}...'" if len(p) > 50 else f" '{p}' → '{r}'")
|
|
380
|
+
if length_chop:
|
|
381
|
+
print(f" → LengthChop enabled (will insert JumPJumP every 10k chars)")
|
|
382
|
+
|
|
383
|
+
cleaned_sections = []
|
|
384
|
+
total_modified = 0
|
|
385
|
+
total_regex_matches = 0
|
|
386
|
+
|
|
387
|
+
for text in text_sections:
|
|
388
|
+
original_text = text
|
|
389
|
+
section_matches = 0
|
|
390
|
+
|
|
391
|
+
# Apply replacements and count matches
|
|
392
|
+
for pattern, repl in rules.items():
|
|
393
|
+
if pattern: # Skip empty patterns
|
|
394
|
+
matches_before = len(re.findall(pattern, text, flags=re.DOTALL if use_dotall else 0))
|
|
395
|
+
section_matches += matches_before
|
|
396
|
+
text = re.sub(pattern, repl, text, flags=re.DOTALL if use_dotall else 0)
|
|
397
|
+
|
|
398
|
+
total_regex_matches += section_matches
|
|
399
|
+
|
|
400
|
+
# Track if any changes were made
|
|
401
|
+
if text != original_text:
|
|
402
|
+
total_modified += 1
|
|
403
|
+
|
|
404
|
+
# Insert jump markers for long sections (if LengthChop=1)
|
|
405
|
+
if length_chop:
|
|
406
|
+
text = insert_jumpjump(text, max_chars=10000)
|
|
407
|
+
|
|
408
|
+
# Consolidate multiple consecutive JumPJumP markers into one
|
|
409
|
+
text = re.sub(r'(\n*JumPJumP\n*)+', '\nJumPJumP\n', text)
|
|
410
|
+
|
|
411
|
+
# Merge short sections (less than 500 chars) that were unnecessarily split
|
|
412
|
+
while True:
|
|
413
|
+
new_text = re.sub(r'JumPJumP(.{1,500})JumPJumP', r'JumPJumP\1', text, flags=re.DOTALL)
|
|
414
|
+
if new_text == text:
|
|
415
|
+
break
|
|
416
|
+
text = new_text
|
|
417
|
+
|
|
418
|
+
# Re-split by JumPJumP in case regex patterns added new markers
|
|
419
|
+
sub_sections = [t.strip() for t in re.split(r'\n*JumPJumP\n*', text) if t.strip()]
|
|
420
|
+
cleaned_sections.extend(sub_sections)
|
|
421
|
+
|
|
422
|
+
# Update the inventory with cleaned sections
|
|
423
|
+
file_info["cleaned_subsections"] = cleaned_sections
|
|
424
|
+
_save_inventory(self.json_path, self.inventory)
|
|
425
|
+
|
|
426
|
+
print(f" {total_regex_matches} regex matches applied to {total_modified}/{len(text_sections)} subsection(s)")
|
|
427
|
+
print(f" Final subsections after re-split: {len(cleaned_sections)}")
|
|
428
|
+
|
|
429
|
+
def attach_protocol(self, csv_path: str):
|
|
430
|
+
"""
|
|
431
|
+
Load and cache the protocol CSV that contains file-specific cleaning rules.
|
|
432
|
+
"""
|
|
433
|
+
self.protocol_df = pd.read_csv(csv_path)
|
|
434
|
+
print(f"Protocol CSV loaded with {len(self.protocol_df)} entries.")
|
|
435
|
+
|
|
436
|
+
def export_cleaned_text(self, output_path: str = None, include_empty: bool = False):
|
|
437
|
+
"""
|
|
438
|
+
Exports cleaned subsections to a separate JSON file for downstream use.
|
|
439
|
+
|
|
440
|
+
Parameters:
|
|
441
|
+
- output_path (str): Path to save the cleaned JSON.
|
|
442
|
+
If None, uses '<original_name>_cleaned.json'
|
|
443
|
+
- include_empty (bool): Whether to include files with no cleaned_subsections
|
|
444
|
+
|
|
445
|
+
Returns:
|
|
446
|
+
- str: Path to the exported file
|
|
447
|
+
"""
|
|
448
|
+
if output_path is None:
|
|
449
|
+
# Generate default name from inventory path
|
|
450
|
+
stem = self.json_path.stem
|
|
451
|
+
output_path = self.json_path.parent / f"{stem}_cleaned.json"
|
|
452
|
+
else:
|
|
453
|
+
output_path = Path(output_path)
|
|
454
|
+
|
|
455
|
+
cleaned_data = {
|
|
456
|
+
"source_inventory": str(self.json_path),
|
|
457
|
+
"exported_at": pd.Timestamp.now().isoformat(),
|
|
458
|
+
"files": []
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
files_with_content = 0
|
|
462
|
+
total_subsections = 0
|
|
463
|
+
|
|
464
|
+
for file_info in self.inventory["files"]:
|
|
465
|
+
subsections = file_info.get("cleaned_subsections", [])
|
|
466
|
+
|
|
467
|
+
# Skip empty files unless include_empty is True
|
|
468
|
+
if not subsections and not include_empty:
|
|
469
|
+
continue
|
|
470
|
+
|
|
471
|
+
cleaned_entry = {
|
|
472
|
+
"file_path": file_info["file_path"],
|
|
473
|
+
"filename": Path(file_info["file_path"]).name,
|
|
474
|
+
"subsection_count": len(subsections),
|
|
475
|
+
"cleaned_subsections": subsections
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
cleaned_data["files"].append(cleaned_entry)
|
|
479
|
+
|
|
480
|
+
if subsections:
|
|
481
|
+
files_with_content += 1
|
|
482
|
+
total_subsections += len(subsections)
|
|
483
|
+
|
|
484
|
+
# Save to file
|
|
485
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
486
|
+
json.dump(cleaned_data, f, indent=4, ensure_ascii=False)
|
|
487
|
+
|
|
488
|
+
print(f"Exported cleaned text to: {output_path}")
|
|
489
|
+
print(f" → {files_with_content} file(s) with content")
|
|
490
|
+
print(f" → {total_subsections} total subsection(s)")
|
|
491
|
+
|
|
492
|
+
return str(output_path)
|
|
493
|
+
|
|
494
|
+
def load_fonts_from_csv(self, csv_path: str):
|
|
495
|
+
"""
|
|
496
|
+
Load font classifications from a comprehensive CSV file (like 231228_ProtocolForPreliminaryCleaning.csv)
|
|
497
|
+
and automatically assign them to all files in the inventory.
|
|
498
|
+
|
|
499
|
+
This replaces the manual font assignment workflow with automated loading from the original protocol.
|
|
500
|
+
|
|
501
|
+
Parameters:
|
|
502
|
+
- csv_path (str): Path to the CSV file containing font classifications
|
|
503
|
+
|
|
504
|
+
Returns:
|
|
505
|
+
- dict: Statistics about the font assignment process
|
|
506
|
+
"""
|
|
507
|
+
print("Loading font classification data from CSV...")
|
|
508
|
+
df_fonts = pd.read_csv(csv_path)
|
|
509
|
+
print(f"Loaded {len(df_fonts)} font classification records")
|
|
510
|
+
|
|
511
|
+
# Statistics tracking
|
|
512
|
+
stats = {
|
|
513
|
+
'files_processed': 0,
|
|
514
|
+
'fonts_assigned': 0,
|
|
515
|
+
'files_matched': 0,
|
|
516
|
+
'files_not_found': [],
|
|
517
|
+
'font_usage_summary': {}
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
# Group by file path for easier processing
|
|
521
|
+
print("\nGrouping font data by file...")
|
|
522
|
+
font_groups = df_fonts.groupby('File')
|
|
523
|
+
|
|
524
|
+
# Process each file in the inventory
|
|
525
|
+
print("\nAssigning fonts to inventory files...")
|
|
526
|
+
|
|
527
|
+
for file_idx, file_info in enumerate(self.inventory['files']):
|
|
528
|
+
file_path = file_info['file_path']
|
|
529
|
+
filename = Path(file_path).name
|
|
530
|
+
|
|
531
|
+
print(f"Processing file {file_idx}: {filename}")
|
|
532
|
+
|
|
533
|
+
# Find matching entries in the CSV (try both full path and filename)
|
|
534
|
+
matching_fonts = None
|
|
535
|
+
|
|
536
|
+
# First try exact path match
|
|
537
|
+
if file_path in font_groups.groups:
|
|
538
|
+
matching_fonts = font_groups.get_group(file_path)
|
|
539
|
+
else:
|
|
540
|
+
# Try to find by filename in the CSV paths
|
|
541
|
+
for csv_path_key in font_groups.groups.keys():
|
|
542
|
+
if Path(csv_path_key).name == filename:
|
|
543
|
+
matching_fonts = font_groups.get_group(csv_path_key)
|
|
544
|
+
print(f" Matched by filename: {Path(csv_path_key).name}")
|
|
545
|
+
break
|
|
546
|
+
|
|
547
|
+
if matching_fonts is not None:
|
|
548
|
+
# Create font usage entries
|
|
549
|
+
font_usage = []
|
|
550
|
+
|
|
551
|
+
for _, row in matching_fonts.iterrows():
|
|
552
|
+
font_entry = {
|
|
553
|
+
'class': row['class'],
|
|
554
|
+
'instances': int(row['instances']),
|
|
555
|
+
'average_length': float(row['average_length']),
|
|
556
|
+
'usage': row['Usage'] # MainText, Title, Footnotes, Table, Rest
|
|
557
|
+
}
|
|
558
|
+
font_usage.append(font_entry)
|
|
559
|
+
|
|
560
|
+
# Updateno, it has usage summary statistics
|
|
561
|
+
usage_type = row['Usage']
|
|
562
|
+
if usage_type not in stats['font_usage_summary']:
|
|
563
|
+
stats['font_usage_summary'][usage_type] = 0
|
|
564
|
+
stats['font_usage_summary'][usage_type] += 1
|
|
565
|
+
|
|
566
|
+
# Assign the font usage to the file
|
|
567
|
+
file_info['font_usage'] = font_usage
|
|
568
|
+
|
|
569
|
+
# Also assign relevant bounds if available in CSV
|
|
570
|
+
if not matching_fonts.empty:
|
|
571
|
+
first_row = matching_fonts.iloc[0]
|
|
572
|
+
if pd.notna(first_row['RelevantBeginning']) and pd.notna(first_row['RelevantEnd']):
|
|
573
|
+
file_info['relevant_beginning'] = int(first_row['RelevantBeginning'])
|
|
574
|
+
file_info['relevant_end'] = int(first_row['RelevantEnd'])
|
|
575
|
+
print(f" Set bounds: {first_row['RelevantBeginning']} - {first_row['RelevantEnd']}")
|
|
576
|
+
|
|
577
|
+
stats['files_matched'] += 1
|
|
578
|
+
stats['fonts_assigned'] += len(font_usage)
|
|
579
|
+
print(f" Assigned {len(font_usage)} font classifications")
|
|
580
|
+
|
|
581
|
+
else:
|
|
582
|
+
stats['files_not_found'].append(filename)
|
|
583
|
+
print(f" Warning: No font data found for {filename}")
|
|
584
|
+
|
|
585
|
+
stats['files_processed'] += 1
|
|
586
|
+
|
|
587
|
+
# Save the updated inventory
|
|
588
|
+
print(f"\nSaving updated inventory to {self.json_path}...")
|
|
589
|
+
_save_inventory(self.json_path, self.inventory)
|
|
590
|
+
|
|
591
|
+
# Print summary statistics
|
|
592
|
+
print("\n" + "="*50)
|
|
593
|
+
print("FONT ASSIGNMENT SUMMARY")
|
|
594
|
+
print("="*50)
|
|
595
|
+
print(f"Files processed: {stats['files_processed']}")
|
|
596
|
+
print(f"Files matched with font data: {stats['files_matched']}")
|
|
597
|
+
print(f"Total fonts assigned: {stats['fonts_assigned']}")
|
|
598
|
+
print(f"Files without font data: {len(stats['files_not_found'])}")
|
|
599
|
+
|
|
600
|
+
if stats['files_not_found']:
|
|
601
|
+
print("\nFiles not found in CSV:")
|
|
602
|
+
for filename in stats['files_not_found']:
|
|
603
|
+
print(f" - {filename}")
|
|
604
|
+
|
|
605
|
+
print("\nFont usage distribution:")
|
|
606
|
+
for usage_type, count in stats['font_usage_summary'].items():
|
|
607
|
+
print(f" {usage_type}: {count} fonts")
|
|
608
|
+
|
|
609
|
+
return stats
|
|
610
|
+
|
|
611
|
+
def get_font_assignment_sample(self, csv_path: str, file_index: int = None):
|
|
612
|
+
"""
|
|
613
|
+
Generate a sample assignment dictionary for a specific file, useful for verification.
|
|
614
|
+
|
|
615
|
+
Parameters:
|
|
616
|
+
- csv_path (str): Path to the CSV file
|
|
617
|
+
- file_index (int): Index of file in inventory (if None, uses first file as example)
|
|
618
|
+
|
|
619
|
+
Returns:
|
|
620
|
+
- dict: Assignment dictionary in the format used by classify_fonts_usage()
|
|
621
|
+
"""
|
|
622
|
+
df_fonts = pd.read_csv(csv_path)
|
|
623
|
+
|
|
624
|
+
if file_index is not None:
|
|
625
|
+
file_path = self.inventory['files'][file_index]['file_path']
|
|
626
|
+
filename = Path(file_path).name
|
|
627
|
+
file_fonts = df_fonts[df_fonts['File'].apply(lambda p: Path(p).name == filename)]
|
|
628
|
+
else:
|
|
629
|
+
# Use first file as example
|
|
630
|
+
first_file = df_fonts['File'].iloc[0]
|
|
631
|
+
file_fonts = df_fonts[df_fonts['File'] == first_file]
|
|
632
|
+
print(f"Showing example for: {Path(first_file).name}")
|
|
633
|
+
|
|
634
|
+
assignment_dict = {}
|
|
635
|
+
for _, row in file_fonts.iterrows():
|
|
636
|
+
assignment_dict[row['class']] = row['Usage']
|
|
637
|
+
|
|
638
|
+
return assignment_dict
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
def inspect_chunks(self, file_index, show_content=True, outlier_threshold=0.95, max_content_length=500):
|
|
642
|
+
"""
|
|
643
|
+
Inspect chunks after text processing to identify outliers and patterns.
|
|
644
|
+
|
|
645
|
+
Args:
|
|
646
|
+
file_index (int): Index of the file to inspect
|
|
647
|
+
show_content (bool): Whether to show actual chunk content
|
|
648
|
+
outlier_threshold (float): Percentile threshold for outlier detection (0.95 = top/bottom 5%)
|
|
649
|
+
max_content_length (int): Maximum characters to show in content preview
|
|
650
|
+
"""
|
|
651
|
+
if file_index >= len(self.inventory['files']):
|
|
652
|
+
print(f"Error: File index {file_index} out of range")
|
|
653
|
+
return
|
|
654
|
+
|
|
655
|
+
file_info = self.inventory['files'][file_index]
|
|
656
|
+
|
|
657
|
+
# Check if chunks exist
|
|
658
|
+
if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
|
|
659
|
+
print(f"No chunks found for file index {file_index}. Run selecting_text_chunks() first.")
|
|
660
|
+
return
|
|
661
|
+
|
|
662
|
+
chunks = file_info['cleaned_subsections']
|
|
663
|
+
chunk_lengths = [len(chunk) for chunk in chunks]
|
|
664
|
+
|
|
665
|
+
# Basic statistics
|
|
666
|
+
total_chunks = len(chunks)
|
|
667
|
+
avg_length = sum(chunk_lengths) / total_chunks
|
|
668
|
+
median_length = sorted(chunk_lengths)[total_chunks // 2]
|
|
669
|
+
min_length = min(chunk_lengths)
|
|
670
|
+
max_length = max(chunk_lengths)
|
|
671
|
+
std_dev = (sum((x - avg_length) ** 2 for x in chunk_lengths) / total_chunks) ** 0.5
|
|
672
|
+
|
|
673
|
+
# Print summary
|
|
674
|
+
print(f"\nChunk Statistics for file: \"{file_info['file_path'].split('/')[-1]}\"")
|
|
675
|
+
print("=" * 60)
|
|
676
|
+
print(f"Total chunks: {total_chunks}")
|
|
677
|
+
print(f"Average length: {avg_length:,.0f} characters")
|
|
678
|
+
print(f"Median length: {median_length:,.0f} characters")
|
|
679
|
+
print(f"Shortest chunk: {min_length:,} characters")
|
|
680
|
+
print(f"Longest chunk: {max_length:,} characters")
|
|
681
|
+
print(f"Standard deviation: {std_dev:,.0f} characters")
|
|
682
|
+
|
|
683
|
+
# Length distribution
|
|
684
|
+
print(f"\nLength Distribution:")
|
|
685
|
+
bins = [(0, 500), (500, 2000), (2000, 5000), (5000, float('inf'))]
|
|
686
|
+
for min_len, max_len in bins:
|
|
687
|
+
if max_len == float('inf'):
|
|
688
|
+
count = sum(1 for length in chunk_lengths if length > min_len)
|
|
689
|
+
label = f"> {min_len:,} chars:"
|
|
690
|
+
else:
|
|
691
|
+
count = sum(1 for length in chunk_lengths if min_len <= length < max_len)
|
|
692
|
+
label = f"{min_len:,}-{max_len:,}:"
|
|
693
|
+
|
|
694
|
+
percentage = (count / total_chunks) * 100
|
|
695
|
+
print(f" {label:<15} {count:3d} chunks ({percentage:5.1f}%)")
|
|
696
|
+
|
|
697
|
+
if not show_content:
|
|
698
|
+
return
|
|
699
|
+
|
|
700
|
+
# Find outliers
|
|
701
|
+
sorted_indices = sorted(range(len(chunk_lengths)), key=lambda i: chunk_lengths[i])
|
|
702
|
+
|
|
703
|
+
# Bottom outliers
|
|
704
|
+
bottom_count = max(1, int((1 - outlier_threshold) * total_chunks))
|
|
705
|
+
shortest_indices = sorted_indices[:bottom_count]
|
|
706
|
+
|
|
707
|
+
# Top outliers
|
|
708
|
+
top_count = max(1, int((1 - outlier_threshold) * total_chunks))
|
|
709
|
+
longest_indices = sorted_indices[-top_count:]
|
|
710
|
+
|
|
711
|
+
# Show shortest chunks
|
|
712
|
+
print(f"\n=== SHORTEST CHUNKS (bottom {((1-outlier_threshold)*100):.0f}%) ===")
|
|
713
|
+
for i, chunk_idx in enumerate(shortest_indices):
|
|
714
|
+
chunk = chunks[chunk_idx]
|
|
715
|
+
print(f"\nChunk {chunk_idx}: {len(chunk):,} characters")
|
|
716
|
+
preview = chunk[:max_content_length]
|
|
717
|
+
if len(chunk) > max_content_length:
|
|
718
|
+
preview += "..."
|
|
719
|
+
print(f"Content: '{preview}'")
|
|
720
|
+
|
|
721
|
+
# Show longest chunks
|
|
722
|
+
print(f"\n=== LONGEST CHUNKS (top {((1-outlier_threshold)*100):.0f}%) ===")
|
|
723
|
+
for i, chunk_idx in enumerate(longest_indices):
|
|
724
|
+
chunk = chunks[chunk_idx]
|
|
725
|
+
print(f"\nChunk {chunk_idx}: {len(chunk):,} characters")
|
|
726
|
+
preview = chunk[:max_content_length]
|
|
727
|
+
if len(chunk) > max_content_length:
|
|
728
|
+
preview += "..."
|
|
729
|
+
print(f"Content: '{preview}'")
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
def show_chunk(self, file_index, chunk_index, full_content=False, context_chars=100):
|
|
733
|
+
"""
|
|
734
|
+
Display specific chunk content with optional context.
|
|
735
|
+
|
|
736
|
+
Args:
|
|
737
|
+
file_index (int): Index of the file
|
|
738
|
+
chunk_index (int): Index of the chunk to display
|
|
739
|
+
full_content (bool): Whether to show the full chunk or truncated version
|
|
740
|
+
context_chars (int): Number of characters to show before/after if not full_content
|
|
741
|
+
"""
|
|
742
|
+
if file_index >= len(self.inventory['files']):
|
|
743
|
+
print(f"Error: File index {file_index} out of range")
|
|
744
|
+
return
|
|
745
|
+
|
|
746
|
+
file_info = self.inventory['files'][file_index]
|
|
747
|
+
|
|
748
|
+
# Check if chunks exist
|
|
749
|
+
if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
|
|
750
|
+
print(f"No chunks found for file index {file_index}. Run selecting_text_chunks() first.")
|
|
751
|
+
return
|
|
752
|
+
|
|
753
|
+
chunks = file_info['cleaned_subsections']
|
|
754
|
+
|
|
755
|
+
if chunk_index >= len(chunks):
|
|
756
|
+
print(f"Error: Chunk index {chunk_index} out of range. File has {len(chunks)} chunks.")
|
|
757
|
+
return
|
|
758
|
+
|
|
759
|
+
chunk = chunks[chunk_index]
|
|
760
|
+
|
|
761
|
+
print(f"\nChunk {chunk_index} from file: \"{file_info['file_path'].split('/')[-1]}\"")
|
|
762
|
+
print(f"Length: {len(chunk):,} characters")
|
|
763
|
+
print("=" * 60)
|
|
764
|
+
|
|
765
|
+
if full_content:
|
|
766
|
+
print(chunk)
|
|
767
|
+
else:
|
|
768
|
+
# Show truncated content with context
|
|
769
|
+
if len(chunk) <= context_chars * 2:
|
|
770
|
+
print(chunk)
|
|
771
|
+
else:
|
|
772
|
+
start_part = chunk[:context_chars]
|
|
773
|
+
end_part = chunk[-context_chars:]
|
|
774
|
+
middle_chars = len(chunk) - (context_chars * 2)
|
|
775
|
+
print(f"{start_part}")
|
|
776
|
+
print(f"\n... [{middle_chars:,} characters hidden] ...\n")
|
|
777
|
+
print(f"{end_part}")
|
|
778
|
+
|
|
779
|
+
|
|
780
|
+
def get_chunks_summary(self, file_index=None):
|
|
781
|
+
"""
|
|
782
|
+
Get a quick summary of chunks across all files or a specific file.
|
|
783
|
+
|
|
784
|
+
Args:
|
|
785
|
+
file_index (int, optional): Index of specific file. If None, summarizes all files.
|
|
786
|
+
|
|
787
|
+
Returns:
|
|
788
|
+
dict: Summary statistics
|
|
789
|
+
"""
|
|
790
|
+
if file_index is not None:
|
|
791
|
+
# Summary for specific file
|
|
792
|
+
if file_index >= len(self.inventory['files']):
|
|
793
|
+
return {"error": f"File index {file_index} out of range"}
|
|
794
|
+
|
|
795
|
+
file_info = self.inventory['files'][file_index]
|
|
796
|
+
if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
|
|
797
|
+
return {"error": "No chunks found for this file"}
|
|
798
|
+
|
|
799
|
+
chunks = file_info['cleaned_subsections']
|
|
800
|
+
chunk_lengths = [len(chunk) for chunk in chunks]
|
|
801
|
+
|
|
802
|
+
return {
|
|
803
|
+
"file_path": file_info['file_path'].split('/')[-1],
|
|
804
|
+
"total_chunks": len(chunks),
|
|
805
|
+
"total_characters": sum(chunk_lengths),
|
|
806
|
+
"avg_length": sum(chunk_lengths) / len(chunks),
|
|
807
|
+
"min_length": min(chunk_lengths),
|
|
808
|
+
"max_length": max(chunk_lengths),
|
|
809
|
+
}
|
|
810
|
+
|
|
811
|
+
else:
|
|
812
|
+
# Summary for all files
|
|
813
|
+
all_summaries = []
|
|
814
|
+
for i, file_info in enumerate(self.inventory['files']):
|
|
815
|
+
if 'cleaned_subsections' in file_info and file_info['cleaned_subsections']:
|
|
816
|
+
summary = self.get_chunks_summary(i)
|
|
817
|
+
summary['file_index'] = i
|
|
818
|
+
all_summaries.append(summary)
|
|
819
|
+
|
|
820
|
+
return {
|
|
821
|
+
"files_with_chunks": len(all_summaries),
|
|
822
|
+
"total_files": len(self.inventory['files']),
|
|
823
|
+
"files": all_summaries
|
|
824
|
+
}
|
|
825
|
+
|
|
826
|
+
#FUNCTIONS USED FOR GPT-BASED EXTRACTION
|
|
827
|
+
#
|
|
828
|
+
def _save_extraction_results(self, extracted_data: Dict, output_path: str, file_index: int):
|
|
829
|
+
"""Save extraction results to file with metadata"""
|
|
830
|
+
file_info = self.inventory['files'][file_index]
|
|
831
|
+
|
|
832
|
+
result_data = {
|
|
833
|
+
"file_index": file_index,
|
|
834
|
+
"filename": Path(file_info['file_path']).name,
|
|
835
|
+
"extraction_timestamp": pd.Timestamp.now().isoformat(),
|
|
836
|
+
"total_chunks_processed": len(file_info.get('cleaned_subsections', [])),
|
|
837
|
+
"extracted_data": extracted_data
|
|
838
|
+
}
|
|
839
|
+
|
|
840
|
+
with open(output_path, 'w', encoding='utf-8') as f:
|
|
841
|
+
json.dump(result_data, f, indent=2, ensure_ascii=False)
|
|
842
|
+
def is_file_ready_for_llm(self, file_index: int) -> bool:
|
|
843
|
+
"""Check if a file is ready for LLM processing"""
|
|
844
|
+
if file_index >= len(self.inventory['files']):
|
|
845
|
+
return False
|
|
846
|
+
|
|
847
|
+
file_info = self.inventory['files'][file_index]
|
|
848
|
+
|
|
849
|
+
has_bounds = 'relevant_beginning' in file_info and 'relevant_end' in file_info
|
|
850
|
+
has_fonts = any(font.get('usage') for font in file_info.get('fonts', {}).values()) if 'fonts' in file_info else any(font.get('usage') for font in file_info.get('font_usage', []))
|
|
851
|
+
has_chunks = len(file_info.get('cleaned_subsections', [])) > 0
|
|
852
|
+
|
|
853
|
+
return has_bounds and has_fonts and has_chunks
|
|
854
|
+
|
|
855
|
+
def process_with_llm(self, file_index: int,
|
|
856
|
+
llm_processor: 'LLMProcessor',
|
|
857
|
+
extraction_type: str = "relations",
|
|
858
|
+
output_path: str = None) -> Dict:
|
|
859
|
+
"""
|
|
860
|
+
Process cleaned subsections with LLM for information extraction
|
|
861
|
+
"""
|
|
862
|
+
if file_index >= len(self.inventory['files']):
|
|
863
|
+
raise IndexError(f"File index {file_index} out of range")
|
|
864
|
+
|
|
865
|
+
file_info = self.inventory['files'][file_index]
|
|
866
|
+
|
|
867
|
+
if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
|
|
868
|
+
raise ValueError(f"No cleaned subsections found. Run selecting_text_chunks() first.")
|
|
869
|
+
|
|
870
|
+
if not llm_processor:
|
|
871
|
+
raise ValueError("LLM processor is required")
|
|
872
|
+
|
|
873
|
+
# Use the correct method signature
|
|
874
|
+
extracted_data = llm_processor.extract_information(
|
|
875
|
+
file_info['cleaned_subsections'],
|
|
876
|
+
extraction_schema=None, # Will use defaults
|
|
877
|
+
extraction_type=extraction_type
|
|
878
|
+
)
|
|
879
|
+
|
|
880
|
+
# Save results
|
|
881
|
+
if output_path:
|
|
882
|
+
self._save_extraction_results(extracted_data, output_path, file_index)
|
|
883
|
+
|
|
884
|
+
return extracted_data
|
|
885
|
+
|
|
886
|
+
def batch_process_with_llm(self, llm_processor: 'LLMProcessor',
|
|
887
|
+
extraction_type: str = "relations",
|
|
888
|
+
output_dir: str = "llm_results") -> Dict:
|
|
889
|
+
"""
|
|
890
|
+
Process all files with cleaned subsections using LLM
|
|
891
|
+
"""
|
|
892
|
+
results = {}
|
|
893
|
+
output_path = Path(output_dir)
|
|
894
|
+
output_path.mkdir(exist_ok=True)
|
|
895
|
+
|
|
896
|
+
for idx, file_info in enumerate(self.inventory['files']):
|
|
897
|
+
if 'cleaned_subsections' in file_info and file_info['cleaned_subsections']:
|
|
898
|
+
try:
|
|
899
|
+
file_results = self.process_with_llm(
|
|
900
|
+
idx, llm_processor, extraction_type,
|
|
901
|
+
output_path / f"file_{idx}_results.json"
|
|
902
|
+
)
|
|
903
|
+
results[idx] = file_results
|
|
904
|
+
print(f"✓ Processed file {idx}: {Path(file_info['file_path']).name}")
|
|
905
|
+
except Exception as e:
|
|
906
|
+
print(f"✗ Error processing file {idx}: {e}")
|
|
907
|
+
results[idx] = {"error": str(e)}
|
|
908
|
+
|
|
909
|
+
return results
|