text2rel 0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
text2rel/cleaner.py ADDED
@@ -0,0 +1,909 @@
1
+
2
+ import os
3
+ import sys
4
+ import json
5
+ import numpy as np
6
+ import pandas as pd
7
+ import re
8
+ from bs4 import BeautifulSoup
9
+ import ast
10
+ import string
11
+ from pathlib import Path
12
+ from text2rel.utils import insert_jumpjump
13
+ from text2rel.io import _load_inventory, _save_inventory
14
+ from typing import Dict
15
+
16
+
17
+
18
+ class HTMLCleaner:
19
+ def __init__(self, json_path: str, html_directory: str = None, extension: str = ".htm"):
20
+ """
21
+ Initializes the cleaner. If the JSON doesn't exist but a directory is given, it builds the inventory.
22
+ """
23
+ if not json_path:
24
+ raise ValueError("json_path parameter is required. Please provide a path to your JSON inventory file. It can be simply and empty JSON file")
25
+ if not html_directory:
26
+ raise ValueError("html_directory parameter is required. Please provide a path to your HTML files directory.")
27
+ self.json_path = Path(json_path)
28
+
29
+ if not self.json_path.exists():
30
+ if html_directory is None:
31
+ raise FileNotFoundError(f"JSON not found at {json_path} and no directory given to create one.")
32
+ self._create_inventory(html_directory, extension)
33
+ print(f"Created new inventory at {json_path}")
34
+
35
+ self.inventory = _load_inventory(self.json_path)
36
+
37
+ #Creates the inventory from a given directory of HTML files.
38
+ def _create_inventory(self, directory: str, extension: str = ".htm"):
39
+ directory = Path(directory)
40
+ if not directory.exists():
41
+ raise FileNotFoundError(f"Directory {directory} does not exist.")
42
+
43
+ data = []
44
+ for root, _, files in os.walk(directory):
45
+ for file in files:
46
+ if file.endswith(extension):
47
+ data.append({
48
+ "file_path": str(Path(root) / file),
49
+ "relevant_beginning": None, #Needs to be manually set by the user.
50
+ "relevant_end": None, #Needs to be manually set by the user.
51
+ "font_usage": [] #Gets updated during cleaning.
52
+ })
53
+
54
+ inventory = {"files": data}
55
+ with open(self.json_path, 'w', encoding='utf-8') as f:
56
+ json.dump(inventory, f, indent=4)
57
+
58
+ #Lists all files in the inventory with their index for easy reference.
59
+ def list_files(self):
60
+ """
61
+ Prints all file paths with their corresponding index for easy reference.
62
+ """
63
+ print("Inventory File Index:")
64
+ for idx, file_info in enumerate(self.inventory['files']):
65
+ print(f"{idx}: {Path(file_info['file_path']).name}")
66
+
67
+ #Allowing the user to set the relevant beginning and end for each file, the files can be accessed by filename
68
+ def set_relevant_bounds(self, filename: str, beginning, end):
69
+ for file in self.inventory['files']:
70
+ if file['file_path'].endswith(filename):
71
+ file['relevant_beginning'] = beginning
72
+ file['relevant_end'] = end
73
+ _save_inventory(self.json_path,self.inventory)
74
+ return
75
+ raise ValueError(f"File {filename} not found in inventory.")
76
+
77
+ def analyze_html_fonts(self, file_index: int):
78
+ """
79
+ Analyzes the HTML content of a file's relevant section to extract font usage statistics.
80
+ Parameters:
81
+ - file_index (int): Index of the file in the inventory
82
+ Updates:
83
+ - Updates the `font_usage` field in the JSON for the selected file.
84
+ """
85
+ file_info = self.inventory['files'][file_index]
86
+ html_path = Path(file_info['file_path'])
87
+
88
+ if not html_path.exists():
89
+ print(f"File not found: {html_path}")
90
+ return
91
+
92
+ if file_info['relevant_beginning'] is None or file_info['relevant_end'] is None:
93
+ print(f"relevant_beginning and relevant_end must be set first.")
94
+ return
95
+
96
+ with open(html_path, 'r', encoding='utf-8') as f:
97
+ html_content = f.read()
98
+
99
+ relevant_text = html_content[file_info['relevant_beginning']:file_info['relevant_end']-1]
100
+ soup = BeautifulSoup(relevant_text, 'html.parser')
101
+
102
+ # Match font classes like font1, font12, etc.
103
+ pattern = re.compile(r'font\d{1,2}')
104
+ class_stats = {}
105
+
106
+ for tag in soup.find_all(class_=pattern):
107
+ class_list = tag.get('class', [])
108
+ for cls_name in class_list:
109
+ if not pattern.match(cls_name):
110
+ continue
111
+ text = tag.get_text(strip=True)
112
+ if cls_name in class_stats:
113
+ class_stats[cls_name]['total_length'] += len(text)
114
+ class_stats[cls_name]['count'] += 1
115
+ else:
116
+ class_stats[cls_name] = {'total_length': len(text), 'count': 1}
117
+
118
+ # Format and save to inventory
119
+ font_usage = []
120
+ for cls, stats in class_stats.items():
121
+ font_usage.append({
122
+ 'class': cls,
123
+ 'instances': stats['count'],
124
+ 'average_length': stats['total_length'] / stats['count'],
125
+ 'usage': "" # Still to be set by user
126
+ })
127
+
128
+ file_info['font_usage'] = font_usage
129
+ _save_inventory(self.json_path, self.inventory)
130
+ #Display the fonts to the user
131
+ df = pd.DataFrame(font_usage)
132
+ print(df)
133
+ print(f"Font usage statistics for {Path(file_info['file_path']).name} updated.")
134
+
135
+
136
+ #Display the results as a DataFrame
137
+
138
+
139
+ def classify_fonts_usage(self, index: int, usages: dict):
140
+ """
141
+ Classifies font usage for a specific file in the inventory.
142
+
143
+ Parameters:
144
+ - index (int): The index of the file in the inventory.
145
+ - usages (dict): Dictionary of font classes to usage labels (e.g., {'font3': 'MainText'}).
146
+ """
147
+
148
+ if 0 <= index < len(self.inventory['files']):
149
+ file_info = self.inventory['files'][index]
150
+ for font, usage_type in usages.items():
151
+ for font_entry in file_info.get('font_usage', []):
152
+ if font_entry['class'] == font:
153
+ font_entry['usage'] = usage_type
154
+ break
155
+
156
+ _save_inventory(self.json_path,self.inventory)
157
+ print("Updated font usage for file index", index)
158
+ else:
159
+ print(f"Invalid index: {index}")
160
+
161
+ #Remove all the
162
+
163
+ #Converts the inventory into a pandas DataFrame for easy analysis if the user desires it.
164
+
165
+ def to_dataframe(self):
166
+ """
167
+ Converts the inventory into a pandas DataFrame.
168
+
169
+ Returns:
170
+ - pd.DataFrame: Columns include File, RelevantBeginning, RelevantEnd, class, instances, average_length, Usage.
171
+ """
172
+ records = []
173
+
174
+ for file_info in self.inventory['files']:
175
+ for font_entry in file_info.get('font_usage', []):
176
+ records.append({
177
+ 'File': file_info['file_path'],
178
+ 'RelevantBeginning': file_info['relevant_beginning'],
179
+ 'RelevantEnd': file_info['relevant_end'],
180
+ 'class': font_entry['class'],
181
+ 'instances': font_entry['instances'],
182
+ 'average_length': font_entry['average_length'],
183
+ 'Usage': font_entry['usage']
184
+ })
185
+
186
+ def selecting_text_chunks(self):
187
+ json_path = self.json_path
188
+
189
+ for file_info in self.inventory['files']:
190
+ html_file_path = file_info['file_path']
191
+
192
+ # Skip files without bounds or font usage set
193
+ if file_info.get('relevant_beginning') is None or file_info.get('relevant_end') is None:
194
+ continue
195
+ if not any(entry.get('usage') == 'MainText' for entry in file_info.get('font_usage', [])):
196
+ continue
197
+
198
+ try:
199
+ with open(html_file_path, 'r', encoding='utf-8') as f:
200
+ html_content = f.read()
201
+ except FileNotFoundError:
202
+ print(f"File not found: {html_file_path}")
203
+ continue
204
+
205
+ print(f"Processing: {Path(html_file_path).name}")
206
+
207
+ # Slice relevant section, based on the markings in relevant beginning and relevant end
208
+ relevant_text = html_content[file_info['relevant_beginning']:file_info['relevant_end']]
209
+ soup = BeautifulSoup(relevant_text, 'html.parser')
210
+
211
+ # Step 1: Remove (decompose) elements with 'Rest' usage - they are not needed
212
+ for font_entry in file_info.get('font_usage', []):
213
+ if font_entry.get('usage') == 'Rest':
214
+ class_name = font_entry['class']
215
+ for el in soup.find_all(class_=class_name):
216
+ el.decompose()
217
+
218
+ # Step 2: Insert JumPJumP before Title classes
219
+ title_count = 0
220
+ for font_entry in file_info.get('font_usage', []):
221
+ if font_entry.get('usage') == 'Title':
222
+ class_name = font_entry['class']
223
+ for el in soup.find_all(class_=class_name):
224
+ el.insert_before(soup.new_string("JumPJumP"))
225
+ title_count += 1
226
+
227
+ # Step 3: Extract ALL remaining text (includes JumPJumP markers and titles)
228
+ text = soup.get_text()
229
+
230
+ print(f" → {title_count} title elements found (JumPJumP inserted before each)")
231
+
232
+
233
+ # Clean up the text
234
+ text = re.sub(r'\n+', '\n', text)
235
+ text = re.sub(r'[ ]+', ' ', text)
236
+
237
+ excluded_punctuation = re.escape(string.punctuation.replace('.', ''))
238
+ text = re.sub(
239
+ fr"([a-záéíóúü]) ?([{excluded_punctuation}])?\n([{excluded_punctuation}])? ?([a-záéíóúü])",
240
+ r'\1 \4',
241
+ text,
242
+ flags=re.IGNORECASE
243
+ )
244
+
245
+ # Count JumPJumP from titles only
246
+ jumps_from_titles = text.count("JumPJumP")
247
+
248
+ # Note: Length-based chopping is handled in apply_file_specific_rules()
249
+ # based on the LengthChop flag in the protocol CSV
250
+
251
+ # Merge short sections unnecessarily split by jumps
252
+ while True:
253
+ new_text = re.sub(r'JumPJumP(.{1,500})JumPJumP', r'JumPJumP\1', text, flags=re.DOTALL)
254
+ if new_text == text:
255
+ break
256
+
257
+ text = new_text
258
+
259
+ # Consolidate multiple consecutive JumPJumP markers into one
260
+ text = re.sub(r'(\n*JumPJumP\n*)+', '\nJumPJumP\n', text)
261
+
262
+ # Split by marker and embed directly in JSON
263
+ text_sections = [t.strip() for t in re.split(r'\n*JumPJumP\n*', text) if t.strip()]
264
+ file_info['cleaned_subsections'] = text_sections
265
+
266
+ print(f" → JumPJumP from titles: {jumps_from_titles}")
267
+ print(f" → Initial subsections: {len(text_sections)}")
268
+
269
+ # Save updated JSON
270
+ _save_inventory(json_path, self.inventory)
271
+
272
+ print("Cleaned text added to each file entry in the JSON.")
273
+
274
+
275
+ ###_____________REMOVE WHEN EVERYTHING REGARDING FILE-SPECIFIC RULES IS FULLY TESTED AND WORKING________________####
276
+ # def apply_cleaning_rules(self, file_index, rule_keys: list = None, multiline: bool = False):
277
+ # """
278
+ # Applies selected regex cleaning rules from cleaning_rules.py to each subsection.
279
+
280
+ # Parameters:
281
+ # - file_index (int): The index of the file in the inventory
282
+ # - rule_keys (list[str]): List of rule category keys to apply.
283
+ # If None or empty, no rules will be applied.
284
+ # - multiline (bool): Whether to apply rules across multiple lines using re.DOTALL
285
+
286
+ # Returns:
287
+ # - list[str]: List of cleaned text sections
288
+ # """
289
+ # # Get the file info from inventory
290
+ # file_info = self.inventory['files'][file_index]
291
+
292
+ # # Skip if no rules specified
293
+ # if not rule_keys:
294
+ # return file_info["cleaned_subsections"]
295
+
296
+ # cleaned_sections = []
297
+
298
+ # # Process each subsection
299
+ # for section in file_info["cleaned_subsections"]:
300
+ # cleaned_text = section
301
+
302
+ # # Apply each rule category
303
+ # for key in rule_keys:
304
+ # rules = CLEANING_RULES.get(key, [])
305
+ # print(f"Applying rules for {key}")
306
+
307
+ # # Apply each pattern/replacement pair in the rule category
308
+ # for pattern, repl in rules:
309
+ # if multiline:
310
+ # print(f"Applying multiline rule: {pattern} -> {repl}")
311
+ # # Use re.DOTALL to match across multiple lines
312
+ # cleaned_text = re.sub(pattern, repl, cleaned_text, flags=re.DOTALL)
313
+ # else:
314
+ # # Use default behavior (single line)
315
+ # cleaned_text = re.sub(pattern, repl, cleaned_text)
316
+
317
+ # cleaned_sections.append(cleaned_text)
318
+
319
+ # # Update the cleaned subsections in the inventory
320
+ # file_info["cleaned_subsections"] = cleaned_sections
321
+
322
+ # # Save changes to JSON
323
+ # with open(self.json_path, 'w', encoding='utf-8') as f:
324
+ # json.dump(self.inventory, f, indent=4, ensure_ascii=False)
325
+
326
+ # return cleaned_sections
327
+
328
+ def apply_file_specific_rules(self, file_index: int):
329
+ """
330
+ Applies file-specific regex rules from the attached protocol CSV.
331
+ """
332
+ if not hasattr(self, "protocol_df"):
333
+ print("No protocol attached. Use attach_protocol() first.")
334
+ return
335
+
336
+ file_info = self.inventory["files"][file_index]
337
+ filename = Path(file_info["file_path"]).name
338
+
339
+ # Find matching protocol entry by filename
340
+ matches = self.protocol_df[
341
+ self.protocol_df['HTML_Location'].apply(lambda p: Path(p).name == filename)
342
+ ]
343
+
344
+ if len(matches) == 0:
345
+ print(f"No protocol entry found for file: {filename}")
346
+ return
347
+ elif len(matches) > 1:
348
+ print(f"Multiple entries found for {filename}. Using the first match.")
349
+
350
+ match = matches.iloc[0]
351
+
352
+ # Parse the rules dictionary from the CSV
353
+ try:
354
+ rules = ast.literal_eval(match['patterns_replacements'])
355
+ if not isinstance(rules, dict):
356
+ raise ValueError("Parsed rules are not a dictionary")
357
+ except Exception as e:
358
+ print(f"Error parsing rules for {filename}: {e}")
359
+ return
360
+
361
+ # Skip if no rules or empty rules
362
+ if not rules or rules == {'': ''}:
363
+ return
364
+
365
+ # Get flags from protocol
366
+ length_chop = bool(int(match.get('LengthChop', 0)))
367
+ use_dotall = bool(int(match.get('LongReplacement', 0)))
368
+
369
+ # Get current subsections
370
+ text_sections = file_info.get("cleaned_subsections", [])
371
+ if not text_sections:
372
+ print(f"⚠ No cleaned_subsections found for: {filename} (skipping)")
373
+ return
374
+
375
+ print(f"Processing: {filename}")
376
+ print(f" → {len(text_sections)} subsection(s) to process")
377
+ print(f" → Rules to apply:")
378
+ for p, r in rules.items():
379
+ print(f" '{p[:50]}...' → '{r[:30]}...'" if len(p) > 50 else f" '{p}' → '{r}'")
380
+ if length_chop:
381
+ print(f" → LengthChop enabled (will insert JumPJumP every 10k chars)")
382
+
383
+ cleaned_sections = []
384
+ total_modified = 0
385
+ total_regex_matches = 0
386
+
387
+ for text in text_sections:
388
+ original_text = text
389
+ section_matches = 0
390
+
391
+ # Apply replacements and count matches
392
+ for pattern, repl in rules.items():
393
+ if pattern: # Skip empty patterns
394
+ matches_before = len(re.findall(pattern, text, flags=re.DOTALL if use_dotall else 0))
395
+ section_matches += matches_before
396
+ text = re.sub(pattern, repl, text, flags=re.DOTALL if use_dotall else 0)
397
+
398
+ total_regex_matches += section_matches
399
+
400
+ # Track if any changes were made
401
+ if text != original_text:
402
+ total_modified += 1
403
+
404
+ # Insert jump markers for long sections (if LengthChop=1)
405
+ if length_chop:
406
+ text = insert_jumpjump(text, max_chars=10000)
407
+
408
+ # Consolidate multiple consecutive JumPJumP markers into one
409
+ text = re.sub(r'(\n*JumPJumP\n*)+', '\nJumPJumP\n', text)
410
+
411
+ # Merge short sections (less than 500 chars) that were unnecessarily split
412
+ while True:
413
+ new_text = re.sub(r'JumPJumP(.{1,500})JumPJumP', r'JumPJumP\1', text, flags=re.DOTALL)
414
+ if new_text == text:
415
+ break
416
+ text = new_text
417
+
418
+ # Re-split by JumPJumP in case regex patterns added new markers
419
+ sub_sections = [t.strip() for t in re.split(r'\n*JumPJumP\n*', text) if t.strip()]
420
+ cleaned_sections.extend(sub_sections)
421
+
422
+ # Update the inventory with cleaned sections
423
+ file_info["cleaned_subsections"] = cleaned_sections
424
+ _save_inventory(self.json_path, self.inventory)
425
+
426
+ print(f" {total_regex_matches} regex matches applied to {total_modified}/{len(text_sections)} subsection(s)")
427
+ print(f" Final subsections after re-split: {len(cleaned_sections)}")
428
+
429
+ def attach_protocol(self, csv_path: str):
430
+ """
431
+ Load and cache the protocol CSV that contains file-specific cleaning rules.
432
+ """
433
+ self.protocol_df = pd.read_csv(csv_path)
434
+ print(f"Protocol CSV loaded with {len(self.protocol_df)} entries.")
435
+
436
+ def export_cleaned_text(self, output_path: str = None, include_empty: bool = False):
437
+ """
438
+ Exports cleaned subsections to a separate JSON file for downstream use.
439
+
440
+ Parameters:
441
+ - output_path (str): Path to save the cleaned JSON.
442
+ If None, uses '<original_name>_cleaned.json'
443
+ - include_empty (bool): Whether to include files with no cleaned_subsections
444
+
445
+ Returns:
446
+ - str: Path to the exported file
447
+ """
448
+ if output_path is None:
449
+ # Generate default name from inventory path
450
+ stem = self.json_path.stem
451
+ output_path = self.json_path.parent / f"{stem}_cleaned.json"
452
+ else:
453
+ output_path = Path(output_path)
454
+
455
+ cleaned_data = {
456
+ "source_inventory": str(self.json_path),
457
+ "exported_at": pd.Timestamp.now().isoformat(),
458
+ "files": []
459
+ }
460
+
461
+ files_with_content = 0
462
+ total_subsections = 0
463
+
464
+ for file_info in self.inventory["files"]:
465
+ subsections = file_info.get("cleaned_subsections", [])
466
+
467
+ # Skip empty files unless include_empty is True
468
+ if not subsections and not include_empty:
469
+ continue
470
+
471
+ cleaned_entry = {
472
+ "file_path": file_info["file_path"],
473
+ "filename": Path(file_info["file_path"]).name,
474
+ "subsection_count": len(subsections),
475
+ "cleaned_subsections": subsections
476
+ }
477
+
478
+ cleaned_data["files"].append(cleaned_entry)
479
+
480
+ if subsections:
481
+ files_with_content += 1
482
+ total_subsections += len(subsections)
483
+
484
+ # Save to file
485
+ with open(output_path, 'w', encoding='utf-8') as f:
486
+ json.dump(cleaned_data, f, indent=4, ensure_ascii=False)
487
+
488
+ print(f"Exported cleaned text to: {output_path}")
489
+ print(f" → {files_with_content} file(s) with content")
490
+ print(f" → {total_subsections} total subsection(s)")
491
+
492
+ return str(output_path)
493
+
494
+ def load_fonts_from_csv(self, csv_path: str):
495
+ """
496
+ Load font classifications from a comprehensive CSV file (like 231228_ProtocolForPreliminaryCleaning.csv)
497
+ and automatically assign them to all files in the inventory.
498
+
499
+ This replaces the manual font assignment workflow with automated loading from the original protocol.
500
+
501
+ Parameters:
502
+ - csv_path (str): Path to the CSV file containing font classifications
503
+
504
+ Returns:
505
+ - dict: Statistics about the font assignment process
506
+ """
507
+ print("Loading font classification data from CSV...")
508
+ df_fonts = pd.read_csv(csv_path)
509
+ print(f"Loaded {len(df_fonts)} font classification records")
510
+
511
+ # Statistics tracking
512
+ stats = {
513
+ 'files_processed': 0,
514
+ 'fonts_assigned': 0,
515
+ 'files_matched': 0,
516
+ 'files_not_found': [],
517
+ 'font_usage_summary': {}
518
+ }
519
+
520
+ # Group by file path for easier processing
521
+ print("\nGrouping font data by file...")
522
+ font_groups = df_fonts.groupby('File')
523
+
524
+ # Process each file in the inventory
525
+ print("\nAssigning fonts to inventory files...")
526
+
527
+ for file_idx, file_info in enumerate(self.inventory['files']):
528
+ file_path = file_info['file_path']
529
+ filename = Path(file_path).name
530
+
531
+ print(f"Processing file {file_idx}: {filename}")
532
+
533
+ # Find matching entries in the CSV (try both full path and filename)
534
+ matching_fonts = None
535
+
536
+ # First try exact path match
537
+ if file_path in font_groups.groups:
538
+ matching_fonts = font_groups.get_group(file_path)
539
+ else:
540
+ # Try to find by filename in the CSV paths
541
+ for csv_path_key in font_groups.groups.keys():
542
+ if Path(csv_path_key).name == filename:
543
+ matching_fonts = font_groups.get_group(csv_path_key)
544
+ print(f" Matched by filename: {Path(csv_path_key).name}")
545
+ break
546
+
547
+ if matching_fonts is not None:
548
+ # Create font usage entries
549
+ font_usage = []
550
+
551
+ for _, row in matching_fonts.iterrows():
552
+ font_entry = {
553
+ 'class': row['class'],
554
+ 'instances': int(row['instances']),
555
+ 'average_length': float(row['average_length']),
556
+ 'usage': row['Usage'] # MainText, Title, Footnotes, Table, Rest
557
+ }
558
+ font_usage.append(font_entry)
559
+
560
+ # Updateno, it has usage summary statistics
561
+ usage_type = row['Usage']
562
+ if usage_type not in stats['font_usage_summary']:
563
+ stats['font_usage_summary'][usage_type] = 0
564
+ stats['font_usage_summary'][usage_type] += 1
565
+
566
+ # Assign the font usage to the file
567
+ file_info['font_usage'] = font_usage
568
+
569
+ # Also assign relevant bounds if available in CSV
570
+ if not matching_fonts.empty:
571
+ first_row = matching_fonts.iloc[0]
572
+ if pd.notna(first_row['RelevantBeginning']) and pd.notna(first_row['RelevantEnd']):
573
+ file_info['relevant_beginning'] = int(first_row['RelevantBeginning'])
574
+ file_info['relevant_end'] = int(first_row['RelevantEnd'])
575
+ print(f" Set bounds: {first_row['RelevantBeginning']} - {first_row['RelevantEnd']}")
576
+
577
+ stats['files_matched'] += 1
578
+ stats['fonts_assigned'] += len(font_usage)
579
+ print(f" Assigned {len(font_usage)} font classifications")
580
+
581
+ else:
582
+ stats['files_not_found'].append(filename)
583
+ print(f" Warning: No font data found for {filename}")
584
+
585
+ stats['files_processed'] += 1
586
+
587
+ # Save the updated inventory
588
+ print(f"\nSaving updated inventory to {self.json_path}...")
589
+ _save_inventory(self.json_path, self.inventory)
590
+
591
+ # Print summary statistics
592
+ print("\n" + "="*50)
593
+ print("FONT ASSIGNMENT SUMMARY")
594
+ print("="*50)
595
+ print(f"Files processed: {stats['files_processed']}")
596
+ print(f"Files matched with font data: {stats['files_matched']}")
597
+ print(f"Total fonts assigned: {stats['fonts_assigned']}")
598
+ print(f"Files without font data: {len(stats['files_not_found'])}")
599
+
600
+ if stats['files_not_found']:
601
+ print("\nFiles not found in CSV:")
602
+ for filename in stats['files_not_found']:
603
+ print(f" - {filename}")
604
+
605
+ print("\nFont usage distribution:")
606
+ for usage_type, count in stats['font_usage_summary'].items():
607
+ print(f" {usage_type}: {count} fonts")
608
+
609
+ return stats
610
+
611
+ def get_font_assignment_sample(self, csv_path: str, file_index: int = None):
612
+ """
613
+ Generate a sample assignment dictionary for a specific file, useful for verification.
614
+
615
+ Parameters:
616
+ - csv_path (str): Path to the CSV file
617
+ - file_index (int): Index of file in inventory (if None, uses first file as example)
618
+
619
+ Returns:
620
+ - dict: Assignment dictionary in the format used by classify_fonts_usage()
621
+ """
622
+ df_fonts = pd.read_csv(csv_path)
623
+
624
+ if file_index is not None:
625
+ file_path = self.inventory['files'][file_index]['file_path']
626
+ filename = Path(file_path).name
627
+ file_fonts = df_fonts[df_fonts['File'].apply(lambda p: Path(p).name == filename)]
628
+ else:
629
+ # Use first file as example
630
+ first_file = df_fonts['File'].iloc[0]
631
+ file_fonts = df_fonts[df_fonts['File'] == first_file]
632
+ print(f"Showing example for: {Path(first_file).name}")
633
+
634
+ assignment_dict = {}
635
+ for _, row in file_fonts.iterrows():
636
+ assignment_dict[row['class']] = row['Usage']
637
+
638
+ return assignment_dict
639
+
640
+
641
+ def inspect_chunks(self, file_index, show_content=True, outlier_threshold=0.95, max_content_length=500):
642
+ """
643
+ Inspect chunks after text processing to identify outliers and patterns.
644
+
645
+ Args:
646
+ file_index (int): Index of the file to inspect
647
+ show_content (bool): Whether to show actual chunk content
648
+ outlier_threshold (float): Percentile threshold for outlier detection (0.95 = top/bottom 5%)
649
+ max_content_length (int): Maximum characters to show in content preview
650
+ """
651
+ if file_index >= len(self.inventory['files']):
652
+ print(f"Error: File index {file_index} out of range")
653
+ return
654
+
655
+ file_info = self.inventory['files'][file_index]
656
+
657
+ # Check if chunks exist
658
+ if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
659
+ print(f"No chunks found for file index {file_index}. Run selecting_text_chunks() first.")
660
+ return
661
+
662
+ chunks = file_info['cleaned_subsections']
663
+ chunk_lengths = [len(chunk) for chunk in chunks]
664
+
665
+ # Basic statistics
666
+ total_chunks = len(chunks)
667
+ avg_length = sum(chunk_lengths) / total_chunks
668
+ median_length = sorted(chunk_lengths)[total_chunks // 2]
669
+ min_length = min(chunk_lengths)
670
+ max_length = max(chunk_lengths)
671
+ std_dev = (sum((x - avg_length) ** 2 for x in chunk_lengths) / total_chunks) ** 0.5
672
+
673
+ # Print summary
674
+ print(f"\nChunk Statistics for file: \"{file_info['file_path'].split('/')[-1]}\"")
675
+ print("=" * 60)
676
+ print(f"Total chunks: {total_chunks}")
677
+ print(f"Average length: {avg_length:,.0f} characters")
678
+ print(f"Median length: {median_length:,.0f} characters")
679
+ print(f"Shortest chunk: {min_length:,} characters")
680
+ print(f"Longest chunk: {max_length:,} characters")
681
+ print(f"Standard deviation: {std_dev:,.0f} characters")
682
+
683
+ # Length distribution
684
+ print(f"\nLength Distribution:")
685
+ bins = [(0, 500), (500, 2000), (2000, 5000), (5000, float('inf'))]
686
+ for min_len, max_len in bins:
687
+ if max_len == float('inf'):
688
+ count = sum(1 for length in chunk_lengths if length > min_len)
689
+ label = f"> {min_len:,} chars:"
690
+ else:
691
+ count = sum(1 for length in chunk_lengths if min_len <= length < max_len)
692
+ label = f"{min_len:,}-{max_len:,}:"
693
+
694
+ percentage = (count / total_chunks) * 100
695
+ print(f" {label:<15} {count:3d} chunks ({percentage:5.1f}%)")
696
+
697
+ if not show_content:
698
+ return
699
+
700
+ # Find outliers
701
+ sorted_indices = sorted(range(len(chunk_lengths)), key=lambda i: chunk_lengths[i])
702
+
703
+ # Bottom outliers
704
+ bottom_count = max(1, int((1 - outlier_threshold) * total_chunks))
705
+ shortest_indices = sorted_indices[:bottom_count]
706
+
707
+ # Top outliers
708
+ top_count = max(1, int((1 - outlier_threshold) * total_chunks))
709
+ longest_indices = sorted_indices[-top_count:]
710
+
711
+ # Show shortest chunks
712
+ print(f"\n=== SHORTEST CHUNKS (bottom {((1-outlier_threshold)*100):.0f}%) ===")
713
+ for i, chunk_idx in enumerate(shortest_indices):
714
+ chunk = chunks[chunk_idx]
715
+ print(f"\nChunk {chunk_idx}: {len(chunk):,} characters")
716
+ preview = chunk[:max_content_length]
717
+ if len(chunk) > max_content_length:
718
+ preview += "..."
719
+ print(f"Content: '{preview}'")
720
+
721
+ # Show longest chunks
722
+ print(f"\n=== LONGEST CHUNKS (top {((1-outlier_threshold)*100):.0f}%) ===")
723
+ for i, chunk_idx in enumerate(longest_indices):
724
+ chunk = chunks[chunk_idx]
725
+ print(f"\nChunk {chunk_idx}: {len(chunk):,} characters")
726
+ preview = chunk[:max_content_length]
727
+ if len(chunk) > max_content_length:
728
+ preview += "..."
729
+ print(f"Content: '{preview}'")
730
+
731
+
732
+ def show_chunk(self, file_index, chunk_index, full_content=False, context_chars=100):
733
+ """
734
+ Display specific chunk content with optional context.
735
+
736
+ Args:
737
+ file_index (int): Index of the file
738
+ chunk_index (int): Index of the chunk to display
739
+ full_content (bool): Whether to show the full chunk or truncated version
740
+ context_chars (int): Number of characters to show before/after if not full_content
741
+ """
742
+ if file_index >= len(self.inventory['files']):
743
+ print(f"Error: File index {file_index} out of range")
744
+ return
745
+
746
+ file_info = self.inventory['files'][file_index]
747
+
748
+ # Check if chunks exist
749
+ if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
750
+ print(f"No chunks found for file index {file_index}. Run selecting_text_chunks() first.")
751
+ return
752
+
753
+ chunks = file_info['cleaned_subsections']
754
+
755
+ if chunk_index >= len(chunks):
756
+ print(f"Error: Chunk index {chunk_index} out of range. File has {len(chunks)} chunks.")
757
+ return
758
+
759
+ chunk = chunks[chunk_index]
760
+
761
+ print(f"\nChunk {chunk_index} from file: \"{file_info['file_path'].split('/')[-1]}\"")
762
+ print(f"Length: {len(chunk):,} characters")
763
+ print("=" * 60)
764
+
765
+ if full_content:
766
+ print(chunk)
767
+ else:
768
+ # Show truncated content with context
769
+ if len(chunk) <= context_chars * 2:
770
+ print(chunk)
771
+ else:
772
+ start_part = chunk[:context_chars]
773
+ end_part = chunk[-context_chars:]
774
+ middle_chars = len(chunk) - (context_chars * 2)
775
+ print(f"{start_part}")
776
+ print(f"\n... [{middle_chars:,} characters hidden] ...\n")
777
+ print(f"{end_part}")
778
+
779
+
780
+ def get_chunks_summary(self, file_index=None):
781
+ """
782
+ Get a quick summary of chunks across all files or a specific file.
783
+
784
+ Args:
785
+ file_index (int, optional): Index of specific file. If None, summarizes all files.
786
+
787
+ Returns:
788
+ dict: Summary statistics
789
+ """
790
+ if file_index is not None:
791
+ # Summary for specific file
792
+ if file_index >= len(self.inventory['files']):
793
+ return {"error": f"File index {file_index} out of range"}
794
+
795
+ file_info = self.inventory['files'][file_index]
796
+ if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
797
+ return {"error": "No chunks found for this file"}
798
+
799
+ chunks = file_info['cleaned_subsections']
800
+ chunk_lengths = [len(chunk) for chunk in chunks]
801
+
802
+ return {
803
+ "file_path": file_info['file_path'].split('/')[-1],
804
+ "total_chunks": len(chunks),
805
+ "total_characters": sum(chunk_lengths),
806
+ "avg_length": sum(chunk_lengths) / len(chunks),
807
+ "min_length": min(chunk_lengths),
808
+ "max_length": max(chunk_lengths),
809
+ }
810
+
811
+ else:
812
+ # Summary for all files
813
+ all_summaries = []
814
+ for i, file_info in enumerate(self.inventory['files']):
815
+ if 'cleaned_subsections' in file_info and file_info['cleaned_subsections']:
816
+ summary = self.get_chunks_summary(i)
817
+ summary['file_index'] = i
818
+ all_summaries.append(summary)
819
+
820
+ return {
821
+ "files_with_chunks": len(all_summaries),
822
+ "total_files": len(self.inventory['files']),
823
+ "files": all_summaries
824
+ }
825
+
826
+ #FUNCTIONS USED FOR GPT-BASED EXTRACTION
827
+ #
828
+ def _save_extraction_results(self, extracted_data: Dict, output_path: str, file_index: int):
829
+ """Save extraction results to file with metadata"""
830
+ file_info = self.inventory['files'][file_index]
831
+
832
+ result_data = {
833
+ "file_index": file_index,
834
+ "filename": Path(file_info['file_path']).name,
835
+ "extraction_timestamp": pd.Timestamp.now().isoformat(),
836
+ "total_chunks_processed": len(file_info.get('cleaned_subsections', [])),
837
+ "extracted_data": extracted_data
838
+ }
839
+
840
+ with open(output_path, 'w', encoding='utf-8') as f:
841
+ json.dump(result_data, f, indent=2, ensure_ascii=False)
842
+ def is_file_ready_for_llm(self, file_index: int) -> bool:
843
+ """Check if a file is ready for LLM processing"""
844
+ if file_index >= len(self.inventory['files']):
845
+ return False
846
+
847
+ file_info = self.inventory['files'][file_index]
848
+
849
+ has_bounds = 'relevant_beginning' in file_info and 'relevant_end' in file_info
850
+ has_fonts = any(font.get('usage') for font in file_info.get('fonts', {}).values()) if 'fonts' in file_info else any(font.get('usage') for font in file_info.get('font_usage', []))
851
+ has_chunks = len(file_info.get('cleaned_subsections', [])) > 0
852
+
853
+ return has_bounds and has_fonts and has_chunks
854
+
855
+ def process_with_llm(self, file_index: int,
856
+ llm_processor: 'LLMProcessor',
857
+ extraction_type: str = "relations",
858
+ output_path: str = None) -> Dict:
859
+ """
860
+ Process cleaned subsections with LLM for information extraction
861
+ """
862
+ if file_index >= len(self.inventory['files']):
863
+ raise IndexError(f"File index {file_index} out of range")
864
+
865
+ file_info = self.inventory['files'][file_index]
866
+
867
+ if 'cleaned_subsections' not in file_info or not file_info['cleaned_subsections']:
868
+ raise ValueError(f"No cleaned subsections found. Run selecting_text_chunks() first.")
869
+
870
+ if not llm_processor:
871
+ raise ValueError("LLM processor is required")
872
+
873
+ # Use the correct method signature
874
+ extracted_data = llm_processor.extract_information(
875
+ file_info['cleaned_subsections'],
876
+ extraction_schema=None, # Will use defaults
877
+ extraction_type=extraction_type
878
+ )
879
+
880
+ # Save results
881
+ if output_path:
882
+ self._save_extraction_results(extracted_data, output_path, file_index)
883
+
884
+ return extracted_data
885
+
886
+ def batch_process_with_llm(self, llm_processor: 'LLMProcessor',
887
+ extraction_type: str = "relations",
888
+ output_dir: str = "llm_results") -> Dict:
889
+ """
890
+ Process all files with cleaned subsections using LLM
891
+ """
892
+ results = {}
893
+ output_path = Path(output_dir)
894
+ output_path.mkdir(exist_ok=True)
895
+
896
+ for idx, file_info in enumerate(self.inventory['files']):
897
+ if 'cleaned_subsections' in file_info and file_info['cleaned_subsections']:
898
+ try:
899
+ file_results = self.process_with_llm(
900
+ idx, llm_processor, extraction_type,
901
+ output_path / f"file_{idx}_results.json"
902
+ )
903
+ results[idx] = file_results
904
+ print(f"✓ Processed file {idx}: {Path(file_info['file_path']).name}")
905
+ except Exception as e:
906
+ print(f"✗ Error processing file {idx}: {e}")
907
+ results[idx] = {"error": str(e)}
908
+
909
+ return results