text2rel 0.1__tar.gz → 0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: text2rel
3
- Version: 0.1
4
- Summary: An HTML cleaner customized for kinship ties extraction from large text corpra.
3
+ Version: 0.2
4
+ Summary: An HTML cleaner, LLM processor and Reliability Assessment customized for kinship ties extraction from large text corpra.
5
5
  Requires-Python: >=3.10,<3.13
6
6
  Classifier: Programming Language :: Python :: 3
7
7
  Classifier: Programming Language :: Python :: 3.10
@@ -1,7 +1,7 @@
1
1
  [tool.poetry]
2
2
  name = "text2rel"
3
- version = "0.1"
4
- description = "An HTML cleaner customized for kinship ties extraction from large text corpra."
3
+ version = "0.2"
4
+ description = "An HTML cleaner, LLM processor and Reliability Assessment customized for kinship ties extraction from large text corpra."
5
5
 
6
6
  readme = "readme.md"
7
7
 
@@ -0,0 +1,12 @@
1
+ from .cleaner import HTMLCleaner
2
+ from .llm_processor import LLMProcessor
3
+ from .reliability_assessor import ReliabilityAssessment
4
+ from .state import set_default_inventory_path, get_default_inventory_path
5
+
6
+ __all__ = [
7
+ "HTMLCleaner",
8
+ "LLMProcessor",
9
+ "ReliabilityAssessment",
10
+ "set_default_inventory_path",
11
+ "get_default_inventory_path",
12
+ ]
@@ -2,6 +2,7 @@
2
2
  import os
3
3
  import sys
4
4
  import json
5
+ import webbrowser
5
6
  import numpy as np
6
7
  import pandas as pd
7
8
  import re
@@ -11,26 +12,32 @@ import string
11
12
  from pathlib import Path
12
13
  from text2rel.utils import insert_jumpjump
13
14
  from text2rel.io import _load_inventory, _save_inventory
14
- from typing import Dict
15
+ from text2rel.state import set_default_inventory_path
16
+ from typing import Dict, TYPE_CHECKING, Optional, List, Any
17
+
18
+ if TYPE_CHECKING:
19
+ from text2rel.llm_processor import LLMProcessor
15
20
 
16
21
 
17
22
 
18
23
  class HTMLCleaner:
19
- def __init__(self, json_path: str, html_directory: str = None, extension: str = ".htm"):
24
+ def __init__(self, json_name: str, html_directory: str = None, extension: str = ".htm"):
20
25
  """
21
26
  Initializes the cleaner. If the JSON doesn't exist but a directory is given, it builds the inventory.
22
27
  """
23
- if not json_path:
24
- raise ValueError("json_path parameter is required. Please provide a path to your JSON inventory file. It can be simply and empty JSON file")
28
+ if not json_name:
29
+ raise ValueError("json_name parameter is required. Please provide a name for your JSON inventory file. It can be simply an empty JSON file")
25
30
  if not html_directory:
26
31
  raise ValueError("html_directory parameter is required. Please provide a path to your HTML files directory.")
27
- self.json_path = Path(json_path)
32
+
33
+ self.json_path = Path(json_name)
34
+ set_default_inventory_path(self.json_path)
28
35
 
29
36
  if not self.json_path.exists():
30
37
  if html_directory is None:
31
38
  raise FileNotFoundError(f"JSON not found at {json_path} and no directory given to create one.")
32
39
  self._create_inventory(html_directory, extension)
33
- print(f"Created new inventory at {json_path}")
40
+ print(f"Created new inventory at {json_name}")
34
41
 
35
42
  self.inventory = _load_inventory(self.json_path)
36
43
 
@@ -64,6 +71,29 @@ class HTMLCleaner:
64
71
  for idx, file_info in enumerate(self.inventory['files']):
65
72
  print(f"{idx}: {Path(file_info['file_path']).name}")
66
73
 
74
+ def show_file(self, file_index: int, open_in_browser: bool = True) -> str:
75
+ """
76
+ Open an indexed HTML file from the inventory.
77
+
78
+ Parameters:
79
+ - file_index (int): index in self.inventory['files']
80
+ - open_in_browser (bool): if True, opens the file in the default browser
81
+
82
+ Returns:
83
+ - str: absolute path of the selected file
84
+ """
85
+ if file_index < 0 or file_index >= len(self.inventory['files']):
86
+ raise IndexError(f"file_index {file_index} out of range")
87
+
88
+ file_path = Path(self.inventory['files'][file_index]['file_path']).resolve()
89
+ if not file_path.exists():
90
+ raise FileNotFoundError(f"File not found: {file_path}")
91
+
92
+ if open_in_browser:
93
+ webbrowser.open(file_path.as_uri())
94
+
95
+ return str(file_path)
96
+
67
97
  #Allowing the user to set the relevant beginning and end for each file, the files can be accessed by filename
68
98
  def set_relevant_bounds(self, filename: str, beginning, end):
69
99
  for file in self.inventory['files']:
@@ -73,6 +103,179 @@ class HTMLCleaner:
73
103
  _save_inventory(self.json_path,self.inventory)
74
104
  return
75
105
  raise ValueError(f"File {filename} not found in inventory.")
106
+
107
+ def find_str(
108
+ self,
109
+ what: str,
110
+ where: Optional[str] = None,
111
+ window_left: int = 120,
112
+ window_right: int = 120,
113
+ *,
114
+ file_index: Optional[int] = None,
115
+ use_regex: bool = False,
116
+ html_flexible: bool = False,
117
+ robust_html: bool = False,
118
+ ignore_case: bool = True,
119
+ max_matches: Optional[int] = None,
120
+ return_matches: bool = False,
121
+ ) -> Optional[List[Dict[str, Any]]]:
122
+ """
123
+ Search text (or an inventory HTML file) and print contextual matches.
124
+
125
+ Parameters:
126
+ - what: query string or regex pattern.
127
+ - where: raw text to search. If None, file_index is used to load a file.
128
+ - window_left/window_right: chars shown before/after each match.
129
+ - file_index: inventory index used when where is None.
130
+ - use_regex: interpret `what` as regex pattern.
131
+ - html_flexible: for literal search, allow whitespace/newline drift.
132
+ This uses a fast token-based pattern by default.
133
+ - robust_html: if True with html_flexible, use a heavier HTML-aware
134
+ matcher (slower on very large files, but more tolerant).
135
+ - ignore_case: case-insensitive search.
136
+ - max_matches: optional cap on matches returned/printed.
137
+ - return_matches: return structured match metadata.
138
+ """
139
+ if where is None:
140
+ if file_index is None:
141
+ raise ValueError("Provide either `where` or `file_index`.")
142
+ if file_index < 0 or file_index >= len(self.inventory['files']):
143
+ raise IndexError(f"file_index {file_index} out of range")
144
+
145
+ file_path = Path(self.inventory['files'][file_index]['file_path'])
146
+ if not file_path.exists():
147
+ raise FileNotFoundError(f"File not found: {file_path}")
148
+ with open(file_path, 'r', encoding='utf-8') as f:
149
+ where = f.read()
150
+
151
+ if what is None or str(what).strip() == "":
152
+ raise ValueError("Search query cannot be empty.")
153
+
154
+ # If query was copied from repr(...) output, decode common escape sequences
155
+ # so patterns like "\\n" and escaped quotes do not block matching.
156
+ raw_query = str(what)
157
+ if any(token in raw_query for token in [r"\n", r"\t", r"\r", r"\"", r"\'"]):
158
+ try:
159
+ raw_query = bytes(raw_query, "utf-8").decode("unicode_escape")
160
+ except UnicodeDecodeError:
161
+ pass
162
+
163
+ flags = re.IGNORECASE if ignore_case else 0
164
+
165
+ if use_regex:
166
+ pattern = raw_query
167
+ else:
168
+ if html_flexible:
169
+ # Fast path: token-based whitespace-flexible matching.
170
+ # This is much faster on large HTML files.
171
+ tokens = re.findall(r"\S+", raw_query)
172
+ pattern = r"\s*".join(re.escape(token) for token in tokens)
173
+
174
+ # Optional robust path for difficult HTML snippets.
175
+ if robust_html and "<" in raw_query and ">" in raw_query:
176
+ parts = re.findall(r"<[^>]+>|[^<]+", raw_query)
177
+ part_patterns: List[str] = []
178
+
179
+ for part in parts:
180
+ if not part:
181
+ continue
182
+
183
+ if part.startswith("<") and part.endswith(">"):
184
+ inner = part[1:-1].strip()
185
+ if not inner:
186
+ continue
187
+
188
+ is_closing = inner.startswith("/")
189
+ if is_closing:
190
+ tag_name = inner[1:].strip().split()[0].strip("/").lower()
191
+ if tag_name:
192
+ part_patterns.append(rf"<\s*/\s*{re.escape(tag_name)}\s*>")
193
+ continue
194
+
195
+ tag_name = inner.split()[0].strip("/").lower()
196
+ if not tag_name:
197
+ continue
198
+
199
+ attrs = re.findall(
200
+ r"([a-zA-Z_:][-a-zA-Z0-9_:.]*)\s*=\s*([\"\'])(.*?)\2",
201
+ inner,
202
+ )
203
+ lookaheads = "".join(
204
+ rf"(?=[^>]*{re.escape(attr)}\s*=\s*[\"\']{re.escape(val)}[\"\'])"
205
+ for attr, _quote, val in attrs
206
+ )
207
+ part_patterns.append(
208
+ rf"<\s*{re.escape(tag_name)}\b{lookaheads}[^>]*>"
209
+ )
210
+ else:
211
+ text_tokens = re.findall(r"\S+", part)
212
+ if text_tokens:
213
+ part_patterns.append(
214
+ r"\s+".join(re.escape(token) for token in text_tokens)
215
+ )
216
+
217
+ if part_patterns:
218
+ pattern = r".*?".join(part_patterns)
219
+ else:
220
+ pattern = re.escape(raw_query)
221
+
222
+ search_flags = flags | (re.DOTALL if html_flexible else 0)
223
+ all_matches = list(re.finditer(pattern, where, flags=search_flags))
224
+
225
+ # Fallback for copied HTML fragments: if tag-level matching fails,
226
+ # search by the visible text contained in the fragment.
227
+ if not all_matches and html_flexible and not use_regex:
228
+ text_query = BeautifulSoup(raw_query, 'html.parser').get_text(" ", strip=True)
229
+ if text_query:
230
+ text_tokens = re.findall(r"\S+", text_query)
231
+ if text_tokens:
232
+ text_pattern = r"\s+".join(re.escape(token) for token in text_tokens)
233
+ all_matches = list(re.finditer(text_pattern, where, flags=flags | re.DOTALL))
234
+
235
+ if all_matches:
236
+ print(
237
+ "No exact HTML-fragment match was found; "
238
+ "falling back to visible-text matching."
239
+ )
240
+
241
+ total_found = len(all_matches)
242
+
243
+ if total_found == 0:
244
+ print("No matches found.")
245
+ return [] if return_matches else None
246
+
247
+ if max_matches is not None:
248
+ all_matches = all_matches[:max_matches]
249
+
250
+ matches_out: List[Dict[str, Any]] = []
251
+ for idx, match in enumerate(all_matches, start=1):
252
+ start = match.start()
253
+ end = match.end()
254
+ context_start = max(0, start - window_left)
255
+ context_end = min(len(where), end + window_right)
256
+ context = where[context_start:context_end]
257
+
258
+ print(
259
+ f"\n\n-------\n\nMatch number {idx} of {total_found} "
260
+ f"for\n\n|-> {what} <-| in {start}:{end}"
261
+ )
262
+ print(
263
+ "Context: \n\n",
264
+ repr(context),
265
+ "\n\n",
266
+ end='\n_________________\n EndOfPrinting\n_________________'
267
+ )
268
+
269
+ matches_out.append({
270
+ "match_number": idx,
271
+ "total_matches": total_found,
272
+ "start": start,
273
+ "end": end,
274
+ "matched_text": match.group(0),
275
+ "context": context,
276
+ })
277
+
278
+ return matches_out if return_matches else None
76
279
 
77
280
  def analyze_html_fonts(self, file_index: int):
78
281
  """