text2rel 0.1__tar.gz → 0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {text2rel-0.1 → text2rel-0.2}/PKG-INFO +2 -2
- {text2rel-0.1 → text2rel-0.2}/pyproject.toml +2 -2
- text2rel-0.2/text2rel/__init__.py +12 -0
- {text2rel-0.1 → text2rel-0.2}/text2rel/cleaner.py +209 -6
- text2rel-0.2/text2rel/reliability_assessor.py +3049 -0
- text2rel-0.2/text2rel/state.py +18 -0
- text2rel-0.2/text2rel/utils.py +175 -0
- text2rel-0.1/text2rel/__init__.py +0 -3
- text2rel-0.1/text2rel/reliability_assessor.py +0 -364
- text2rel-0.1/text2rel/utils.py +0 -39
- {text2rel-0.1 → text2rel-0.2}/readme.md +0 -0
- {text2rel-0.1 → text2rel-0.2}/text2rel/io.py +0 -0
- {text2rel-0.1 → text2rel-0.2}/text2rel/llm_processor.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: text2rel
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: An HTML cleaner customized for kinship ties extraction from large text corpra.
|
|
3
|
+
Version: 0.2
|
|
4
|
+
Summary: An HTML cleaner, LLM processor and Reliability Assessment customized for kinship ties extraction from large text corpra.
|
|
5
5
|
Requires-Python: >=3.10,<3.13
|
|
6
6
|
Classifier: Programming Language :: Python :: 3
|
|
7
7
|
Classifier: Programming Language :: Python :: 3.10
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "text2rel"
|
|
3
|
-
version = "0.
|
|
4
|
-
description = "An HTML cleaner customized for kinship ties extraction from large text corpra."
|
|
3
|
+
version = "0.2"
|
|
4
|
+
description = "An HTML cleaner, LLM processor and Reliability Assessment customized for kinship ties extraction from large text corpra."
|
|
5
5
|
|
|
6
6
|
readme = "readme.md"
|
|
7
7
|
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from .cleaner import HTMLCleaner
|
|
2
|
+
from .llm_processor import LLMProcessor
|
|
3
|
+
from .reliability_assessor import ReliabilityAssessment
|
|
4
|
+
from .state import set_default_inventory_path, get_default_inventory_path
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"HTMLCleaner",
|
|
8
|
+
"LLMProcessor",
|
|
9
|
+
"ReliabilityAssessment",
|
|
10
|
+
"set_default_inventory_path",
|
|
11
|
+
"get_default_inventory_path",
|
|
12
|
+
]
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
import os
|
|
3
3
|
import sys
|
|
4
4
|
import json
|
|
5
|
+
import webbrowser
|
|
5
6
|
import numpy as np
|
|
6
7
|
import pandas as pd
|
|
7
8
|
import re
|
|
@@ -11,26 +12,32 @@ import string
|
|
|
11
12
|
from pathlib import Path
|
|
12
13
|
from text2rel.utils import insert_jumpjump
|
|
13
14
|
from text2rel.io import _load_inventory, _save_inventory
|
|
14
|
-
from
|
|
15
|
+
from text2rel.state import set_default_inventory_path
|
|
16
|
+
from typing import Dict, TYPE_CHECKING, Optional, List, Any
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from text2rel.llm_processor import LLMProcessor
|
|
15
20
|
|
|
16
21
|
|
|
17
22
|
|
|
18
23
|
class HTMLCleaner:
|
|
19
|
-
def __init__(self,
|
|
24
|
+
def __init__(self, json_name: str, html_directory: str = None, extension: str = ".htm"):
|
|
20
25
|
"""
|
|
21
26
|
Initializes the cleaner. If the JSON doesn't exist but a directory is given, it builds the inventory.
|
|
22
27
|
"""
|
|
23
|
-
if not
|
|
24
|
-
raise ValueError("
|
|
28
|
+
if not json_name:
|
|
29
|
+
raise ValueError("json_name parameter is required. Please provide a name for your JSON inventory file. It can be simply an empty JSON file")
|
|
25
30
|
if not html_directory:
|
|
26
31
|
raise ValueError("html_directory parameter is required. Please provide a path to your HTML files directory.")
|
|
27
|
-
|
|
32
|
+
|
|
33
|
+
self.json_path = Path(json_name)
|
|
34
|
+
set_default_inventory_path(self.json_path)
|
|
28
35
|
|
|
29
36
|
if not self.json_path.exists():
|
|
30
37
|
if html_directory is None:
|
|
31
38
|
raise FileNotFoundError(f"JSON not found at {json_path} and no directory given to create one.")
|
|
32
39
|
self._create_inventory(html_directory, extension)
|
|
33
|
-
print(f"Created new inventory at {
|
|
40
|
+
print(f"Created new inventory at {json_name}")
|
|
34
41
|
|
|
35
42
|
self.inventory = _load_inventory(self.json_path)
|
|
36
43
|
|
|
@@ -64,6 +71,29 @@ class HTMLCleaner:
|
|
|
64
71
|
for idx, file_info in enumerate(self.inventory['files']):
|
|
65
72
|
print(f"{idx}: {Path(file_info['file_path']).name}")
|
|
66
73
|
|
|
74
|
+
def show_file(self, file_index: int, open_in_browser: bool = True) -> str:
|
|
75
|
+
"""
|
|
76
|
+
Open an indexed HTML file from the inventory.
|
|
77
|
+
|
|
78
|
+
Parameters:
|
|
79
|
+
- file_index (int): index in self.inventory['files']
|
|
80
|
+
- open_in_browser (bool): if True, opens the file in the default browser
|
|
81
|
+
|
|
82
|
+
Returns:
|
|
83
|
+
- str: absolute path of the selected file
|
|
84
|
+
"""
|
|
85
|
+
if file_index < 0 or file_index >= len(self.inventory['files']):
|
|
86
|
+
raise IndexError(f"file_index {file_index} out of range")
|
|
87
|
+
|
|
88
|
+
file_path = Path(self.inventory['files'][file_index]['file_path']).resolve()
|
|
89
|
+
if not file_path.exists():
|
|
90
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
91
|
+
|
|
92
|
+
if open_in_browser:
|
|
93
|
+
webbrowser.open(file_path.as_uri())
|
|
94
|
+
|
|
95
|
+
return str(file_path)
|
|
96
|
+
|
|
67
97
|
#Allowing the user to set the relevant beginning and end for each file, the files can be accessed by filename
|
|
68
98
|
def set_relevant_bounds(self, filename: str, beginning, end):
|
|
69
99
|
for file in self.inventory['files']:
|
|
@@ -73,6 +103,179 @@ class HTMLCleaner:
|
|
|
73
103
|
_save_inventory(self.json_path,self.inventory)
|
|
74
104
|
return
|
|
75
105
|
raise ValueError(f"File {filename} not found in inventory.")
|
|
106
|
+
|
|
107
|
+
def find_str(
|
|
108
|
+
self,
|
|
109
|
+
what: str,
|
|
110
|
+
where: Optional[str] = None,
|
|
111
|
+
window_left: int = 120,
|
|
112
|
+
window_right: int = 120,
|
|
113
|
+
*,
|
|
114
|
+
file_index: Optional[int] = None,
|
|
115
|
+
use_regex: bool = False,
|
|
116
|
+
html_flexible: bool = False,
|
|
117
|
+
robust_html: bool = False,
|
|
118
|
+
ignore_case: bool = True,
|
|
119
|
+
max_matches: Optional[int] = None,
|
|
120
|
+
return_matches: bool = False,
|
|
121
|
+
) -> Optional[List[Dict[str, Any]]]:
|
|
122
|
+
"""
|
|
123
|
+
Search text (or an inventory HTML file) and print contextual matches.
|
|
124
|
+
|
|
125
|
+
Parameters:
|
|
126
|
+
- what: query string or regex pattern.
|
|
127
|
+
- where: raw text to search. If None, file_index is used to load a file.
|
|
128
|
+
- window_left/window_right: chars shown before/after each match.
|
|
129
|
+
- file_index: inventory index used when where is None.
|
|
130
|
+
- use_regex: interpret `what` as regex pattern.
|
|
131
|
+
- html_flexible: for literal search, allow whitespace/newline drift.
|
|
132
|
+
This uses a fast token-based pattern by default.
|
|
133
|
+
- robust_html: if True with html_flexible, use a heavier HTML-aware
|
|
134
|
+
matcher (slower on very large files, but more tolerant).
|
|
135
|
+
- ignore_case: case-insensitive search.
|
|
136
|
+
- max_matches: optional cap on matches returned/printed.
|
|
137
|
+
- return_matches: return structured match metadata.
|
|
138
|
+
"""
|
|
139
|
+
if where is None:
|
|
140
|
+
if file_index is None:
|
|
141
|
+
raise ValueError("Provide either `where` or `file_index`.")
|
|
142
|
+
if file_index < 0 or file_index >= len(self.inventory['files']):
|
|
143
|
+
raise IndexError(f"file_index {file_index} out of range")
|
|
144
|
+
|
|
145
|
+
file_path = Path(self.inventory['files'][file_index]['file_path'])
|
|
146
|
+
if not file_path.exists():
|
|
147
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
148
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
149
|
+
where = f.read()
|
|
150
|
+
|
|
151
|
+
if what is None or str(what).strip() == "":
|
|
152
|
+
raise ValueError("Search query cannot be empty.")
|
|
153
|
+
|
|
154
|
+
# If query was copied from repr(...) output, decode common escape sequences
|
|
155
|
+
# so patterns like "\\n" and escaped quotes do not block matching.
|
|
156
|
+
raw_query = str(what)
|
|
157
|
+
if any(token in raw_query for token in [r"\n", r"\t", r"\r", r"\"", r"\'"]):
|
|
158
|
+
try:
|
|
159
|
+
raw_query = bytes(raw_query, "utf-8").decode("unicode_escape")
|
|
160
|
+
except UnicodeDecodeError:
|
|
161
|
+
pass
|
|
162
|
+
|
|
163
|
+
flags = re.IGNORECASE if ignore_case else 0
|
|
164
|
+
|
|
165
|
+
if use_regex:
|
|
166
|
+
pattern = raw_query
|
|
167
|
+
else:
|
|
168
|
+
if html_flexible:
|
|
169
|
+
# Fast path: token-based whitespace-flexible matching.
|
|
170
|
+
# This is much faster on large HTML files.
|
|
171
|
+
tokens = re.findall(r"\S+", raw_query)
|
|
172
|
+
pattern = r"\s*".join(re.escape(token) for token in tokens)
|
|
173
|
+
|
|
174
|
+
# Optional robust path for difficult HTML snippets.
|
|
175
|
+
if robust_html and "<" in raw_query and ">" in raw_query:
|
|
176
|
+
parts = re.findall(r"<[^>]+>|[^<]+", raw_query)
|
|
177
|
+
part_patterns: List[str] = []
|
|
178
|
+
|
|
179
|
+
for part in parts:
|
|
180
|
+
if not part:
|
|
181
|
+
continue
|
|
182
|
+
|
|
183
|
+
if part.startswith("<") and part.endswith(">"):
|
|
184
|
+
inner = part[1:-1].strip()
|
|
185
|
+
if not inner:
|
|
186
|
+
continue
|
|
187
|
+
|
|
188
|
+
is_closing = inner.startswith("/")
|
|
189
|
+
if is_closing:
|
|
190
|
+
tag_name = inner[1:].strip().split()[0].strip("/").lower()
|
|
191
|
+
if tag_name:
|
|
192
|
+
part_patterns.append(rf"<\s*/\s*{re.escape(tag_name)}\s*>")
|
|
193
|
+
continue
|
|
194
|
+
|
|
195
|
+
tag_name = inner.split()[0].strip("/").lower()
|
|
196
|
+
if not tag_name:
|
|
197
|
+
continue
|
|
198
|
+
|
|
199
|
+
attrs = re.findall(
|
|
200
|
+
r"([a-zA-Z_:][-a-zA-Z0-9_:.]*)\s*=\s*([\"\'])(.*?)\2",
|
|
201
|
+
inner,
|
|
202
|
+
)
|
|
203
|
+
lookaheads = "".join(
|
|
204
|
+
rf"(?=[^>]*{re.escape(attr)}\s*=\s*[\"\']{re.escape(val)}[\"\'])"
|
|
205
|
+
for attr, _quote, val in attrs
|
|
206
|
+
)
|
|
207
|
+
part_patterns.append(
|
|
208
|
+
rf"<\s*{re.escape(tag_name)}\b{lookaheads}[^>]*>"
|
|
209
|
+
)
|
|
210
|
+
else:
|
|
211
|
+
text_tokens = re.findall(r"\S+", part)
|
|
212
|
+
if text_tokens:
|
|
213
|
+
part_patterns.append(
|
|
214
|
+
r"\s+".join(re.escape(token) for token in text_tokens)
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
if part_patterns:
|
|
218
|
+
pattern = r".*?".join(part_patterns)
|
|
219
|
+
else:
|
|
220
|
+
pattern = re.escape(raw_query)
|
|
221
|
+
|
|
222
|
+
search_flags = flags | (re.DOTALL if html_flexible else 0)
|
|
223
|
+
all_matches = list(re.finditer(pattern, where, flags=search_flags))
|
|
224
|
+
|
|
225
|
+
# Fallback for copied HTML fragments: if tag-level matching fails,
|
|
226
|
+
# search by the visible text contained in the fragment.
|
|
227
|
+
if not all_matches and html_flexible and not use_regex:
|
|
228
|
+
text_query = BeautifulSoup(raw_query, 'html.parser').get_text(" ", strip=True)
|
|
229
|
+
if text_query:
|
|
230
|
+
text_tokens = re.findall(r"\S+", text_query)
|
|
231
|
+
if text_tokens:
|
|
232
|
+
text_pattern = r"\s+".join(re.escape(token) for token in text_tokens)
|
|
233
|
+
all_matches = list(re.finditer(text_pattern, where, flags=flags | re.DOTALL))
|
|
234
|
+
|
|
235
|
+
if all_matches:
|
|
236
|
+
print(
|
|
237
|
+
"No exact HTML-fragment match was found; "
|
|
238
|
+
"falling back to visible-text matching."
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
total_found = len(all_matches)
|
|
242
|
+
|
|
243
|
+
if total_found == 0:
|
|
244
|
+
print("No matches found.")
|
|
245
|
+
return [] if return_matches else None
|
|
246
|
+
|
|
247
|
+
if max_matches is not None:
|
|
248
|
+
all_matches = all_matches[:max_matches]
|
|
249
|
+
|
|
250
|
+
matches_out: List[Dict[str, Any]] = []
|
|
251
|
+
for idx, match in enumerate(all_matches, start=1):
|
|
252
|
+
start = match.start()
|
|
253
|
+
end = match.end()
|
|
254
|
+
context_start = max(0, start - window_left)
|
|
255
|
+
context_end = min(len(where), end + window_right)
|
|
256
|
+
context = where[context_start:context_end]
|
|
257
|
+
|
|
258
|
+
print(
|
|
259
|
+
f"\n\n-------\n\nMatch number {idx} of {total_found} "
|
|
260
|
+
f"for\n\n|-> {what} <-| in {start}:{end}"
|
|
261
|
+
)
|
|
262
|
+
print(
|
|
263
|
+
"Context: \n\n",
|
|
264
|
+
repr(context),
|
|
265
|
+
"\n\n",
|
|
266
|
+
end='\n_________________\n EndOfPrinting\n_________________'
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
matches_out.append({
|
|
270
|
+
"match_number": idx,
|
|
271
|
+
"total_matches": total_found,
|
|
272
|
+
"start": start,
|
|
273
|
+
"end": end,
|
|
274
|
+
"matched_text": match.group(0),
|
|
275
|
+
"context": context,
|
|
276
|
+
})
|
|
277
|
+
|
|
278
|
+
return matches_out if return_matches else None
|
|
76
279
|
|
|
77
280
|
def analyze_html_fonts(self, file_index: int):
|
|
78
281
|
"""
|