universal-doc-parser 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
- universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
- universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
- universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
- universal_parser/__init__.py +28 -0
- universal_parser/adaptive/__init__.py +17 -0
- universal_parser/adaptive/config_cache.py +127 -0
- universal_parser/adaptive/fingerprint.py +133 -0
- universal_parser/adaptive/tuner.py +97 -0
- universal_parser/core/__init__.py +0 -0
- universal_parser/core/engine.py +55 -0
- universal_parser/core/router.py +38 -0
- universal_parser/core/schema.py +58 -0
- universal_parser/core/sniffer.py +125 -0
- universal_parser/enrichment/__init__.py +0 -0
- universal_parser/enrichment/vlm_enricher.py +0 -0
- universal_parser/exports/__init__.py +1 -0
- universal_parser/exports/to_chunks.py +108 -0
- universal_parser/exports/to_graph.py +114 -0
- universal_parser/exports/to_markdown.py +31 -0
- universal_parser/extractors/__init__.py +0 -0
- universal_parser/extractors/base.py +39 -0
- universal_parser/extractors/images/__init__.py +0 -0
- universal_parser/extractors/images/scan_extractor.py +79 -0
- universal_parser/extractors/mail/__init__.py +0 -0
- universal_parser/extractors/mail/mail_extractor.py +206 -0
- universal_parser/extractors/office/__init__.py +0 -0
- universal_parser/extractors/office/docx_extractor.py +131 -0
- universal_parser/extractors/office/legacy_extractor.py +112 -0
- universal_parser/extractors/office/pptx_extractor.py +109 -0
- universal_parser/extractors/office/xlsx_extractor.py +93 -0
- universal_parser/extractors/pdf/__init__.py +0 -0
- universal_parser/extractors/pdf/native.py +331 -0
- universal_parser/extractors/pdf/tables.py +155 -0
- universal_parser/extractors/pdf/visual_onnx.py +0 -0
- universal_parser/extractors/structured/__init__.py +0 -0
- universal_parser/extractors/structured/csv_extractor.py +86 -0
- universal_parser/extractors/structured/json_xml_extractor.py +104 -0
- universal_parser/extractors/structured/parquet_extractor.py +59 -0
- universal_parser/extractors/web/__init__.py +1 -0
- universal_parser/extractors/web/epub_extractor.py +81 -0
- universal_parser/extractors/web/html_extractor.py +111 -0
- universal_parser/mcp/__init__.py +1 -0
- universal_parser/mcp/server.py +61 -0
- universal_parser/observability/__init__.py +12 -0
- universal_parser/observability/dashboard.py +231 -0
- universal_parser/observability/logger.py +0 -0
- universal_parser/observability/metrics.py +99 -0
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import ClassVar
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pdfplumber
|
|
9
|
+
import pypdfium2 as pdfium
|
|
10
|
+
from rapidocr_onnxruntime import RapidOCR
|
|
11
|
+
|
|
12
|
+
from universal_parser.core.router import register
|
|
13
|
+
from universal_parser.core.schema import BBox, Element
|
|
14
|
+
from universal_parser.core.sniffer import FileType
|
|
15
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
16
|
+
from universal_parser.extractors.pdf.tables import PDFTableExtractor
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@register
|
|
20
|
+
class NativePDFExtractor(BaseExtractor):
|
|
21
|
+
"""Extractor for PDF files (both native text and scanned pages).
|
|
22
|
+
|
|
23
|
+
Handles:
|
|
24
|
+
- Multi-column reading order via column-band clustering
|
|
25
|
+
- Dynamic header hierarchy using font-size percentiles
|
|
26
|
+
- Bordered & borderless table extraction integrated into the reading flow
|
|
27
|
+
- Automatic OCR fallback (via RapidOCR ONNX) for scanned pages with no text
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
supported_types: ClassVar[list[FileType]] = [FileType.PDF]
|
|
31
|
+
|
|
32
|
+
def __init__(self) -> None:
|
|
33
|
+
super().__init__()
|
|
34
|
+
self._ocr = RapidOCR()
|
|
35
|
+
|
|
36
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
37
|
+
"""Stream elements from a PDF page by page."""
|
|
38
|
+
path_str = str(path)
|
|
39
|
+
|
|
40
|
+
try:
|
|
41
|
+
plumber_doc = pdfplumber.open(path_str)
|
|
42
|
+
pdfium_doc = pdfium.PdfDocument(path_str)
|
|
43
|
+
except Exception: # noqa: BLE001
|
|
44
|
+
return
|
|
45
|
+
|
|
46
|
+
try:
|
|
47
|
+
# Collect font sizes for header hierarchy
|
|
48
|
+
font_sizes = self._collect_font_sizes(plumber_doc, max_pages=10)
|
|
49
|
+
thresholds = self._compute_header_thresholds(font_sizes)
|
|
50
|
+
|
|
51
|
+
for page_num in range(len(plumber_doc.pages)):
|
|
52
|
+
plumber_page = plumber_doc.pages[page_num]
|
|
53
|
+
page_width = float(plumber_page.width)
|
|
54
|
+
|
|
55
|
+
# A. Extract tables and their bounding boxes using pdfplumber
|
|
56
|
+
try:
|
|
57
|
+
table_extractor = PDFTableExtractor(plumber_page)
|
|
58
|
+
page_tables = table_extractor.extract_tables()
|
|
59
|
+
except Exception: # noqa: BLE001
|
|
60
|
+
page_tables = []
|
|
61
|
+
|
|
62
|
+
table_bboxes = [table["bbox"] for table in page_tables]
|
|
63
|
+
|
|
64
|
+
# B. Extract native text words/spans
|
|
65
|
+
words = plumber_page.extract_words(extra_attrs=["size"], keep_blank_chars=False)
|
|
66
|
+
spans = []
|
|
67
|
+
|
|
68
|
+
# Group adjacent words on same line into phrase spans
|
|
69
|
+
current_line: list[dict] = []
|
|
70
|
+
for word in words:
|
|
71
|
+
bbox = (
|
|
72
|
+
float(word["x0"]),
|
|
73
|
+
float(word["top"]),
|
|
74
|
+
float(word["x1"]),
|
|
75
|
+
float(word["bottom"]),
|
|
76
|
+
)
|
|
77
|
+
if self._is_inside_any_bbox(bbox, table_bboxes):
|
|
78
|
+
continue
|
|
79
|
+
|
|
80
|
+
if not current_line:
|
|
81
|
+
current_line.append(word)
|
|
82
|
+
else:
|
|
83
|
+
prev = current_line[-1]
|
|
84
|
+
# Check if on same line (vertical overlap) and close horizontal gap
|
|
85
|
+
same_line = abs(float(word["top"]) - float(prev["top"])) < 4.0
|
|
86
|
+
gap = float(word["x0"]) - float(prev["x1"])
|
|
87
|
+
if same_line and gap < 12.0:
|
|
88
|
+
current_line.append(word)
|
|
89
|
+
else:
|
|
90
|
+
span_text = " ".join(w["text"] for w in current_line).strip()
|
|
91
|
+
if span_text:
|
|
92
|
+
spans.append(
|
|
93
|
+
{
|
|
94
|
+
"is_table": False,
|
|
95
|
+
"text": span_text,
|
|
96
|
+
"size": round(float(current_line[0].get("size", 10.0)), 1),
|
|
97
|
+
"bbox": (
|
|
98
|
+
float(current_line[0]["x0"]),
|
|
99
|
+
float(min(w["top"] for w in current_line)),
|
|
100
|
+
float(current_line[-1]["x1"]),
|
|
101
|
+
float(max(w["bottom"] for w in current_line)),
|
|
102
|
+
),
|
|
103
|
+
}
|
|
104
|
+
)
|
|
105
|
+
current_line = [word]
|
|
106
|
+
|
|
107
|
+
if current_line:
|
|
108
|
+
span_text = " ".join(w["text"] for w in current_line).strip()
|
|
109
|
+
if span_text:
|
|
110
|
+
spans.append(
|
|
111
|
+
{
|
|
112
|
+
"is_table": False,
|
|
113
|
+
"text": span_text,
|
|
114
|
+
"size": round(float(current_line[0].get("size", 10.0)), 1),
|
|
115
|
+
"bbox": (
|
|
116
|
+
float(current_line[0]["x0"]),
|
|
117
|
+
float(min(w["top"] for w in current_line)),
|
|
118
|
+
float(current_line[-1]["x1"]),
|
|
119
|
+
float(max(w["bottom"] for w in current_line)),
|
|
120
|
+
),
|
|
121
|
+
}
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
# C. Wrap extracted tables
|
|
125
|
+
for table in page_tables:
|
|
126
|
+
spans.append(
|
|
127
|
+
{
|
|
128
|
+
"is_table": True,
|
|
129
|
+
"table_data": table["data"],
|
|
130
|
+
"confidence": table["confidence"],
|
|
131
|
+
"bbox": table["bbox"],
|
|
132
|
+
}
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
# D. SCANNED PAGE FALLBACK: If page has NO native text or tables, run OCR
|
|
136
|
+
if not spans:
|
|
137
|
+
if page_num < len(pdfium_doc):
|
|
138
|
+
yield from self._ocr_scanned_page(
|
|
139
|
+
pdfium_doc[page_num],
|
|
140
|
+
page_num + 1,
|
|
141
|
+
page_width,
|
|
142
|
+
float(plumber_page.height),
|
|
143
|
+
)
|
|
144
|
+
continue
|
|
145
|
+
|
|
146
|
+
# Step 3: Sort in natural reading order
|
|
147
|
+
ordered_elements = self._sort_reading_order(spans, page_width)
|
|
148
|
+
|
|
149
|
+
# Step 4: Yield sorted elements
|
|
150
|
+
for el in ordered_elements:
|
|
151
|
+
bbox_coords = el["bbox"]
|
|
152
|
+
|
|
153
|
+
if el["is_table"]:
|
|
154
|
+
headers = el["table_data"].headers
|
|
155
|
+
rows = el["table_data"].rows
|
|
156
|
+
md_header = "| " + " | ".join(headers) + " |"
|
|
157
|
+
md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
|
|
158
|
+
md_rows = ["| " + " | ".join(row) + " |" for row in rows]
|
|
159
|
+
markdown_repr = "\n".join([md_header, md_separator] + md_rows)
|
|
160
|
+
|
|
161
|
+
yield Element(
|
|
162
|
+
type="table",
|
|
163
|
+
page=page_num + 1,
|
|
164
|
+
bbox=BBox(
|
|
165
|
+
x0=bbox_coords[0],
|
|
166
|
+
y0=bbox_coords[1],
|
|
167
|
+
x1=bbox_coords[2],
|
|
168
|
+
y1=bbox_coords[3],
|
|
169
|
+
),
|
|
170
|
+
data=el["table_data"],
|
|
171
|
+
markdown_repr=markdown_repr,
|
|
172
|
+
confidence=el["confidence"],
|
|
173
|
+
)
|
|
174
|
+
else:
|
|
175
|
+
size = el["size"]
|
|
176
|
+
text = el["text"]
|
|
177
|
+
|
|
178
|
+
el_type = "paragraph"
|
|
179
|
+
level = None
|
|
180
|
+
|
|
181
|
+
if size >= thresholds["h1"]:
|
|
182
|
+
el_type = "heading"
|
|
183
|
+
level = 1
|
|
184
|
+
elif size >= thresholds["h2"]:
|
|
185
|
+
el_type = "heading"
|
|
186
|
+
level = 2
|
|
187
|
+
elif size >= thresholds["h3"]:
|
|
188
|
+
el_type = "heading"
|
|
189
|
+
level = 3
|
|
190
|
+
|
|
191
|
+
yield Element(
|
|
192
|
+
type=el_type,
|
|
193
|
+
level=level,
|
|
194
|
+
text=text,
|
|
195
|
+
page=page_num + 1,
|
|
196
|
+
bbox=BBox(
|
|
197
|
+
x0=bbox_coords[0],
|
|
198
|
+
y0=bbox_coords[1],
|
|
199
|
+
x1=bbox_coords[2],
|
|
200
|
+
y1=bbox_coords[3],
|
|
201
|
+
),
|
|
202
|
+
)
|
|
203
|
+
finally:
|
|
204
|
+
pdfium_doc.close()
|
|
205
|
+
plumber_doc.close()
|
|
206
|
+
|
|
207
|
+
def _ocr_scanned_page(
|
|
208
|
+
self, page: pdfium.PdfPage, page_num: int, page_w: float, page_h: float
|
|
209
|
+
) -> Iterator[Element]:
|
|
210
|
+
"""Render a scanned PDF page to an image and run RapidOCR."""
|
|
211
|
+
try:
|
|
212
|
+
scale = 200.0 / 72.0
|
|
213
|
+
bitmap = page.render(scale=scale)
|
|
214
|
+
pil_img = bitmap.to_pil().convert("RGB")
|
|
215
|
+
img_np = np.array(pil_img)
|
|
216
|
+
bitmap.close()
|
|
217
|
+
|
|
218
|
+
ocr_results, _ = self._ocr(img_np)
|
|
219
|
+
if not ocr_results:
|
|
220
|
+
return
|
|
221
|
+
|
|
222
|
+
scale_x = page_w / pil_img.width
|
|
223
|
+
scale_y = page_h / pil_img.height
|
|
224
|
+
|
|
225
|
+
for item in ocr_results:
|
|
226
|
+
dt_boxes, text, score = item
|
|
227
|
+
clean_text = text.strip()
|
|
228
|
+
if not clean_text:
|
|
229
|
+
continue
|
|
230
|
+
|
|
231
|
+
pts = np.array(dt_boxes, dtype=np.float32)
|
|
232
|
+
x0 = float(np.min(pts[:, 0])) * scale_x
|
|
233
|
+
y0 = float(np.min(pts[:, 1])) * scale_y
|
|
234
|
+
x1 = float(np.max(pts[:, 0])) * scale_x
|
|
235
|
+
y1 = float(np.max(pts[:, 1])) * scale_y
|
|
236
|
+
|
|
237
|
+
yield Element(
|
|
238
|
+
type="paragraph",
|
|
239
|
+
text=clean_text,
|
|
240
|
+
page=page_num,
|
|
241
|
+
bbox=BBox(x0=x0, y0=y0, x1=x1, y1=y1),
|
|
242
|
+
markdown_repr=clean_text,
|
|
243
|
+
confidence=round(float(score), 3),
|
|
244
|
+
)
|
|
245
|
+
except Exception: # noqa: BLE001
|
|
246
|
+
return
|
|
247
|
+
|
|
248
|
+
def _is_inside_any_bbox(
|
|
249
|
+
self,
|
|
250
|
+
span_bbox: tuple[float, float, float, float],
|
|
251
|
+
table_bboxes: list[tuple[float, float, float, float]],
|
|
252
|
+
) -> bool:
|
|
253
|
+
"""Check if a text span falls inside any table bounding box."""
|
|
254
|
+
sx0, sy0, sx1, sy1 = span_bbox
|
|
255
|
+
for tx0, ty0, tx1, ty1 in table_bboxes:
|
|
256
|
+
if sx0 >= (tx0 - 2) and sy0 >= (ty0 - 2) and sx1 <= (tx1 + 2) and sy1 <= (ty1 + 2):
|
|
257
|
+
return True
|
|
258
|
+
return False
|
|
259
|
+
|
|
260
|
+
def _collect_font_sizes(self, doc: pdfplumber.PDF, max_pages: int) -> list[float]:
|
|
261
|
+
"""Collect all font sizes from the start of the document."""
|
|
262
|
+
sizes = []
|
|
263
|
+
pages_to_scan = min(len(doc.pages), max_pages)
|
|
264
|
+
for i in range(pages_to_scan):
|
|
265
|
+
page = doc.pages[i]
|
|
266
|
+
words = page.extract_words(extra_attrs=["size"])
|
|
267
|
+
for w in words:
|
|
268
|
+
sizes.append(float(w.get("size", 10.0)))
|
|
269
|
+
return sizes
|
|
270
|
+
|
|
271
|
+
def _compute_header_thresholds(self, font_sizes: list[float]) -> dict[str, float]:
|
|
272
|
+
"""Compute font size cutoffs for H1, H2, H3 using percentiles."""
|
|
273
|
+
if not font_sizes:
|
|
274
|
+
return {"h1": 16.0, "h2": 14.0, "h3": 12.0}
|
|
275
|
+
|
|
276
|
+
arr = np.array(font_sizes)
|
|
277
|
+
h1_val = float(np.percentile(arr, 95))
|
|
278
|
+
h2_val = float(np.percentile(arr, 85))
|
|
279
|
+
h3_val = float(np.percentile(arr, 75))
|
|
280
|
+
|
|
281
|
+
median = float(np.median(arr))
|
|
282
|
+
|
|
283
|
+
h1_val = max(h1_val, median + 3.0)
|
|
284
|
+
h2_val = max(h2_val, median + 1.5)
|
|
285
|
+
h3_val = max(h3_val, median + 0.5)
|
|
286
|
+
|
|
287
|
+
return {"h1": h1_val, "h2": h2_val, "h3": h3_val}
|
|
288
|
+
|
|
289
|
+
def _sort_reading_order(self, elements: list[dict], page_width: float) -> list[dict]:
|
|
290
|
+
"""Sort elements to preserve natural multi-column reading order."""
|
|
291
|
+
if len(elements) < 5:
|
|
292
|
+
return sorted(elements, key=lambda e: (e["bbox"][1], e["bbox"][0]))
|
|
293
|
+
|
|
294
|
+
midpoint = page_width / 2.0
|
|
295
|
+
|
|
296
|
+
left_col = []
|
|
297
|
+
right_col = []
|
|
298
|
+
full_width = []
|
|
299
|
+
|
|
300
|
+
for el in elements:
|
|
301
|
+
x0, _, x1, _ = el["bbox"]
|
|
302
|
+
if x1 <= (midpoint + 15):
|
|
303
|
+
left_col.append(el)
|
|
304
|
+
elif x0 >= (midpoint - 15):
|
|
305
|
+
right_col.append(el)
|
|
306
|
+
else:
|
|
307
|
+
full_width.append(el)
|
|
308
|
+
|
|
309
|
+
def key_y(e):
|
|
310
|
+
return (e["bbox"][1], e["bbox"][0])
|
|
311
|
+
|
|
312
|
+
left_sorted = sorted(left_col, key=key_y)
|
|
313
|
+
right_sorted = sorted(right_col, key=key_y)
|
|
314
|
+
|
|
315
|
+
if len(left_sorted) >= 2 and len(right_sorted) >= 2:
|
|
316
|
+
col_top = min(left_sorted[0]["bbox"][1], right_sorted[0]["bbox"][1])
|
|
317
|
+
col_bottom = max(left_sorted[-1]["bbox"][3], right_sorted[-1]["bbox"][3])
|
|
318
|
+
|
|
319
|
+
top_full = [e for e in full_width if e["bbox"][3] <= col_top + 10]
|
|
320
|
+
bottom_full = [e for e in full_width if e["bbox"][1] >= col_bottom - 10]
|
|
321
|
+
mid_full = [e for e in full_width if e not in top_full and e not in bottom_full]
|
|
322
|
+
|
|
323
|
+
return (
|
|
324
|
+
sorted(top_full, key=key_y)
|
|
325
|
+
+ sorted(mid_full, key=key_y)
|
|
326
|
+
+ left_sorted
|
|
327
|
+
+ right_sorted
|
|
328
|
+
+ sorted(bottom_full, key=key_y)
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
return sorted(elements, key=key_y)
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import pdfplumber
|
|
4
|
+
|
|
5
|
+
from universal_parser.core.schema import TableData
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class PDFTableExtractor:
|
|
9
|
+
"""
|
|
10
|
+
Handles bordered (Lattice) and borderless (Stream) table extraction from PDF pages.
|
|
11
|
+
|
|
12
|
+
Provides:
|
|
13
|
+
- Bordered table extraction via coordinate grid mapping
|
|
14
|
+
- Borderless table extraction via whitespace column clustering
|
|
15
|
+
- Confidence scoring based on layout density and cell consistency
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
def __init__(self, page: pdfplumber.page.Page):
|
|
19
|
+
self.page = page
|
|
20
|
+
|
|
21
|
+
def extract_tables(self) -> list[dict]:
|
|
22
|
+
"""
|
|
23
|
+
Extract all tables from the page.
|
|
24
|
+
|
|
25
|
+
Returns:
|
|
26
|
+
list[dict]: A list of tables found, each formatted as:
|
|
27
|
+
{
|
|
28
|
+
"data": TableData,
|
|
29
|
+
"bbox": tuple(x0, y0, x1, y1),
|
|
30
|
+
"confidence": float (0.0 to 1.0)
|
|
31
|
+
}
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
tables = []
|
|
35
|
+
|
|
36
|
+
# 1. Try Lattice extraction first (bordered tables)
|
|
37
|
+
lattice_tables = self._extract_lattice()
|
|
38
|
+
if lattice_tables:
|
|
39
|
+
tables.extend(lattice_tables)
|
|
40
|
+
|
|
41
|
+
# 2. Try Stream extraction (borderless tables)
|
|
42
|
+
# We only run stream extraction if we don't find lattice tables
|
|
43
|
+
# to prevent duplicate extractions on the same area.
|
|
44
|
+
if not lattice_tables:
|
|
45
|
+
stream_tables = self._extract_stream()
|
|
46
|
+
if stream_tables:
|
|
47
|
+
tables.extend(stream_tables)
|
|
48
|
+
|
|
49
|
+
return tables
|
|
50
|
+
|
|
51
|
+
def _extract_lattice(self) -> list[dict]:
|
|
52
|
+
"""Extract tables using vertical and horizontal vector lines."""
|
|
53
|
+
extracted = []
|
|
54
|
+
# pdfplumber vertical and horizontal line settings
|
|
55
|
+
table_settings = {
|
|
56
|
+
"vertical_strategy": "lines",
|
|
57
|
+
"horizontal_strategy": "lines",
|
|
58
|
+
"snap_tolerance": 3,
|
|
59
|
+
"join_tolerance": 3,
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
plumber_tables = self.page.find_tables(table_settings=table_settings)
|
|
63
|
+
for table in plumber_tables:
|
|
64
|
+
raw_data = table.extract()
|
|
65
|
+
if not raw_data or len(raw_data) < 2:
|
|
66
|
+
continue
|
|
67
|
+
|
|
68
|
+
clean_rows = []
|
|
69
|
+
for row in raw_data:
|
|
70
|
+
# Convert None to empty string
|
|
71
|
+
clean_row = [str(cell).strip() if cell is not None else "" for cell in row]
|
|
72
|
+
clean_rows.append(clean_row)
|
|
73
|
+
|
|
74
|
+
headers = clean_rows[0]
|
|
75
|
+
data_rows = clean_rows[1:]
|
|
76
|
+
|
|
77
|
+
# Lattice table confidence is high (0.95+) since lines physically define cells
|
|
78
|
+
confidence = 0.98 if all(len(row) == len(headers) for row in data_rows) else 0.90
|
|
79
|
+
|
|
80
|
+
extracted.append(
|
|
81
|
+
{
|
|
82
|
+
"data": TableData(headers=headers, rows=data_rows),
|
|
83
|
+
"bbox": table.bbox, # (x0, y0, x1, y1)
|
|
84
|
+
"confidence": confidence,
|
|
85
|
+
}
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
return extracted
|
|
89
|
+
|
|
90
|
+
def _extract_stream(self) -> list[dict]:
|
|
91
|
+
"""Extract borderless tables using whitespace distance clustering and text alignment."""
|
|
92
|
+
extracted = []
|
|
93
|
+
table_settings = {
|
|
94
|
+
"vertical_strategy": "text",
|
|
95
|
+
"horizontal_strategy": "text",
|
|
96
|
+
"snap_tolerance": 4,
|
|
97
|
+
"join_tolerance": 4,
|
|
98
|
+
"min_words_vertical": 3,
|
|
99
|
+
"min_words_horizontal": 2,
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
try:
|
|
103
|
+
plumber_tables = self.page.find_tables(table_settings=table_settings)
|
|
104
|
+
for table in plumber_tables:
|
|
105
|
+
raw_data = table.extract()
|
|
106
|
+
if not raw_data or len(raw_data) < 2:
|
|
107
|
+
continue
|
|
108
|
+
|
|
109
|
+
clean_rows = []
|
|
110
|
+
total_words = 0
|
|
111
|
+
non_empty_cells = 0
|
|
112
|
+
|
|
113
|
+
for row in raw_data:
|
|
114
|
+
clean_row = [str(cell).strip() if cell is not None else "" for cell in row]
|
|
115
|
+
if any(clean_row):
|
|
116
|
+
clean_rows.append(clean_row)
|
|
117
|
+
for cell in clean_row:
|
|
118
|
+
if cell:
|
|
119
|
+
word_count = len(cell.split())
|
|
120
|
+
total_words += word_count
|
|
121
|
+
non_empty_cells += 1
|
|
122
|
+
|
|
123
|
+
if len(clean_rows) < 2 or non_empty_cells < 4:
|
|
124
|
+
continue
|
|
125
|
+
|
|
126
|
+
# Heuristic: Genuine tables contain concise cell values (labels, numbers, codes <= 2.5 words/cell).
|
|
127
|
+
# If cells contain full paragraph sentences, it is multi-column text, not a table.
|
|
128
|
+
avg_words_per_cell = total_words / max(1, non_empty_cells)
|
|
129
|
+
if avg_words_per_cell > 2.5:
|
|
130
|
+
continue
|
|
131
|
+
|
|
132
|
+
# Filter out tables that swallow more than 40% of page height (multi-column text pages)
|
|
133
|
+
page_h = float(self.page.height)
|
|
134
|
+
table_h = float(table.bbox[3] - table.bbox[1])
|
|
135
|
+
if table_h > (0.40 * page_h):
|
|
136
|
+
continue
|
|
137
|
+
|
|
138
|
+
max_cols = max(len(r) for r in clean_rows)
|
|
139
|
+
if max_cols < 2:
|
|
140
|
+
continue
|
|
141
|
+
|
|
142
|
+
headers = clean_rows[0]
|
|
143
|
+
data_rows = clean_rows[1:]
|
|
144
|
+
|
|
145
|
+
extracted.append(
|
|
146
|
+
{
|
|
147
|
+
"data": TableData(headers=headers, rows=data_rows),
|
|
148
|
+
"bbox": table.bbox, # (x0, y0, x1, y1)
|
|
149
|
+
"confidence": 0.85,
|
|
150
|
+
}
|
|
151
|
+
)
|
|
152
|
+
except Exception: # noqa: BLE001
|
|
153
|
+
return []
|
|
154
|
+
|
|
155
|
+
return extracted
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
from collections.abc import Iterator
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import ClassVar
|
|
7
|
+
|
|
8
|
+
from universal_parser.core.router import register
|
|
9
|
+
from universal_parser.core.schema import Element, TableData
|
|
10
|
+
from universal_parser.core.sniffer import FileType
|
|
11
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@register
|
|
15
|
+
class CSVExtractor(BaseExtractor):
|
|
16
|
+
"""
|
|
17
|
+
Extractor for delimited text files (.csv, .tsv).
|
|
18
|
+
Handles:
|
|
19
|
+
- Auto-dialect detection (delimiter, quotechar) via csv.Sniffer
|
|
20
|
+
- UTF-8, Latin-1, and Windows-1252 encodings
|
|
21
|
+
- Streams rows into structured TableData
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
supported_types: ClassVar[list[FileType]] = [FileType.CSV, FileType.TSV]
|
|
25
|
+
|
|
26
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
27
|
+
"""Stream table element from a CSV/TSV file."""
|
|
28
|
+
path_obj = Path(path)
|
|
29
|
+
path_str = str(path)
|
|
30
|
+
|
|
31
|
+
# Step 1: Detect encoding safely
|
|
32
|
+
encoding = self._detect_encoding(path_str)
|
|
33
|
+
|
|
34
|
+
try:
|
|
35
|
+
with open(path_str, encoding=encoding, errors="replace") as f:
|
|
36
|
+
sample = f.read(4096)
|
|
37
|
+
f.seek(0)
|
|
38
|
+
|
|
39
|
+
if not sample.strip():
|
|
40
|
+
return
|
|
41
|
+
|
|
42
|
+
# Step 2: Auto-detect delimiter using Sniffer, fallback to extension
|
|
43
|
+
try:
|
|
44
|
+
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|")
|
|
45
|
+
delimiter = dialect.delimiter
|
|
46
|
+
except Exception: # noqa: BLE001
|
|
47
|
+
delimiter = "\t" if path_obj.suffix.lower() == ".tsv" else ","
|
|
48
|
+
|
|
49
|
+
reader = csv.reader(f, delimiter=delimiter)
|
|
50
|
+
rows = []
|
|
51
|
+
for row in reader:
|
|
52
|
+
clean_row = [cell.strip() for cell in row]
|
|
53
|
+
if any(clean_row):
|
|
54
|
+
rows.append(clean_row)
|
|
55
|
+
|
|
56
|
+
if not rows:
|
|
57
|
+
return
|
|
58
|
+
|
|
59
|
+
headers = rows[0]
|
|
60
|
+
data_rows = rows[1:]
|
|
61
|
+
|
|
62
|
+
# Generate markdown representation
|
|
63
|
+
md_header = "| " + " | ".join(headers) + " |"
|
|
64
|
+
md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
|
|
65
|
+
md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
|
|
66
|
+
markdown_repr = "\n".join([md_header, md_separator] + md_rows)
|
|
67
|
+
|
|
68
|
+
yield Element(
|
|
69
|
+
type="table",
|
|
70
|
+
text=path_obj.name,
|
|
71
|
+
data=TableData(headers=headers, rows=data_rows),
|
|
72
|
+
markdown_repr=markdown_repr,
|
|
73
|
+
confidence=1.0,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
except Exception: # noqa: BLE001
|
|
77
|
+
return
|
|
78
|
+
|
|
79
|
+
def _detect_encoding(self, path: str) -> str:
|
|
80
|
+
"""Try decoding a small chunk with UTF-8, fallback to latin-1."""
|
|
81
|
+
try:
|
|
82
|
+
with open(path, "rb") as f:
|
|
83
|
+
f.read(2048).decode("utf-8")
|
|
84
|
+
return "utf-8"
|
|
85
|
+
except UnicodeDecodeError:
|
|
86
|
+
return "latin-1"
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import xml.etree.ElementTree as ET
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import ClassVar
|
|
8
|
+
|
|
9
|
+
from universal_parser.core.router import register
|
|
10
|
+
from universal_parser.core.schema import Element, TableData
|
|
11
|
+
from universal_parser.core.sniffer import FileType
|
|
12
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@register
|
|
16
|
+
class JSONXMLExtractor(BaseExtractor):
|
|
17
|
+
"""
|
|
18
|
+
Extractor for JSON (.json) and XML (.xml) files.
|
|
19
|
+
|
|
20
|
+
Handles:
|
|
21
|
+
- List of objects in JSON -> Table element (if tabular schema)
|
|
22
|
+
- Nested JSON objects -> formatted code block element
|
|
23
|
+
- XML -> element nodes parsed into headings, paragraphs, and code blocks
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
supported_types: ClassVar[list[FileType]] = [FileType.JSON, FileType.XML]
|
|
27
|
+
|
|
28
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
29
|
+
path_obj = Path(path)
|
|
30
|
+
ext = path_obj.suffix.lower()
|
|
31
|
+
|
|
32
|
+
if ext == ".json":
|
|
33
|
+
yield from self._stream_json(path_obj)
|
|
34
|
+
elif ext == ".xml":
|
|
35
|
+
yield from self._stream_xml(path_obj)
|
|
36
|
+
|
|
37
|
+
def _stream_json(self, path: Path) -> Iterator[Element]:
|
|
38
|
+
try:
|
|
39
|
+
with open(path, encoding="utf-8", errors="replace") as f:
|
|
40
|
+
data = json.load(f)
|
|
41
|
+
|
|
42
|
+
# Check if it's a list of uniform dictionaries (tabular JSON)
|
|
43
|
+
if isinstance(data, list) and data and all(isinstance(row, dict) for row in data):
|
|
44
|
+
headers = list(data[0].keys())
|
|
45
|
+
rows = [[str(row.get(h, "")).strip() for h in headers] for row in data]
|
|
46
|
+
|
|
47
|
+
md_header = "| " + " | ".join(headers) + " |"
|
|
48
|
+
md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
|
|
49
|
+
md_rows = ["| " + " | ".join(r) + " |" for r in rows]
|
|
50
|
+
markdown_repr = "\n".join([md_header, md_separator] + md_rows)
|
|
51
|
+
|
|
52
|
+
yield Element(
|
|
53
|
+
type="table",
|
|
54
|
+
text=path.name,
|
|
55
|
+
data=TableData(headers=headers, rows=rows),
|
|
56
|
+
markdown_repr=markdown_repr,
|
|
57
|
+
confidence=1.0,
|
|
58
|
+
)
|
|
59
|
+
else:
|
|
60
|
+
# General JSON: render formatted code block
|
|
61
|
+
formatted_json = json.dumps(data, indent=2)
|
|
62
|
+
yield Element(
|
|
63
|
+
type="code_block",
|
|
64
|
+
text=formatted_json,
|
|
65
|
+
markdown_repr=f"```json\n{formatted_json}\n```",
|
|
66
|
+
confidence=1.0,
|
|
67
|
+
)
|
|
68
|
+
except Exception: # noqa: BLE001
|
|
69
|
+
return
|
|
70
|
+
|
|
71
|
+
def _stream_xml(self, path: Path) -> Iterator[Element]:
|
|
72
|
+
try:
|
|
73
|
+
tree = ET.parse(path)
|
|
74
|
+
root = tree.getroot()
|
|
75
|
+
|
|
76
|
+
# Yield root tag as level 1 heading
|
|
77
|
+
yield Element(
|
|
78
|
+
type="heading",
|
|
79
|
+
level=1,
|
|
80
|
+
text=root.tag,
|
|
81
|
+
markdown_repr=f"# {root.tag}",
|
|
82
|
+
confidence=1.0,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
for child in root:
|
|
86
|
+
text_val = (child.text or "").strip()
|
|
87
|
+
if text_val:
|
|
88
|
+
yield Element(
|
|
89
|
+
type="paragraph",
|
|
90
|
+
text=f"{child.tag}: {text_val}",
|
|
91
|
+
markdown_repr=f"**{child.tag}**: {text_val}",
|
|
92
|
+
confidence=1.0,
|
|
93
|
+
)
|
|
94
|
+
else:
|
|
95
|
+
child_str = ET.tostring(child, encoding="unicode").strip()
|
|
96
|
+
if child_str:
|
|
97
|
+
yield Element(
|
|
98
|
+
type="code_block",
|
|
99
|
+
text=child_str,
|
|
100
|
+
markdown_repr=f"```xml\n{child_str}\n```",
|
|
101
|
+
confidence=1.0,
|
|
102
|
+
)
|
|
103
|
+
except Exception: # noqa: BLE001
|
|
104
|
+
return
|