universal-doc-parser 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
  2. universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
  3. universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
  4. universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
  5. universal_parser/__init__.py +28 -0
  6. universal_parser/adaptive/__init__.py +17 -0
  7. universal_parser/adaptive/config_cache.py +127 -0
  8. universal_parser/adaptive/fingerprint.py +133 -0
  9. universal_parser/adaptive/tuner.py +97 -0
  10. universal_parser/core/__init__.py +0 -0
  11. universal_parser/core/engine.py +55 -0
  12. universal_parser/core/router.py +38 -0
  13. universal_parser/core/schema.py +58 -0
  14. universal_parser/core/sniffer.py +125 -0
  15. universal_parser/enrichment/__init__.py +0 -0
  16. universal_parser/enrichment/vlm_enricher.py +0 -0
  17. universal_parser/exports/__init__.py +1 -0
  18. universal_parser/exports/to_chunks.py +108 -0
  19. universal_parser/exports/to_graph.py +114 -0
  20. universal_parser/exports/to_markdown.py +31 -0
  21. universal_parser/extractors/__init__.py +0 -0
  22. universal_parser/extractors/base.py +39 -0
  23. universal_parser/extractors/images/__init__.py +0 -0
  24. universal_parser/extractors/images/scan_extractor.py +79 -0
  25. universal_parser/extractors/mail/__init__.py +0 -0
  26. universal_parser/extractors/mail/mail_extractor.py +206 -0
  27. universal_parser/extractors/office/__init__.py +0 -0
  28. universal_parser/extractors/office/docx_extractor.py +131 -0
  29. universal_parser/extractors/office/legacy_extractor.py +112 -0
  30. universal_parser/extractors/office/pptx_extractor.py +109 -0
  31. universal_parser/extractors/office/xlsx_extractor.py +93 -0
  32. universal_parser/extractors/pdf/__init__.py +0 -0
  33. universal_parser/extractors/pdf/native.py +331 -0
  34. universal_parser/extractors/pdf/tables.py +155 -0
  35. universal_parser/extractors/pdf/visual_onnx.py +0 -0
  36. universal_parser/extractors/structured/__init__.py +0 -0
  37. universal_parser/extractors/structured/csv_extractor.py +86 -0
  38. universal_parser/extractors/structured/json_xml_extractor.py +104 -0
  39. universal_parser/extractors/structured/parquet_extractor.py +59 -0
  40. universal_parser/extractors/web/__init__.py +1 -0
  41. universal_parser/extractors/web/epub_extractor.py +81 -0
  42. universal_parser/extractors/web/html_extractor.py +111 -0
  43. universal_parser/mcp/__init__.py +1 -0
  44. universal_parser/mcp/server.py +61 -0
  45. universal_parser/observability/__init__.py +12 -0
  46. universal_parser/observability/dashboard.py +231 -0
  47. universal_parser/observability/logger.py +0 -0
  48. universal_parser/observability/metrics.py +99 -0
@@ -0,0 +1,331 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterator
4
+ from pathlib import Path
5
+ from typing import ClassVar
6
+
7
+ import numpy as np
8
+ import pdfplumber
9
+ import pypdfium2 as pdfium
10
+ from rapidocr_onnxruntime import RapidOCR
11
+
12
+ from universal_parser.core.router import register
13
+ from universal_parser.core.schema import BBox, Element
14
+ from universal_parser.core.sniffer import FileType
15
+ from universal_parser.extractors.base import BaseExtractor
16
+ from universal_parser.extractors.pdf.tables import PDFTableExtractor
17
+
18
+
19
+ @register
20
+ class NativePDFExtractor(BaseExtractor):
21
+ """Extractor for PDF files (both native text and scanned pages).
22
+
23
+ Handles:
24
+ - Multi-column reading order via column-band clustering
25
+ - Dynamic header hierarchy using font-size percentiles
26
+ - Bordered & borderless table extraction integrated into the reading flow
27
+ - Automatic OCR fallback (via RapidOCR ONNX) for scanned pages with no text
28
+ """
29
+
30
+ supported_types: ClassVar[list[FileType]] = [FileType.PDF]
31
+
32
+ def __init__(self) -> None:
33
+ super().__init__()
34
+ self._ocr = RapidOCR()
35
+
36
+ def stream(self, path: str | Path) -> Iterator[Element]:
37
+ """Stream elements from a PDF page by page."""
38
+ path_str = str(path)
39
+
40
+ try:
41
+ plumber_doc = pdfplumber.open(path_str)
42
+ pdfium_doc = pdfium.PdfDocument(path_str)
43
+ except Exception: # noqa: BLE001
44
+ return
45
+
46
+ try:
47
+ # Collect font sizes for header hierarchy
48
+ font_sizes = self._collect_font_sizes(plumber_doc, max_pages=10)
49
+ thresholds = self._compute_header_thresholds(font_sizes)
50
+
51
+ for page_num in range(len(plumber_doc.pages)):
52
+ plumber_page = plumber_doc.pages[page_num]
53
+ page_width = float(plumber_page.width)
54
+
55
+ # A. Extract tables and their bounding boxes using pdfplumber
56
+ try:
57
+ table_extractor = PDFTableExtractor(plumber_page)
58
+ page_tables = table_extractor.extract_tables()
59
+ except Exception: # noqa: BLE001
60
+ page_tables = []
61
+
62
+ table_bboxes = [table["bbox"] for table in page_tables]
63
+
64
+ # B. Extract native text words/spans
65
+ words = plumber_page.extract_words(extra_attrs=["size"], keep_blank_chars=False)
66
+ spans = []
67
+
68
+ # Group adjacent words on same line into phrase spans
69
+ current_line: list[dict] = []
70
+ for word in words:
71
+ bbox = (
72
+ float(word["x0"]),
73
+ float(word["top"]),
74
+ float(word["x1"]),
75
+ float(word["bottom"]),
76
+ )
77
+ if self._is_inside_any_bbox(bbox, table_bboxes):
78
+ continue
79
+
80
+ if not current_line:
81
+ current_line.append(word)
82
+ else:
83
+ prev = current_line[-1]
84
+ # Check if on same line (vertical overlap) and close horizontal gap
85
+ same_line = abs(float(word["top"]) - float(prev["top"])) < 4.0
86
+ gap = float(word["x0"]) - float(prev["x1"])
87
+ if same_line and gap < 12.0:
88
+ current_line.append(word)
89
+ else:
90
+ span_text = " ".join(w["text"] for w in current_line).strip()
91
+ if span_text:
92
+ spans.append(
93
+ {
94
+ "is_table": False,
95
+ "text": span_text,
96
+ "size": round(float(current_line[0].get("size", 10.0)), 1),
97
+ "bbox": (
98
+ float(current_line[0]["x0"]),
99
+ float(min(w["top"] for w in current_line)),
100
+ float(current_line[-1]["x1"]),
101
+ float(max(w["bottom"] for w in current_line)),
102
+ ),
103
+ }
104
+ )
105
+ current_line = [word]
106
+
107
+ if current_line:
108
+ span_text = " ".join(w["text"] for w in current_line).strip()
109
+ if span_text:
110
+ spans.append(
111
+ {
112
+ "is_table": False,
113
+ "text": span_text,
114
+ "size": round(float(current_line[0].get("size", 10.0)), 1),
115
+ "bbox": (
116
+ float(current_line[0]["x0"]),
117
+ float(min(w["top"] for w in current_line)),
118
+ float(current_line[-1]["x1"]),
119
+ float(max(w["bottom"] for w in current_line)),
120
+ ),
121
+ }
122
+ )
123
+
124
+ # C. Wrap extracted tables
125
+ for table in page_tables:
126
+ spans.append(
127
+ {
128
+ "is_table": True,
129
+ "table_data": table["data"],
130
+ "confidence": table["confidence"],
131
+ "bbox": table["bbox"],
132
+ }
133
+ )
134
+
135
+ # D. SCANNED PAGE FALLBACK: If page has NO native text or tables, run OCR
136
+ if not spans:
137
+ if page_num < len(pdfium_doc):
138
+ yield from self._ocr_scanned_page(
139
+ pdfium_doc[page_num],
140
+ page_num + 1,
141
+ page_width,
142
+ float(plumber_page.height),
143
+ )
144
+ continue
145
+
146
+ # Step 3: Sort in natural reading order
147
+ ordered_elements = self._sort_reading_order(spans, page_width)
148
+
149
+ # Step 4: Yield sorted elements
150
+ for el in ordered_elements:
151
+ bbox_coords = el["bbox"]
152
+
153
+ if el["is_table"]:
154
+ headers = el["table_data"].headers
155
+ rows = el["table_data"].rows
156
+ md_header = "| " + " | ".join(headers) + " |"
157
+ md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
158
+ md_rows = ["| " + " | ".join(row) + " |" for row in rows]
159
+ markdown_repr = "\n".join([md_header, md_separator] + md_rows)
160
+
161
+ yield Element(
162
+ type="table",
163
+ page=page_num + 1,
164
+ bbox=BBox(
165
+ x0=bbox_coords[0],
166
+ y0=bbox_coords[1],
167
+ x1=bbox_coords[2],
168
+ y1=bbox_coords[3],
169
+ ),
170
+ data=el["table_data"],
171
+ markdown_repr=markdown_repr,
172
+ confidence=el["confidence"],
173
+ )
174
+ else:
175
+ size = el["size"]
176
+ text = el["text"]
177
+
178
+ el_type = "paragraph"
179
+ level = None
180
+
181
+ if size >= thresholds["h1"]:
182
+ el_type = "heading"
183
+ level = 1
184
+ elif size >= thresholds["h2"]:
185
+ el_type = "heading"
186
+ level = 2
187
+ elif size >= thresholds["h3"]:
188
+ el_type = "heading"
189
+ level = 3
190
+
191
+ yield Element(
192
+ type=el_type,
193
+ level=level,
194
+ text=text,
195
+ page=page_num + 1,
196
+ bbox=BBox(
197
+ x0=bbox_coords[0],
198
+ y0=bbox_coords[1],
199
+ x1=bbox_coords[2],
200
+ y1=bbox_coords[3],
201
+ ),
202
+ )
203
+ finally:
204
+ pdfium_doc.close()
205
+ plumber_doc.close()
206
+
207
+ def _ocr_scanned_page(
208
+ self, page: pdfium.PdfPage, page_num: int, page_w: float, page_h: float
209
+ ) -> Iterator[Element]:
210
+ """Render a scanned PDF page to an image and run RapidOCR."""
211
+ try:
212
+ scale = 200.0 / 72.0
213
+ bitmap = page.render(scale=scale)
214
+ pil_img = bitmap.to_pil().convert("RGB")
215
+ img_np = np.array(pil_img)
216
+ bitmap.close()
217
+
218
+ ocr_results, _ = self._ocr(img_np)
219
+ if not ocr_results:
220
+ return
221
+
222
+ scale_x = page_w / pil_img.width
223
+ scale_y = page_h / pil_img.height
224
+
225
+ for item in ocr_results:
226
+ dt_boxes, text, score = item
227
+ clean_text = text.strip()
228
+ if not clean_text:
229
+ continue
230
+
231
+ pts = np.array(dt_boxes, dtype=np.float32)
232
+ x0 = float(np.min(pts[:, 0])) * scale_x
233
+ y0 = float(np.min(pts[:, 1])) * scale_y
234
+ x1 = float(np.max(pts[:, 0])) * scale_x
235
+ y1 = float(np.max(pts[:, 1])) * scale_y
236
+
237
+ yield Element(
238
+ type="paragraph",
239
+ text=clean_text,
240
+ page=page_num,
241
+ bbox=BBox(x0=x0, y0=y0, x1=x1, y1=y1),
242
+ markdown_repr=clean_text,
243
+ confidence=round(float(score), 3),
244
+ )
245
+ except Exception: # noqa: BLE001
246
+ return
247
+
248
+ def _is_inside_any_bbox(
249
+ self,
250
+ span_bbox: tuple[float, float, float, float],
251
+ table_bboxes: list[tuple[float, float, float, float]],
252
+ ) -> bool:
253
+ """Check if a text span falls inside any table bounding box."""
254
+ sx0, sy0, sx1, sy1 = span_bbox
255
+ for tx0, ty0, tx1, ty1 in table_bboxes:
256
+ if sx0 >= (tx0 - 2) and sy0 >= (ty0 - 2) and sx1 <= (tx1 + 2) and sy1 <= (ty1 + 2):
257
+ return True
258
+ return False
259
+
260
+ def _collect_font_sizes(self, doc: pdfplumber.PDF, max_pages: int) -> list[float]:
261
+ """Collect all font sizes from the start of the document."""
262
+ sizes = []
263
+ pages_to_scan = min(len(doc.pages), max_pages)
264
+ for i in range(pages_to_scan):
265
+ page = doc.pages[i]
266
+ words = page.extract_words(extra_attrs=["size"])
267
+ for w in words:
268
+ sizes.append(float(w.get("size", 10.0)))
269
+ return sizes
270
+
271
+ def _compute_header_thresholds(self, font_sizes: list[float]) -> dict[str, float]:
272
+ """Compute font size cutoffs for H1, H2, H3 using percentiles."""
273
+ if not font_sizes:
274
+ return {"h1": 16.0, "h2": 14.0, "h3": 12.0}
275
+
276
+ arr = np.array(font_sizes)
277
+ h1_val = float(np.percentile(arr, 95))
278
+ h2_val = float(np.percentile(arr, 85))
279
+ h3_val = float(np.percentile(arr, 75))
280
+
281
+ median = float(np.median(arr))
282
+
283
+ h1_val = max(h1_val, median + 3.0)
284
+ h2_val = max(h2_val, median + 1.5)
285
+ h3_val = max(h3_val, median + 0.5)
286
+
287
+ return {"h1": h1_val, "h2": h2_val, "h3": h3_val}
288
+
289
+ def _sort_reading_order(self, elements: list[dict], page_width: float) -> list[dict]:
290
+ """Sort elements to preserve natural multi-column reading order."""
291
+ if len(elements) < 5:
292
+ return sorted(elements, key=lambda e: (e["bbox"][1], e["bbox"][0]))
293
+
294
+ midpoint = page_width / 2.0
295
+
296
+ left_col = []
297
+ right_col = []
298
+ full_width = []
299
+
300
+ for el in elements:
301
+ x0, _, x1, _ = el["bbox"]
302
+ if x1 <= (midpoint + 15):
303
+ left_col.append(el)
304
+ elif x0 >= (midpoint - 15):
305
+ right_col.append(el)
306
+ else:
307
+ full_width.append(el)
308
+
309
+ def key_y(e):
310
+ return (e["bbox"][1], e["bbox"][0])
311
+
312
+ left_sorted = sorted(left_col, key=key_y)
313
+ right_sorted = sorted(right_col, key=key_y)
314
+
315
+ if len(left_sorted) >= 2 and len(right_sorted) >= 2:
316
+ col_top = min(left_sorted[0]["bbox"][1], right_sorted[0]["bbox"][1])
317
+ col_bottom = max(left_sorted[-1]["bbox"][3], right_sorted[-1]["bbox"][3])
318
+
319
+ top_full = [e for e in full_width if e["bbox"][3] <= col_top + 10]
320
+ bottom_full = [e for e in full_width if e["bbox"][1] >= col_bottom - 10]
321
+ mid_full = [e for e in full_width if e not in top_full and e not in bottom_full]
322
+
323
+ return (
324
+ sorted(top_full, key=key_y)
325
+ + sorted(mid_full, key=key_y)
326
+ + left_sorted
327
+ + right_sorted
328
+ + sorted(bottom_full, key=key_y)
329
+ )
330
+
331
+ return sorted(elements, key=key_y)
@@ -0,0 +1,155 @@
1
+ from __future__ import annotations
2
+
3
+ import pdfplumber
4
+
5
+ from universal_parser.core.schema import TableData
6
+
7
+
8
+ class PDFTableExtractor:
9
+ """
10
+ Handles bordered (Lattice) and borderless (Stream) table extraction from PDF pages.
11
+
12
+ Provides:
13
+ - Bordered table extraction via coordinate grid mapping
14
+ - Borderless table extraction via whitespace column clustering
15
+ - Confidence scoring based on layout density and cell consistency
16
+ """
17
+
18
+ def __init__(self, page: pdfplumber.page.Page):
19
+ self.page = page
20
+
21
+ def extract_tables(self) -> list[dict]:
22
+ """
23
+ Extract all tables from the page.
24
+
25
+ Returns:
26
+ list[dict]: A list of tables found, each formatted as:
27
+ {
28
+ "data": TableData,
29
+ "bbox": tuple(x0, y0, x1, y1),
30
+ "confidence": float (0.0 to 1.0)
31
+ }
32
+ """
33
+
34
+ tables = []
35
+
36
+ # 1. Try Lattice extraction first (bordered tables)
37
+ lattice_tables = self._extract_lattice()
38
+ if lattice_tables:
39
+ tables.extend(lattice_tables)
40
+
41
+ # 2. Try Stream extraction (borderless tables)
42
+ # We only run stream extraction if we don't find lattice tables
43
+ # to prevent duplicate extractions on the same area.
44
+ if not lattice_tables:
45
+ stream_tables = self._extract_stream()
46
+ if stream_tables:
47
+ tables.extend(stream_tables)
48
+
49
+ return tables
50
+
51
+ def _extract_lattice(self) -> list[dict]:
52
+ """Extract tables using vertical and horizontal vector lines."""
53
+ extracted = []
54
+ # pdfplumber vertical and horizontal line settings
55
+ table_settings = {
56
+ "vertical_strategy": "lines",
57
+ "horizontal_strategy": "lines",
58
+ "snap_tolerance": 3,
59
+ "join_tolerance": 3,
60
+ }
61
+
62
+ plumber_tables = self.page.find_tables(table_settings=table_settings)
63
+ for table in plumber_tables:
64
+ raw_data = table.extract()
65
+ if not raw_data or len(raw_data) < 2:
66
+ continue
67
+
68
+ clean_rows = []
69
+ for row in raw_data:
70
+ # Convert None to empty string
71
+ clean_row = [str(cell).strip() if cell is not None else "" for cell in row]
72
+ clean_rows.append(clean_row)
73
+
74
+ headers = clean_rows[0]
75
+ data_rows = clean_rows[1:]
76
+
77
+ # Lattice table confidence is high (0.95+) since lines physically define cells
78
+ confidence = 0.98 if all(len(row) == len(headers) for row in data_rows) else 0.90
79
+
80
+ extracted.append(
81
+ {
82
+ "data": TableData(headers=headers, rows=data_rows),
83
+ "bbox": table.bbox, # (x0, y0, x1, y1)
84
+ "confidence": confidence,
85
+ }
86
+ )
87
+
88
+ return extracted
89
+
90
+ def _extract_stream(self) -> list[dict]:
91
+ """Extract borderless tables using whitespace distance clustering and text alignment."""
92
+ extracted = []
93
+ table_settings = {
94
+ "vertical_strategy": "text",
95
+ "horizontal_strategy": "text",
96
+ "snap_tolerance": 4,
97
+ "join_tolerance": 4,
98
+ "min_words_vertical": 3,
99
+ "min_words_horizontal": 2,
100
+ }
101
+
102
+ try:
103
+ plumber_tables = self.page.find_tables(table_settings=table_settings)
104
+ for table in plumber_tables:
105
+ raw_data = table.extract()
106
+ if not raw_data or len(raw_data) < 2:
107
+ continue
108
+
109
+ clean_rows = []
110
+ total_words = 0
111
+ non_empty_cells = 0
112
+
113
+ for row in raw_data:
114
+ clean_row = [str(cell).strip() if cell is not None else "" for cell in row]
115
+ if any(clean_row):
116
+ clean_rows.append(clean_row)
117
+ for cell in clean_row:
118
+ if cell:
119
+ word_count = len(cell.split())
120
+ total_words += word_count
121
+ non_empty_cells += 1
122
+
123
+ if len(clean_rows) < 2 or non_empty_cells < 4:
124
+ continue
125
+
126
+ # Heuristic: Genuine tables contain concise cell values (labels, numbers, codes <= 2.5 words/cell).
127
+ # If cells contain full paragraph sentences, it is multi-column text, not a table.
128
+ avg_words_per_cell = total_words / max(1, non_empty_cells)
129
+ if avg_words_per_cell > 2.5:
130
+ continue
131
+
132
+ # Filter out tables that swallow more than 40% of page height (multi-column text pages)
133
+ page_h = float(self.page.height)
134
+ table_h = float(table.bbox[3] - table.bbox[1])
135
+ if table_h > (0.40 * page_h):
136
+ continue
137
+
138
+ max_cols = max(len(r) for r in clean_rows)
139
+ if max_cols < 2:
140
+ continue
141
+
142
+ headers = clean_rows[0]
143
+ data_rows = clean_rows[1:]
144
+
145
+ extracted.append(
146
+ {
147
+ "data": TableData(headers=headers, rows=data_rows),
148
+ "bbox": table.bbox, # (x0, y0, x1, y1)
149
+ "confidence": 0.85,
150
+ }
151
+ )
152
+ except Exception: # noqa: BLE001
153
+ return []
154
+
155
+ return extracted
File without changes
File without changes
@@ -0,0 +1,86 @@
1
+ from __future__ import annotations
2
+
3
+ import csv
4
+ from collections.abc import Iterator
5
+ from pathlib import Path
6
+ from typing import ClassVar
7
+
8
+ from universal_parser.core.router import register
9
+ from universal_parser.core.schema import Element, TableData
10
+ from universal_parser.core.sniffer import FileType
11
+ from universal_parser.extractors.base import BaseExtractor
12
+
13
+
14
+ @register
15
+ class CSVExtractor(BaseExtractor):
16
+ """
17
+ Extractor for delimited text files (.csv, .tsv).
18
+ Handles:
19
+ - Auto-dialect detection (delimiter, quotechar) via csv.Sniffer
20
+ - UTF-8, Latin-1, and Windows-1252 encodings
21
+ - Streams rows into structured TableData
22
+ """
23
+
24
+ supported_types: ClassVar[list[FileType]] = [FileType.CSV, FileType.TSV]
25
+
26
+ def stream(self, path: str | Path) -> Iterator[Element]:
27
+ """Stream table element from a CSV/TSV file."""
28
+ path_obj = Path(path)
29
+ path_str = str(path)
30
+
31
+ # Step 1: Detect encoding safely
32
+ encoding = self._detect_encoding(path_str)
33
+
34
+ try:
35
+ with open(path_str, encoding=encoding, errors="replace") as f:
36
+ sample = f.read(4096)
37
+ f.seek(0)
38
+
39
+ if not sample.strip():
40
+ return
41
+
42
+ # Step 2: Auto-detect delimiter using Sniffer, fallback to extension
43
+ try:
44
+ dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|")
45
+ delimiter = dialect.delimiter
46
+ except Exception: # noqa: BLE001
47
+ delimiter = "\t" if path_obj.suffix.lower() == ".tsv" else ","
48
+
49
+ reader = csv.reader(f, delimiter=delimiter)
50
+ rows = []
51
+ for row in reader:
52
+ clean_row = [cell.strip() for cell in row]
53
+ if any(clean_row):
54
+ rows.append(clean_row)
55
+
56
+ if not rows:
57
+ return
58
+
59
+ headers = rows[0]
60
+ data_rows = rows[1:]
61
+
62
+ # Generate markdown representation
63
+ md_header = "| " + " | ".join(headers) + " |"
64
+ md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
65
+ md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
66
+ markdown_repr = "\n".join([md_header, md_separator] + md_rows)
67
+
68
+ yield Element(
69
+ type="table",
70
+ text=path_obj.name,
71
+ data=TableData(headers=headers, rows=data_rows),
72
+ markdown_repr=markdown_repr,
73
+ confidence=1.0,
74
+ )
75
+
76
+ except Exception: # noqa: BLE001
77
+ return
78
+
79
+ def _detect_encoding(self, path: str) -> str:
80
+ """Try decoding a small chunk with UTF-8, fallback to latin-1."""
81
+ try:
82
+ with open(path, "rb") as f:
83
+ f.read(2048).decode("utf-8")
84
+ return "utf-8"
85
+ except UnicodeDecodeError:
86
+ return "latin-1"
@@ -0,0 +1,104 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import xml.etree.ElementTree as ET
5
+ from collections.abc import Iterator
6
+ from pathlib import Path
7
+ from typing import ClassVar
8
+
9
+ from universal_parser.core.router import register
10
+ from universal_parser.core.schema import Element, TableData
11
+ from universal_parser.core.sniffer import FileType
12
+ from universal_parser.extractors.base import BaseExtractor
13
+
14
+
15
+ @register
16
+ class JSONXMLExtractor(BaseExtractor):
17
+ """
18
+ Extractor for JSON (.json) and XML (.xml) files.
19
+
20
+ Handles:
21
+ - List of objects in JSON -> Table element (if tabular schema)
22
+ - Nested JSON objects -> formatted code block element
23
+ - XML -> element nodes parsed into headings, paragraphs, and code blocks
24
+ """
25
+
26
+ supported_types: ClassVar[list[FileType]] = [FileType.JSON, FileType.XML]
27
+
28
+ def stream(self, path: str | Path) -> Iterator[Element]:
29
+ path_obj = Path(path)
30
+ ext = path_obj.suffix.lower()
31
+
32
+ if ext == ".json":
33
+ yield from self._stream_json(path_obj)
34
+ elif ext == ".xml":
35
+ yield from self._stream_xml(path_obj)
36
+
37
+ def _stream_json(self, path: Path) -> Iterator[Element]:
38
+ try:
39
+ with open(path, encoding="utf-8", errors="replace") as f:
40
+ data = json.load(f)
41
+
42
+ # Check if it's a list of uniform dictionaries (tabular JSON)
43
+ if isinstance(data, list) and data and all(isinstance(row, dict) for row in data):
44
+ headers = list(data[0].keys())
45
+ rows = [[str(row.get(h, "")).strip() for h in headers] for row in data]
46
+
47
+ md_header = "| " + " | ".join(headers) + " |"
48
+ md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
49
+ md_rows = ["| " + " | ".join(r) + " |" for r in rows]
50
+ markdown_repr = "\n".join([md_header, md_separator] + md_rows)
51
+
52
+ yield Element(
53
+ type="table",
54
+ text=path.name,
55
+ data=TableData(headers=headers, rows=rows),
56
+ markdown_repr=markdown_repr,
57
+ confidence=1.0,
58
+ )
59
+ else:
60
+ # General JSON: render formatted code block
61
+ formatted_json = json.dumps(data, indent=2)
62
+ yield Element(
63
+ type="code_block",
64
+ text=formatted_json,
65
+ markdown_repr=f"```json\n{formatted_json}\n```",
66
+ confidence=1.0,
67
+ )
68
+ except Exception: # noqa: BLE001
69
+ return
70
+
71
+ def _stream_xml(self, path: Path) -> Iterator[Element]:
72
+ try:
73
+ tree = ET.parse(path)
74
+ root = tree.getroot()
75
+
76
+ # Yield root tag as level 1 heading
77
+ yield Element(
78
+ type="heading",
79
+ level=1,
80
+ text=root.tag,
81
+ markdown_repr=f"# {root.tag}",
82
+ confidence=1.0,
83
+ )
84
+
85
+ for child in root:
86
+ text_val = (child.text or "").strip()
87
+ if text_val:
88
+ yield Element(
89
+ type="paragraph",
90
+ text=f"{child.tag}: {text_val}",
91
+ markdown_repr=f"**{child.tag}**: {text_val}",
92
+ confidence=1.0,
93
+ )
94
+ else:
95
+ child_str = ET.tostring(child, encoding="unicode").strip()
96
+ if child_str:
97
+ yield Element(
98
+ type="code_block",
99
+ text=child_str,
100
+ markdown_repr=f"```xml\n{child_str}\n```",
101
+ confidence=1.0,
102
+ )
103
+ except Exception: # noqa: BLE001
104
+ return