universal-doc-parser 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
  2. universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
  3. universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
  4. universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
  5. universal_parser/__init__.py +28 -0
  6. universal_parser/adaptive/__init__.py +17 -0
  7. universal_parser/adaptive/config_cache.py +127 -0
  8. universal_parser/adaptive/fingerprint.py +133 -0
  9. universal_parser/adaptive/tuner.py +97 -0
  10. universal_parser/core/__init__.py +0 -0
  11. universal_parser/core/engine.py +55 -0
  12. universal_parser/core/router.py +38 -0
  13. universal_parser/core/schema.py +58 -0
  14. universal_parser/core/sniffer.py +125 -0
  15. universal_parser/enrichment/__init__.py +0 -0
  16. universal_parser/enrichment/vlm_enricher.py +0 -0
  17. universal_parser/exports/__init__.py +1 -0
  18. universal_parser/exports/to_chunks.py +108 -0
  19. universal_parser/exports/to_graph.py +114 -0
  20. universal_parser/exports/to_markdown.py +31 -0
  21. universal_parser/extractors/__init__.py +0 -0
  22. universal_parser/extractors/base.py +39 -0
  23. universal_parser/extractors/images/__init__.py +0 -0
  24. universal_parser/extractors/images/scan_extractor.py +79 -0
  25. universal_parser/extractors/mail/__init__.py +0 -0
  26. universal_parser/extractors/mail/mail_extractor.py +206 -0
  27. universal_parser/extractors/office/__init__.py +0 -0
  28. universal_parser/extractors/office/docx_extractor.py +131 -0
  29. universal_parser/extractors/office/legacy_extractor.py +112 -0
  30. universal_parser/extractors/office/pptx_extractor.py +109 -0
  31. universal_parser/extractors/office/xlsx_extractor.py +93 -0
  32. universal_parser/extractors/pdf/__init__.py +0 -0
  33. universal_parser/extractors/pdf/native.py +331 -0
  34. universal_parser/extractors/pdf/tables.py +155 -0
  35. universal_parser/extractors/pdf/visual_onnx.py +0 -0
  36. universal_parser/extractors/structured/__init__.py +0 -0
  37. universal_parser/extractors/structured/csv_extractor.py +86 -0
  38. universal_parser/extractors/structured/json_xml_extractor.py +104 -0
  39. universal_parser/extractors/structured/parquet_extractor.py +59 -0
  40. universal_parser/extractors/web/__init__.py +1 -0
  41. universal_parser/extractors/web/epub_extractor.py +81 -0
  42. universal_parser/extractors/web/html_extractor.py +111 -0
  43. universal_parser/mcp/__init__.py +1 -0
  44. universal_parser/mcp/server.py +61 -0
  45. universal_parser/observability/__init__.py +12 -0
  46. universal_parser/observability/dashboard.py +231 -0
  47. universal_parser/observability/logger.py +0 -0
  48. universal_parser/observability/metrics.py +99 -0
@@ -0,0 +1,206 @@
1
+ from __future__ import annotations
2
+
3
+ import email
4
+ import mailbox
5
+ import tempfile
6
+ from collections.abc import Iterator
7
+ from email import policy
8
+ from pathlib import Path
9
+ from typing import ClassVar
10
+
11
+ import extract_msg
12
+ from selectolax.parser import HTMLParser
13
+
14
+ from universal_parser.core.router import get_extractor, register
15
+ from universal_parser.core.schema import Element
16
+ from universal_parser.core.sniffer import FileType, sniff
17
+ from universal_parser.extractors.base import BaseExtractor
18
+
19
+
20
+ @register
21
+ class MailExtractor(BaseExtractor):
22
+ """
23
+ Extractor for email formats (.eml, .msg, .mbox).
24
+
25
+ Handles:
26
+ - Header metadata (Subject, From, To, Date)
27
+ - Body extraction (HTML / Plaintext via selectolax)
28
+ - Recursive attachment parsing through the sniffer/router
29
+ """
30
+
31
+ supported_types: ClassVar[list[FileType]] = [
32
+ FileType.EML,
33
+ FileType.MSG,
34
+ FileType.MBOX,
35
+ ]
36
+
37
+ def stream(self, path: str | Path) -> Iterator[Element]:
38
+ path_obj = Path(path)
39
+ ext = path_obj.suffix.lower()
40
+
41
+ if ext == ".eml":
42
+ yield from self._stream_eml(path_obj)
43
+ elif ext == ".msg":
44
+ yield from self._stream_msg(path_obj)
45
+ elif ext == ".mbox":
46
+ yield from self._stream_mbox(path_obj)
47
+
48
+ def _stream_eml(self, path: Path) -> Iterator[Element]:
49
+ try:
50
+ with open(path, "rb") as f:
51
+ msg = email.message_from_binary_file(f, policy=policy.default)
52
+
53
+ yield from self._process_email_message(msg)
54
+ except Exception: # noqa: BLE001
55
+ return
56
+
57
+ def _stream_msg(self, path: Path) -> Iterator[Element]:
58
+ try:
59
+ msg = extract_msg.Message(str(path))
60
+ subject = msg.subject or "No Subject"
61
+ sender = msg.sender or "Unknown Sender"
62
+ to = msg.to or "Unknown Recipient"
63
+ date = str(msg.date) if msg.date else ""
64
+
65
+ yield Element(
66
+ type="heading",
67
+ level=1,
68
+ text=f"Subject: {subject}",
69
+ markdown_repr=f"# Subject: {subject}",
70
+ confidence=1.0,
71
+ )
72
+
73
+ meta_text = f"From: {sender} | To: {to}" + (f" | Date: {date}" if date else "")
74
+ yield Element(
75
+ type="paragraph",
76
+ text=meta_text,
77
+ markdown_repr=f"**{meta_text}**",
78
+ confidence=1.0,
79
+ )
80
+
81
+ body_text = msg.body
82
+ if body_text:
83
+ for paragraph in body_text.split("\n\n"):
84
+ clean_p = paragraph.strip()
85
+ if clean_p:
86
+ yield Element(
87
+ type="paragraph",
88
+ text=clean_p,
89
+ markdown_repr=clean_p,
90
+ confidence=1.0,
91
+ )
92
+
93
+ # Process attachments recursively
94
+ for att in msg.attachments:
95
+ att_data = att.data
96
+ att_name = att.longFilename or att.shortFilename or "attachment"
97
+ if att_data:
98
+ yield from self._process_attachment_bytes(att_name, att_data)
99
+
100
+ msg.close()
101
+ except Exception: # noqa: BLE001
102
+ return
103
+
104
+ def _stream_mbox(self, path: Path) -> Iterator[Element]:
105
+ try:
106
+ mbox = mailbox.mbox(str(path))
107
+ for msg in mbox.values():
108
+ yield from self._process_email_message(msg)
109
+ except Exception: # noqa: BLE001
110
+ return
111
+
112
+ def _process_email_message(self, msg: email.message.EmailMessage) -> Iterator[Element]:
113
+ subject = str(msg.get("Subject", "No Subject"))
114
+ sender = str(msg.get("From", "Unknown Sender"))
115
+ to = str(msg.get("To", "Unknown Recipient"))
116
+ date = str(msg.get("Date", ""))
117
+
118
+ yield Element(
119
+ type="heading",
120
+ level=1,
121
+ text=f"Subject: {subject}",
122
+ markdown_repr=f"# Subject: {subject}",
123
+ confidence=1.0,
124
+ )
125
+
126
+ meta_text = f"From: {sender} | To: {to}" + (f" | Date: {date}" if date else "")
127
+ yield Element(
128
+ type="paragraph",
129
+ text=meta_text,
130
+ markdown_repr=f"**{meta_text}**",
131
+ confidence=1.0,
132
+ )
133
+
134
+ # Body extraction
135
+ body_part = msg.get_body(preferencelist=("html", "plain"))
136
+ if body_part:
137
+ content = body_part.get_content()
138
+ if body_part.get_content_type() == "text/html":
139
+ parser = HTMLParser(content)
140
+ for tag in parser.css("script, style, noscript"):
141
+ tag.decompose()
142
+ body_elem = parser.body or parser.root
143
+ if body_elem:
144
+ for node in body_elem.iter():
145
+ tag_name = node.tag.lower() if node.tag else ""
146
+ if tag_name in ("h1", "h2", "h3", "h4", "h5", "h6"):
147
+ h_text = node.text(strip=True)
148
+ if h_text:
149
+ yield Element(
150
+ type="heading",
151
+ level=int(tag_name[1]),
152
+ text=h_text,
153
+ markdown_repr=f"{'#' * int(tag_name[1])} {h_text}",
154
+ confidence=1.0,
155
+ )
156
+ elif tag_name in ("p", "li"):
157
+ p_text = node.text(strip=True)
158
+ if p_text:
159
+ yield Element(
160
+ type="paragraph",
161
+ text=p_text,
162
+ markdown_repr=p_text,
163
+ confidence=1.0,
164
+ )
165
+ else:
166
+ for paragraph in str(content).split("\n\n"):
167
+ clean_p = paragraph.strip()
168
+ if clean_p:
169
+ yield Element(
170
+ type="paragraph",
171
+ text=clean_p,
172
+ markdown_repr=clean_p,
173
+ confidence=1.0,
174
+ )
175
+
176
+ # Attachment extraction (recursive)
177
+ for part in msg.iter_attachments():
178
+ filename = part.get_filename() or "attachment"
179
+ payload = part.get_payload(decode=True)
180
+ if payload:
181
+ yield from self._process_attachment_bytes(filename, payload)
182
+
183
+ def _process_attachment_bytes(self, filename: str, data: bytes) -> Iterator[Element]:
184
+ """Save attachment to temp file, sniff type, and route to sub-extractor."""
185
+ try:
186
+ suffix = Path(filename).suffix
187
+ with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as tmp:
188
+ tmp.write(data)
189
+ tmp_path = Path(tmp.name)
190
+
191
+ file_type = sniff(tmp_path)
192
+ extractor = get_extractor(file_type)
193
+
194
+ if extractor:
195
+ yield Element(
196
+ type="heading",
197
+ level=2,
198
+ text=f"Attachment: {filename}",
199
+ markdown_repr=f"## Attachment: {filename}",
200
+ confidence=1.0,
201
+ )
202
+ yield from extractor.stream(tmp_path)
203
+
204
+ tmp_path.unlink(missing_ok=True)
205
+ except Exception: # noqa: BLE001
206
+ return
File without changes
@@ -0,0 +1,131 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterator
4
+ from pathlib import Path
5
+ from typing import ClassVar
6
+
7
+ from docx import Document as DocxDocument
8
+ from docx.oxml.ns import qn
9
+ from docx.table import Table
10
+ from docx.text.paragraph import Paragraph
11
+
12
+ from universal_parser.core.router import register
13
+ from universal_parser.core.schema import Element, TableData
14
+ from universal_parser.core.sniffer import FileType
15
+ from universal_parser.extractors.base import BaseExtractor
16
+
17
+ # Map Word built-in heading style names to heading levels
18
+ _HEADING_STYLE_MAP: dict[str, int] = {
19
+ "heading 1": 1,
20
+ "heading 2": 2,
21
+ "heading 3": 3,
22
+ "heading 4": 4,
23
+ "heading 5": 5,
24
+ "heading 6": 6,
25
+ "title": 1, # Word's "Title" style treated as H1
26
+ "subtitle": 2, # Word's "Subtitle" style treated as H2
27
+ }
28
+
29
+
30
+ @register
31
+ class DocxExtractor(BaseExtractor):
32
+ """
33
+ Extractor for .docx files using python-docx.
34
+ Handles:
35
+ - Paragraphs and headings in correct document order
36
+ - Tables (headers inferred from first row)
37
+ - Preserves reading order by iterating document body directly
38
+
39
+ Does NOT handle:
40
+ - Legacy .doc binary format (needs olefile — Phase 6)
41
+ - Embedded images/charts (Phase 6)
42
+ """
43
+
44
+ supported_types: ClassVar[list[FileType]] = [FileType.DOCX]
45
+
46
+ def stream(self, path: str | Path) -> Iterator[Element]:
47
+ """Stream elements from a DOCX file in document reading order."""
48
+ try:
49
+ doc = DocxDocument(str(path))
50
+ except Exception: # noqa: BLE001
51
+ # Corrupted or password-protected DOCX — exit cleanly
52
+ return
53
+
54
+ try:
55
+ # Iterate body children directly to preserve paragraph + table order.
56
+ # doc.paragraphs and doc.tables are separate lists — using them
57
+ # would give all paragraphs first then all tables, losing real order.
58
+ for child in doc.element.body:
59
+ tag = child.tag
60
+
61
+ # Paragraph element
62
+ if tag == qn("w:p"):
63
+ element = self._process_paragraph(Paragraph(child, doc))
64
+ if element is not None:
65
+ yield element
66
+
67
+ # Table element
68
+ elif tag == qn("w:tbl"):
69
+ element = self._process_table(Table(child, doc))
70
+ if element is not None:
71
+ yield element
72
+
73
+ except Exception: # noqa: BLE001
74
+ # If anything fails mid-document, we stop cleanly.
75
+ # We do not crash — whatever was yielded before the error is valid.
76
+ return
77
+
78
+ def _process_paragraph(self, para: Paragraph) -> Element | None:
79
+ """Convert a python-docx Paragraph into an Element."""
80
+ text = para.text.strip()
81
+
82
+ # Skip empty paragraphs — they carry no information for RAG
83
+ if not text:
84
+ return None
85
+
86
+ style_name = para.style.name.lower() if para.style else ""
87
+ heading_level = _HEADING_STYLE_MAP.get(style_name)
88
+
89
+ if heading_level is not None:
90
+ return Element(
91
+ type="heading",
92
+ level=heading_level,
93
+ text=text,
94
+ )
95
+
96
+ # Check for list items (Word list styles contain "list" in the name)
97
+ if "list" in style_name:
98
+ return Element(
99
+ type="list_item",
100
+ text=text,
101
+ )
102
+ return Element(
103
+ type="paragraph",
104
+ text=text,
105
+ )
106
+
107
+ def _process_table(self, table: Table) -> Element | None:
108
+ """Convert a python-docx Table into an Element with TableData."""
109
+ rows = []
110
+ for row in table.rows:
111
+ row_data = [cell.text.strip() for cell in row.cells]
112
+ rows.append(row_data)
113
+
114
+ if not rows:
115
+ return None
116
+
117
+ # Treat the first row as headers
118
+ headers = rows[0]
119
+ data_rows = rows[1:]
120
+
121
+ # Build a markdown representation for quick use in RAG prompts
122
+ md_header = "| " + " | ".join(headers) + " |"
123
+ md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
124
+ md_rows = ["| " + " | ".join(row) + " |" for row in data_rows]
125
+ markdown_repr = "\n".join([md_header, md_separator] + md_rows)
126
+
127
+ return Element(
128
+ type="table",
129
+ data=TableData(headers=headers, rows=data_rows),
130
+ markdown_repr=markdown_repr,
131
+ )
@@ -0,0 +1,112 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from collections.abc import Iterator
5
+ from pathlib import Path
6
+ from typing import ClassVar
7
+
8
+ import olefile
9
+ import xlrd
10
+
11
+ from universal_parser.core.router import register
12
+ from universal_parser.core.schema import Element, TableData
13
+ from universal_parser.core.sniffer import FileType
14
+ from universal_parser.extractors.base import BaseExtractor
15
+
16
+
17
+ @register
18
+ class LegacyOfficeExtractor(BaseExtractor):
19
+ """
20
+ Extractor for legacy Microsoft Office binary formats (.doc, .xls, .ppt).
21
+ Handles:
22
+ - .xls: Streams sheets into structured TableData & Markdown using xlrd
23
+ - .doc / .ppt: Decodes text streams from OLE2 binary containers via olefile
24
+ """
25
+
26
+ supported_types: ClassVar[list[FileType]] = [
27
+ FileType.DOC,
28
+ FileType.XLS,
29
+ FileType.PPT,
30
+ ]
31
+
32
+ def stream(self, path: str | Path) -> Iterator[Element]:
33
+ path_obj = Path(path)
34
+ ext = path_obj.suffix.lower()
35
+
36
+ if ext == ".xls":
37
+ yield from self._stream_xls(path_obj)
38
+ elif ext in (".doc", ".ppt"):
39
+ yield from self._stream_ole_text(path_obj)
40
+
41
+ def _stream_xls(self, path: Path) -> Iterator[Element]:
42
+ """Stream sheets from legacy .xls files using xlrd."""
43
+ try:
44
+ wb = xlrd.open_workbook(str(path))
45
+ for sheet in wb.sheets():
46
+ rows = []
47
+ for row_idx in range(sheet.nrows):
48
+ rows_vals = [
49
+ str(sheet.cell_value(row_idx, col_idx)).strip()
50
+ for col_idx in range(sheet.ncols)
51
+ ]
52
+ if any(rows_vals):
53
+ rows.append(rows_vals)
54
+
55
+ if not rows:
56
+ continue
57
+
58
+ headers = rows[0]
59
+ data_rows = rows[1:]
60
+
61
+ md_header = "| " + " | ".join(headers) + " |"
62
+ md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
63
+ md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
64
+ markdown_repr = f"### Sheet: {sheet.name}\n\n" + "\n".join(
65
+ [md_header, md_separator] + md_rows
66
+ )
67
+
68
+ yield Element(
69
+ type="table",
70
+ text=f"Sheet: {sheet.name}",
71
+ data=TableData(headers=headers, rows=data_rows),
72
+ markdown_repr=markdown_repr,
73
+ confidence=1.0,
74
+ )
75
+
76
+ except Exception: # noqa: BLE001
77
+ return
78
+
79
+ def _stream_ole_text(self, path: Path) -> Iterator[Element]:
80
+ """Extract readable text streams from OLE2 compound files (.doc / .ppt)."""
81
+ try:
82
+ if not olefile.isOleFile(str(path)):
83
+ return
84
+
85
+ ole = olefile.OleFileIO(str(path))
86
+ extracted_text = []
87
+
88
+ for stream_name in ole.listdir():
89
+ try:
90
+ stream_data = ole.openstream(stream_name).read()
91
+ # Filter ASCII string sequences from binary stream
92
+ raw_strings = re.findall(rb"[\x20-\x7E]{4,}", stream_data)
93
+ for s in raw_strings:
94
+ decoded = s.decode("ascii", errors="ignore").strip()
95
+ if len(decoded) > 3 and not decoded.startswith("Microsoft"):
96
+ extracted_text.append(decoded)
97
+
98
+ except Exception: # noqa: BLE001, S112
99
+ continue
100
+
101
+ ole.close()
102
+
103
+ for text_chunk in extracted_text:
104
+ yield Element(
105
+ type="Paragraph",
106
+ text=text_chunk,
107
+ markdown_repr=text_chunk,
108
+ confidence=0.8,
109
+ )
110
+
111
+ except Exception: # noqa: BLE001
112
+ return
@@ -0,0 +1,109 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterator
4
+ from pathlib import Path
5
+ from typing import ClassVar
6
+
7
+ from pptx import Presentation
8
+
9
+ from universal_parser.core.router import register
10
+ from universal_parser.core.schema import Element, TableData
11
+ from universal_parser.core.sniffer import FileType
12
+ from universal_parser.extractors.base import BaseExtractor
13
+
14
+
15
+ @register
16
+ class PPTXExtractor(BaseExtractor):
17
+ """
18
+ Extractor for PowerPoint presentations (.pptx).
19
+ Handles:
20
+ - Slide titles -> Level 1 Headings with slide numbers
21
+ - Text boxes and bullet points -> Paragraphs and List Items
22
+ - Presentation tables -> Structured TableData & Markdown tables
23
+ """
24
+
25
+ supported_types: ClassVar[list[FileType]] = [FileType.PPTX]
26
+
27
+ def stream(self, path: str | Path) -> Iterator[Element]:
28
+ path_str = str(path)
29
+
30
+ try:
31
+ prs = Presentation(path_str)
32
+
33
+ except Exception: # noqa: BLE001
34
+ return
35
+
36
+ for slide_num, slide in enumerate(prs.slides, start=1):
37
+ # 1. Slide Title (Heading 1)
38
+ if slide.shapes.title and slide.shapes.title.text.strip():
39
+ title = slide.shapes.title.text.strip()
40
+ yield Element(
41
+ type="heading",
42
+ level=1,
43
+ text=f"Slide {slide_num}: {title}",
44
+ page=slide_num,
45
+ markdown_repr=f"# Slide {slide_num}: {title}",
46
+ confidence=1.0,
47
+ )
48
+ else:
49
+ yield Element(
50
+ type="heading",
51
+ level=1,
52
+ text=f"Slide {slide_num}",
53
+ page=slide_num,
54
+ markdown_repr=f"# Slide {slide_num}",
55
+ confidence=1.0,
56
+ )
57
+
58
+ # 2. Iterate through shapes on the slide
59
+ for shape in slide.shapes:
60
+ # Skip title shape since already extracted
61
+ if shape == slide.shapes.title:
62
+ continue
63
+ # A. Handle Tables in slides
64
+ if shape.has_table:
65
+ table = shape.table
66
+ rows = []
67
+ for row in table.rows:
68
+ row_cells = [cell.text.strip() for cell in row.cells]
69
+ if any(row_cells):
70
+ rows.append(row_cells)
71
+
72
+ if not rows:
73
+ continue
74
+
75
+ headers = rows[0]
76
+ data_rows = rows[1:]
77
+
78
+ md_header = "| " + " | ".join(headers) + " |"
79
+ md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
80
+ md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
81
+ markdown_repr = "\n".join([md_header, md_separator] + md_rows)
82
+
83
+ yield Element(
84
+ type="table",
85
+ page=slide_num,
86
+ text=f"Slide {slide_num} Table",
87
+ data=TableData(headers=headers, rows=data_rows),
88
+ markdown_repr=markdown_repr,
89
+ confidence=1.0,
90
+ )
91
+
92
+ # B. Handle Text Frames (paragraphs and bullet points)
93
+ elif shape.has_text_frame:
94
+ for paragraph in shape.text_frame.paragraphs:
95
+ text = paragraph.text.strip()
96
+ if not text:
97
+ continue
98
+
99
+ is_bullet = paragraph.level > 0
100
+ elem_type = "list_item" if is_bullet else "paragraph"
101
+ prefix = " " * paragraph.level + "- " if is_bullet else ""
102
+
103
+ yield Element(
104
+ type=elem_type,
105
+ page=slide_num,
106
+ text=text,
107
+ markdown_repr=f"{prefix} {text}",
108
+ confidence=1.0,
109
+ )
@@ -0,0 +1,93 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterator
4
+ from pathlib import Path
5
+ from typing import ClassVar
6
+
7
+ from openpyxl import load_workbook
8
+
9
+ from universal_parser.core.router import register
10
+ from universal_parser.core.schema import Element, TableData
11
+ from universal_parser.core.sniffer import FileType
12
+ from universal_parser.extractors.base import BaseExtractor
13
+
14
+
15
+ @register
16
+ class XlsxExtractor(BaseExtractor):
17
+ """
18
+ Extractor for .xlsx files using openpyxl.
19
+ Handles:
20
+ - Memory-safe streaming using read_only=True
21
+ - Multiple sheets (each sheet is extracted as a Table element)
22
+ - Converts formula cells to values automatically using data_only=True
23
+ - Forward-fills empty/merged header names
24
+ """
25
+
26
+ supported_types: ClassVar[list[FileType]] = [FileType.XLSX]
27
+
28
+ def stream(self, path: str | Path) -> Iterator[Element]:
29
+ """Stream sheets from an Excel workbook as Table elements."""
30
+ path_str = str(path)
31
+
32
+ try:
33
+ # read_only=True streams cells instead of loading the whole document.
34
+ # data_only=True evaluates Excel formulas and returns values.
35
+ wb = load_workbook(path_str, read_only=True, data_only=True)
36
+
37
+ except Exception: # noqa: BLE001
38
+ # Corrupted or password-protected file
39
+ return
40
+
41
+ try:
42
+ for sheet_name in wb.sheetnames:
43
+ sheet = wb[sheet_name]
44
+
45
+ # Extract rows from sheet
46
+ rows = []
47
+ for row in sheet.iter_rows(values_only=True):
48
+ # Clean the row data (convert None values to empty strings)
49
+ clean_row = [str(cell).strip() if cell is not None else "" for cell in row]
50
+
51
+ # Skip completely empty rows
52
+ if any(clean_row):
53
+ rows.append(clean_row)
54
+
55
+ if not rows:
56
+ continue
57
+
58
+ # Clean headers: handle merged cells by filling empty header slots
59
+ raw_headers = rows[0]
60
+ headers = []
61
+ last_seen_header = ""
62
+ for idx, h in enumerate(raw_headers):
63
+ if h:
64
+ headers.append(h)
65
+ last_seen_header = h
66
+ elif last_seen_header:
67
+ # Merged column continuation: e.g. "Q1 Revenue_2"
68
+ headers.append(f"{last_seen_header}_col{idx + 1}")
69
+ else:
70
+ headers.append(f"Column_{idx + 1}")
71
+
72
+ data_rows = rows[1:]
73
+
74
+ # Build a markdown representation of the spreadsheet
75
+ md_header = "| " + " | ".join(headers) + " |"
76
+ md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
77
+ md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
78
+ markdown_repr = f"### Sheet: {sheet_name}\n\n" + "\n".join(
79
+ [md_header, md_separator] + md_rows
80
+ )
81
+
82
+ yield Element(
83
+ type="table",
84
+ text=f"Sheet: {sheet_name}",
85
+ data=TableData(headers=headers, rows=data_rows),
86
+ markdown_repr=markdown_repr,
87
+ confidence=1.0,
88
+ )
89
+
90
+ except Exception: # noqa: BLE001
91
+ return
92
+ finally:
93
+ wb.close()
File without changes