exstruct 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {exstruct-0.2.0 → exstruct-0.2.1}/PKG-INFO +11 -1
- {exstruct-0.2.0 → exstruct-0.2.1}/README.md +10 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/pyproject.toml +1 -1
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/__init__.py +24 -46
- exstruct-0.2.1/src/exstruct/engine.py +228 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/models/__init__.py +9 -1
- {exstruct-0.2.0 → exstruct-0.2.1}/LICENSE +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/cli/main.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/core/__init__.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/core/cells.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/core/charts.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/core/integrate.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/core/shapes.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/io/__init__.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/models/maps.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/py.typed +0 -0
- {exstruct-0.2.0 → exstruct-0.2.1}/src/exstruct/render/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: exstruct
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
|
|
5
5
|
Keywords: excel,structure,data,exstruct
|
|
6
6
|
Author: harumiWeb
|
|
@@ -112,6 +112,16 @@ for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
|
112
112
|
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
113
113
|
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
114
114
|
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
115
|
+
|
|
116
|
+
# ExStructEngine: per-instance options for extraction/output
|
|
117
|
+
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
118
|
+
|
|
119
|
+
engine = ExStructEngine(
|
|
120
|
+
options=StructOptions(mode="standard"),
|
|
121
|
+
output=OutputOptions(include_shapes=False, pretty=True),
|
|
122
|
+
)
|
|
123
|
+
wb2 = engine.extract("input.xlsx")
|
|
124
|
+
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
115
125
|
```
|
|
116
126
|
|
|
117
127
|
## Table Detection Tuning
|
|
@@ -57,6 +57,16 @@ for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
|
57
57
|
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
58
58
|
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
59
59
|
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
60
|
+
|
|
61
|
+
# ExStructEngine: per-instance options for extraction/output
|
|
62
|
+
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
63
|
+
|
|
64
|
+
engine = ExStructEngine(
|
|
65
|
+
options=StructOptions(mode="standard"),
|
|
66
|
+
output=OutputOptions(include_shapes=False, pretty=True),
|
|
67
|
+
)
|
|
68
|
+
wb2 = engine.extract("input.xlsx")
|
|
69
|
+
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
60
70
|
```
|
|
61
71
|
|
|
62
72
|
## Table Detection Tuning
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
-
from pathlib import Path
|
|
3
|
+
from pathlib import Path
|
|
4
4
|
from typing import Literal, Optional, TextIO
|
|
5
5
|
|
|
6
6
|
from .core.integrate import extract_workbook
|
|
7
7
|
from .core.cells import set_table_detection_params
|
|
8
|
+
from .engine import ExStructEngine, OutputOptions, StructOptions
|
|
8
9
|
from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
|
|
9
10
|
from .models import CellRow, Chart, ChartSeries, Shape, SheetData, WorkbookData
|
|
10
11
|
from .render import export_pdf, export_sheet_images
|
|
@@ -24,6 +25,9 @@ __all__ = [
|
|
|
24
25
|
"SheetData",
|
|
25
26
|
"WorkbookData",
|
|
26
27
|
"set_table_detection_params",
|
|
28
|
+
"ExStructEngine",
|
|
29
|
+
"StructOptions",
|
|
30
|
+
"OutputOptions",
|
|
27
31
|
]
|
|
28
32
|
|
|
29
33
|
|
|
@@ -32,9 +36,8 @@ ExtractionMode = Literal["light", "standard", "verbose"]
|
|
|
32
36
|
|
|
33
37
|
def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> WorkbookData:
|
|
34
38
|
"""Extract workbook semantic structure and return WorkbookData."""
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
return extract_workbook(Path(file_path), mode=mode)
|
|
39
|
+
engine = ExStructEngine(options=StructOptions(mode=mode))
|
|
40
|
+
return engine.extract(file_path, mode=mode)
|
|
38
41
|
|
|
39
42
|
|
|
40
43
|
def export(
|
|
@@ -99,45 +102,20 @@ def process_excel(
|
|
|
99
102
|
- If output_path is None, writes the serialized workbook to stdout (or provided stream).
|
|
100
103
|
- If sheets_dir is given, also writes per-sheet files into that directory.
|
|
101
104
|
"""
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
else:
|
|
120
|
-
if target_stream is None:
|
|
121
|
-
import sys
|
|
122
|
-
|
|
123
|
-
target_stream = sys.stdout
|
|
124
|
-
target_stream.write(text)
|
|
125
|
-
if not text.endswith("\n"):
|
|
126
|
-
target_stream.write("\n")
|
|
127
|
-
|
|
128
|
-
if sheets_dir is not None:
|
|
129
|
-
save_sheets(
|
|
130
|
-
workbook_model,
|
|
131
|
-
sheets_dir,
|
|
132
|
-
fmt=out_fmt,
|
|
133
|
-
pretty=pretty,
|
|
134
|
-
indent=indent,
|
|
135
|
-
)
|
|
136
|
-
|
|
137
|
-
if pdf or image:
|
|
138
|
-
base_target = output_path or file_path.with_suffix(_suffix_for(out_fmt))
|
|
139
|
-
pdf_path = base_target.with_suffix(".pdf")
|
|
140
|
-
export_pdf(file_path, pdf_path)
|
|
141
|
-
if image:
|
|
142
|
-
images_dir = pdf_path.parent / f"{pdf_path.stem}_images"
|
|
143
|
-
export_sheet_images(file_path, images_dir, dpi=dpi)
|
|
105
|
+
engine = ExStructEngine(
|
|
106
|
+
options=StructOptions(mode=mode),
|
|
107
|
+
output=OutputOptions(fmt=out_fmt, pretty=pretty, indent=indent, sheets_dir=sheets_dir, stream=stream),
|
|
108
|
+
)
|
|
109
|
+
engine.process(
|
|
110
|
+
file_path=file_path,
|
|
111
|
+
output_path=output_path,
|
|
112
|
+
out_fmt=out_fmt,
|
|
113
|
+
image=image,
|
|
114
|
+
pdf=pdf,
|
|
115
|
+
dpi=dpi,
|
|
116
|
+
mode=mode,
|
|
117
|
+
pretty=pretty,
|
|
118
|
+
indent=indent,
|
|
119
|
+
sheets_dir=sheets_dir,
|
|
120
|
+
stream=stream,
|
|
121
|
+
)
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Literal, Optional, TextIO
|
|
6
|
+
|
|
7
|
+
from .core.integrate import extract_workbook
|
|
8
|
+
from .core.cells import set_table_detection_params
|
|
9
|
+
from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
|
|
10
|
+
from .models import SheetData, WorkbookData
|
|
11
|
+
from .render import export_pdf, export_sheet_images
|
|
12
|
+
|
|
13
|
+
ExtractionMode = Literal["light", "standard", "verbose"]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass(frozen=True)
|
|
17
|
+
class StructOptions:
|
|
18
|
+
"""
|
|
19
|
+
Extraction-time options for ExStructEngine.
|
|
20
|
+
|
|
21
|
+
Attributes:
|
|
22
|
+
mode: Extraction mode. One of "light", "standard", "verbose".
|
|
23
|
+
- light: cells + table candidates only (no COM, shapes/charts empty)
|
|
24
|
+
- standard: texted shapes + arrows + charts (if COM available)
|
|
25
|
+
- verbose: all shapes (width/height), charts, table candidates
|
|
26
|
+
table_params: Optional dict passed to `set_table_detection_params(**table_params)`
|
|
27
|
+
before extraction. Use this to tweak table detection heuristics
|
|
28
|
+
per engine instance without touching global state.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
mode: ExtractionMode = "standard"
|
|
32
|
+
table_params: Optional[dict] = None # forwarded to set_table_detection_params if provided
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class OutputOptions:
|
|
37
|
+
"""
|
|
38
|
+
Output-time options for ExStructEngine.
|
|
39
|
+
|
|
40
|
+
Attributes:
|
|
41
|
+
fmt: Default export format. One of "json", "yaml", "yml", "toon".
|
|
42
|
+
pretty: Whether to pretty-print JSON; default False (compact).
|
|
43
|
+
indent: Explicit indent size. If None and pretty=True, indent=2 for JSON.
|
|
44
|
+
include_rows: Include SheetData.rows in output (set False to drop).
|
|
45
|
+
include_shapes: Include SheetData.shapes in output.
|
|
46
|
+
include_charts: Include SheetData.charts in output.
|
|
47
|
+
include_tables: Include SheetData.table_candidates in output.
|
|
48
|
+
sheets_dir: Optional directory to write per-sheet files (in the chosen fmt).
|
|
49
|
+
stream: Optional default stream for stdout output when output_path is None.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
fmt: Literal["json", "yaml", "yml", "toon"] = "json"
|
|
53
|
+
pretty: bool = False
|
|
54
|
+
indent: int | None = None
|
|
55
|
+
include_rows: bool = True
|
|
56
|
+
include_shapes: bool = True
|
|
57
|
+
include_charts: bool = True
|
|
58
|
+
include_tables: bool = True
|
|
59
|
+
sheets_dir: Path | None = None
|
|
60
|
+
stream: TextIO | None = None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class ExStructEngine:
|
|
64
|
+
"""
|
|
65
|
+
Configurable engine for ExStruct extraction and export.
|
|
66
|
+
|
|
67
|
+
Instances are immutable; override options per call if needed.
|
|
68
|
+
|
|
69
|
+
Key behaviors:
|
|
70
|
+
- Uses StructOptions for extraction defaults (mode, table_params).
|
|
71
|
+
- Uses OutputOptions for serialization defaults (fmt, pretty/indent, include* filters).
|
|
72
|
+
- Methods:
|
|
73
|
+
extract(path, mode=None) -> WorkbookData
|
|
74
|
+
serialize(workbook, fmt=None, pretty=None, indent=None) -> str
|
|
75
|
+
export(workbook, output_path=None, fmt=None, pretty=None, indent=None,
|
|
76
|
+
sheets_dir=None, stream=None) -> None
|
|
77
|
+
process(file_path, output_path=None, out_fmt=None, image=False, pdf=False,
|
|
78
|
+
dpi=72, mode=None, pretty=None, indent=None, sheets_dir=None,
|
|
79
|
+
stream=None) -> None
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
def __init__(
|
|
83
|
+
self,
|
|
84
|
+
options: StructOptions | None = None,
|
|
85
|
+
output: OutputOptions | None = None,
|
|
86
|
+
) -> None:
|
|
87
|
+
self.options = options or StructOptions()
|
|
88
|
+
self.output = output or OutputOptions()
|
|
89
|
+
|
|
90
|
+
@staticmethod
|
|
91
|
+
def from_defaults() -> "ExStructEngine":
|
|
92
|
+
"""Factory to create an engine with default options."""
|
|
93
|
+
return ExStructEngine()
|
|
94
|
+
|
|
95
|
+
def _apply_table_params(self) -> None:
|
|
96
|
+
if self.options.table_params:
|
|
97
|
+
set_table_detection_params(**self.options.table_params)
|
|
98
|
+
|
|
99
|
+
def _filter_sheet(self, sheet: SheetData) -> SheetData:
|
|
100
|
+
return SheetData(
|
|
101
|
+
rows=sheet.rows if self.output.include_rows else [],
|
|
102
|
+
shapes=sheet.shapes if self.output.include_shapes else [],
|
|
103
|
+
charts=sheet.charts if self.output.include_charts else [],
|
|
104
|
+
table_candidates=sheet.table_candidates if self.output.include_tables else [],
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
def _filter_workbook(self, wb: WorkbookData) -> WorkbookData:
|
|
108
|
+
filtered = {
|
|
109
|
+
name: self._filter_sheet(sheet)
|
|
110
|
+
for name, sheet in wb.sheets.items()
|
|
111
|
+
}
|
|
112
|
+
return WorkbookData(book_name=wb.book_name, sheets=filtered)
|
|
113
|
+
|
|
114
|
+
def extract(self, file_path: str | Path, *, mode: ExtractionMode | None = None) -> WorkbookData:
|
|
115
|
+
"""Extract workbook semantic structure with the configured options."""
|
|
116
|
+
chosen_mode = mode or self.options.mode
|
|
117
|
+
if chosen_mode not in ("light", "standard", "verbose"):
|
|
118
|
+
raise ValueError(f"Unsupported mode: {chosen_mode}")
|
|
119
|
+
self._apply_table_params()
|
|
120
|
+
return extract_workbook(Path(file_path), mode=chosen_mode)
|
|
121
|
+
|
|
122
|
+
def serialize(
|
|
123
|
+
self,
|
|
124
|
+
data: WorkbookData,
|
|
125
|
+
*,
|
|
126
|
+
fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
|
|
127
|
+
pretty: Optional[bool] = None,
|
|
128
|
+
indent: int | None = None,
|
|
129
|
+
) -> str:
|
|
130
|
+
"""
|
|
131
|
+
Serialize WorkbookData using configured output defaults, applying include/exclude filters.
|
|
132
|
+
"""
|
|
133
|
+
filtered = self._filter_workbook(data)
|
|
134
|
+
use_fmt = (fmt or self.output.fmt)
|
|
135
|
+
use_pretty = self.output.pretty if pretty is None else pretty
|
|
136
|
+
use_indent = self.output.indent if indent is None else indent
|
|
137
|
+
return serialize_workbook(filtered, fmt=use_fmt, pretty=use_pretty, indent=use_indent)
|
|
138
|
+
|
|
139
|
+
def export(
|
|
140
|
+
self,
|
|
141
|
+
data: WorkbookData,
|
|
142
|
+
output_path: Path | None = None,
|
|
143
|
+
*,
|
|
144
|
+
fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
|
|
145
|
+
pretty: Optional[bool] = None,
|
|
146
|
+
indent: int | None = None,
|
|
147
|
+
sheets_dir: Path | None = None,
|
|
148
|
+
stream: TextIO | None = None,
|
|
149
|
+
) -> None:
|
|
150
|
+
"""
|
|
151
|
+
Write WorkbookData to disk or stdout (when output_path is None).
|
|
152
|
+
Applies include/exclude filters before serialization.
|
|
153
|
+
"""
|
|
154
|
+
text = self.serialize(data, fmt=fmt, pretty=pretty, indent=indent)
|
|
155
|
+
target_stream = stream or self.output.stream
|
|
156
|
+
chosen_fmt = (fmt or self.output.fmt)
|
|
157
|
+
chosen_sheets_dir = sheets_dir if sheets_dir is not None else self.output.sheets_dir
|
|
158
|
+
|
|
159
|
+
def _suffix_for(fmt_val: str) -> str:
|
|
160
|
+
if fmt_val in ("yaml", "yml"):
|
|
161
|
+
return ".yaml"
|
|
162
|
+
if fmt_val == "toon":
|
|
163
|
+
return ".toon"
|
|
164
|
+
if fmt_val == "json":
|
|
165
|
+
return ".json"
|
|
166
|
+
raise ValueError(f"Unsupported export format: {fmt_val}")
|
|
167
|
+
|
|
168
|
+
if output_path is not None:
|
|
169
|
+
output_path.write_text(text, encoding="utf-8")
|
|
170
|
+
else:
|
|
171
|
+
import sys
|
|
172
|
+
|
|
173
|
+
stream_target = target_stream or sys.stdout
|
|
174
|
+
stream_target.write(text)
|
|
175
|
+
if not text.endswith("\n"):
|
|
176
|
+
stream_target.write("\n")
|
|
177
|
+
|
|
178
|
+
if chosen_sheets_dir is not None:
|
|
179
|
+
filtered = self._filter_workbook(data)
|
|
180
|
+
save_sheets(
|
|
181
|
+
filtered,
|
|
182
|
+
chosen_sheets_dir,
|
|
183
|
+
fmt=chosen_fmt,
|
|
184
|
+
pretty=self.output.pretty if pretty is None else pretty,
|
|
185
|
+
indent=self.output.indent if indent is None else indent,
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
return None
|
|
189
|
+
|
|
190
|
+
def process(
|
|
191
|
+
self,
|
|
192
|
+
file_path: Path,
|
|
193
|
+
output_path: Path | None = None,
|
|
194
|
+
*,
|
|
195
|
+
out_fmt: Optional[str] = None,
|
|
196
|
+
image: bool = False,
|
|
197
|
+
pdf: bool = False,
|
|
198
|
+
dpi: int = 72,
|
|
199
|
+
mode: ExtractionMode | None = None,
|
|
200
|
+
pretty: bool | None = None,
|
|
201
|
+
indent: int | None = None,
|
|
202
|
+
sheets_dir: Path | None = None,
|
|
203
|
+
stream: TextIO | None = None,
|
|
204
|
+
) -> None:
|
|
205
|
+
"""
|
|
206
|
+
Convenience wrapper: extract, export (to file or stdout), and optionally render PDF/PNG.
|
|
207
|
+
"""
|
|
208
|
+
wb = self.extract(file_path, mode=mode)
|
|
209
|
+
chosen_fmt = out_fmt or self.output.fmt
|
|
210
|
+
self.export(
|
|
211
|
+
wb,
|
|
212
|
+
output_path=output_path,
|
|
213
|
+
fmt=chosen_fmt, # type: ignore[arg-type]
|
|
214
|
+
pretty=pretty,
|
|
215
|
+
indent=indent,
|
|
216
|
+
sheets_dir=sheets_dir,
|
|
217
|
+
stream=stream,
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
if pdf or image:
|
|
221
|
+
base_target = output_path or file_path.with_suffix(
|
|
222
|
+
".yaml" if chosen_fmt in ("yaml", "yml") else ".toon" if chosen_fmt == "toon" else ".json"
|
|
223
|
+
)
|
|
224
|
+
pdf_path = base_target.with_suffix(".pdf")
|
|
225
|
+
export_pdf(file_path, pdf_path)
|
|
226
|
+
if image:
|
|
227
|
+
images_dir = pdf_path.parent / f"{pdf_path.stem}_images"
|
|
228
|
+
export_sheet_images(file_path, images_dir, dpi=dpi)
|
|
@@ -163,9 +163,17 @@ class WorkbookData(BaseModel):
|
|
|
163
163
|
case _:
|
|
164
164
|
raise ValueError(f"Unsupported export format: {fmt}")
|
|
165
165
|
return dest
|
|
166
|
+
|
|
167
|
+
def __getitem__(self, sheet_name: str) -> SheetData:
|
|
168
|
+
"""Return the SheetData for the given sheet name."""
|
|
169
|
+
return self.sheets[sheet_name]
|
|
170
|
+
|
|
171
|
+
def __iter__(self):
|
|
172
|
+
"""Iterate over (sheet_name, SheetData) pairs in order."""
|
|
173
|
+
return iter(self.sheets.items())
|
|
166
174
|
|
|
167
175
|
def __iter__(self):
|
|
168
176
|
return iter(self.sheets.items())
|
|
169
177
|
|
|
170
178
|
def __getitem__(self, sheet_name: str) -> SheetData:
|
|
171
|
-
return self.sheets[sheet_name]
|
|
179
|
+
return self.sheets[sheet_name]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|