exstruct 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: exstruct
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
5
5
  Keywords: excel,structure,data,exstruct
6
6
  Author: harumiWeb
@@ -112,6 +112,16 @@ for name, sheet in wb: # __iter__ yields (name, SheetData)
112
112
  wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
113
113
  first_sheet.save("sheet.json") # SheetData → file (by extension)
114
114
  print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
115
+
116
+ # ExStructEngine: per-instance options for extraction/output
117
+ from exstruct import ExStructEngine, StructOptions, OutputOptions
118
+
119
+ engine = ExStructEngine(
120
+ options=StructOptions(mode="standard"),
121
+ output=OutputOptions(include_shapes=False, pretty=True),
122
+ )
123
+ wb2 = engine.extract("input.xlsx")
124
+ engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
115
125
  ```
116
126
 
117
127
  ## Table Detection Tuning
@@ -57,6 +57,16 @@ for name, sheet in wb: # __iter__ yields (name, SheetData)
57
57
  wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
58
58
  first_sheet.save("sheet.json") # SheetData → file (by extension)
59
59
  print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
60
+
61
+ # ExStructEngine: per-instance options for extraction/output
62
+ from exstruct import ExStructEngine, StructOptions, OutputOptions
63
+
64
+ engine = ExStructEngine(
65
+ options=StructOptions(mode="standard"),
66
+ output=OutputOptions(include_shapes=False, pretty=True),
67
+ )
68
+ wb2 = engine.extract("input.xlsx")
69
+ engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
60
70
  ```
61
71
 
62
72
  ## Table Detection Tuning
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "exstruct"
3
- version = "0.2.0"
3
+ version = "0.2.1"
4
4
  description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
5
5
  readme = "README.md"
6
6
  license = { file = "LICENSE" }
@@ -1,10 +1,11 @@
1
1
  from __future__ import annotations
2
2
 
3
- from pathlib import Path
3
+ from pathlib import Path
4
4
  from typing import Literal, Optional, TextIO
5
5
 
6
6
  from .core.integrate import extract_workbook
7
7
  from .core.cells import set_table_detection_params
8
+ from .engine import ExStructEngine, OutputOptions, StructOptions
8
9
  from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
9
10
  from .models import CellRow, Chart, ChartSeries, Shape, SheetData, WorkbookData
10
11
  from .render import export_pdf, export_sheet_images
@@ -24,6 +25,9 @@ __all__ = [
24
25
  "SheetData",
25
26
  "WorkbookData",
26
27
  "set_table_detection_params",
28
+ "ExStructEngine",
29
+ "StructOptions",
30
+ "OutputOptions",
27
31
  ]
28
32
 
29
33
 
@@ -32,9 +36,8 @@ ExtractionMode = Literal["light", "standard", "verbose"]
32
36
 
33
37
  def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> WorkbookData:
34
38
  """Extract workbook semantic structure and return WorkbookData."""
35
- if mode not in ("light", "standard", "verbose"):
36
- raise ValueError(f"Unsupported mode: {mode}")
37
- return extract_workbook(Path(file_path), mode=mode)
39
+ engine = ExStructEngine(options=StructOptions(mode=mode))
40
+ return engine.extract(file_path, mode=mode)
38
41
 
39
42
 
40
43
  def export(
@@ -99,45 +102,20 @@ def process_excel(
99
102
  - If output_path is None, writes the serialized workbook to stdout (or provided stream).
100
103
  - If sheets_dir is given, also writes per-sheet files into that directory.
101
104
  """
102
- if mode not in ("light", "standard", "verbose"):
103
- raise ValueError(f"Unsupported mode: {mode}")
104
- workbook_model = extract(file_path, mode=mode)
105
- text = serialize_workbook(workbook_model, fmt=out_fmt, pretty=pretty, indent=indent)
106
- target_stream = stream
107
-
108
- def _suffix_for(fmt: str) -> str:
109
- if fmt in ("yaml", "yml"):
110
- return ".yaml"
111
- if fmt == "toon":
112
- return ".toon"
113
- if fmt == "json":
114
- return ".json"
115
- raise ValueError(f"Unsupported export format: {fmt}")
116
-
117
- if output_path is not None:
118
- output_path.write_text(text, encoding="utf-8")
119
- else:
120
- if target_stream is None:
121
- import sys
122
-
123
- target_stream = sys.stdout
124
- target_stream.write(text)
125
- if not text.endswith("\n"):
126
- target_stream.write("\n")
127
-
128
- if sheets_dir is not None:
129
- save_sheets(
130
- workbook_model,
131
- sheets_dir,
132
- fmt=out_fmt,
133
- pretty=pretty,
134
- indent=indent,
135
- )
136
-
137
- if pdf or image:
138
- base_target = output_path or file_path.with_suffix(_suffix_for(out_fmt))
139
- pdf_path = base_target.with_suffix(".pdf")
140
- export_pdf(file_path, pdf_path)
141
- if image:
142
- images_dir = pdf_path.parent / f"{pdf_path.stem}_images"
143
- export_sheet_images(file_path, images_dir, dpi=dpi)
105
+ engine = ExStructEngine(
106
+ options=StructOptions(mode=mode),
107
+ output=OutputOptions(fmt=out_fmt, pretty=pretty, indent=indent, sheets_dir=sheets_dir, stream=stream),
108
+ )
109
+ engine.process(
110
+ file_path=file_path,
111
+ output_path=output_path,
112
+ out_fmt=out_fmt,
113
+ image=image,
114
+ pdf=pdf,
115
+ dpi=dpi,
116
+ mode=mode,
117
+ pretty=pretty,
118
+ indent=indent,
119
+ sheets_dir=sheets_dir,
120
+ stream=stream,
121
+ )
@@ -0,0 +1,228 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from pathlib import Path
5
+ from typing import Literal, Optional, TextIO
6
+
7
+ from .core.integrate import extract_workbook
8
+ from .core.cells import set_table_detection_params
9
+ from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
10
+ from .models import SheetData, WorkbookData
11
+ from .render import export_pdf, export_sheet_images
12
+
13
+ ExtractionMode = Literal["light", "standard", "verbose"]
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class StructOptions:
18
+ """
19
+ Extraction-time options for ExStructEngine.
20
+
21
+ Attributes:
22
+ mode: Extraction mode. One of "light", "standard", "verbose".
23
+ - light: cells + table candidates only (no COM, shapes/charts empty)
24
+ - standard: texted shapes + arrows + charts (if COM available)
25
+ - verbose: all shapes (width/height), charts, table candidates
26
+ table_params: Optional dict passed to `set_table_detection_params(**table_params)`
27
+ before extraction. Use this to tweak table detection heuristics
28
+ per engine instance without touching global state.
29
+ """
30
+
31
+ mode: ExtractionMode = "standard"
32
+ table_params: Optional[dict] = None # forwarded to set_table_detection_params if provided
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class OutputOptions:
37
+ """
38
+ Output-time options for ExStructEngine.
39
+
40
+ Attributes:
41
+ fmt: Default export format. One of "json", "yaml", "yml", "toon".
42
+ pretty: Whether to pretty-print JSON; default False (compact).
43
+ indent: Explicit indent size. If None and pretty=True, indent=2 for JSON.
44
+ include_rows: Include SheetData.rows in output (set False to drop).
45
+ include_shapes: Include SheetData.shapes in output.
46
+ include_charts: Include SheetData.charts in output.
47
+ include_tables: Include SheetData.table_candidates in output.
48
+ sheets_dir: Optional directory to write per-sheet files (in the chosen fmt).
49
+ stream: Optional default stream for stdout output when output_path is None.
50
+ """
51
+
52
+ fmt: Literal["json", "yaml", "yml", "toon"] = "json"
53
+ pretty: bool = False
54
+ indent: int | None = None
55
+ include_rows: bool = True
56
+ include_shapes: bool = True
57
+ include_charts: bool = True
58
+ include_tables: bool = True
59
+ sheets_dir: Path | None = None
60
+ stream: TextIO | None = None
61
+
62
+
63
+ class ExStructEngine:
64
+ """
65
+ Configurable engine for ExStruct extraction and export.
66
+
67
+ Instances are immutable; override options per call if needed.
68
+
69
+ Key behaviors:
70
+ - Uses StructOptions for extraction defaults (mode, table_params).
71
+ - Uses OutputOptions for serialization defaults (fmt, pretty/indent, include* filters).
72
+ - Methods:
73
+ extract(path, mode=None) -> WorkbookData
74
+ serialize(workbook, fmt=None, pretty=None, indent=None) -> str
75
+ export(workbook, output_path=None, fmt=None, pretty=None, indent=None,
76
+ sheets_dir=None, stream=None) -> None
77
+ process(file_path, output_path=None, out_fmt=None, image=False, pdf=False,
78
+ dpi=72, mode=None, pretty=None, indent=None, sheets_dir=None,
79
+ stream=None) -> None
80
+ """
81
+
82
+ def __init__(
83
+ self,
84
+ options: StructOptions | None = None,
85
+ output: OutputOptions | None = None,
86
+ ) -> None:
87
+ self.options = options or StructOptions()
88
+ self.output = output or OutputOptions()
89
+
90
+ @staticmethod
91
+ def from_defaults() -> "ExStructEngine":
92
+ """Factory to create an engine with default options."""
93
+ return ExStructEngine()
94
+
95
+ def _apply_table_params(self) -> None:
96
+ if self.options.table_params:
97
+ set_table_detection_params(**self.options.table_params)
98
+
99
+ def _filter_sheet(self, sheet: SheetData) -> SheetData:
100
+ return SheetData(
101
+ rows=sheet.rows if self.output.include_rows else [],
102
+ shapes=sheet.shapes if self.output.include_shapes else [],
103
+ charts=sheet.charts if self.output.include_charts else [],
104
+ table_candidates=sheet.table_candidates if self.output.include_tables else [],
105
+ )
106
+
107
+ def _filter_workbook(self, wb: WorkbookData) -> WorkbookData:
108
+ filtered = {
109
+ name: self._filter_sheet(sheet)
110
+ for name, sheet in wb.sheets.items()
111
+ }
112
+ return WorkbookData(book_name=wb.book_name, sheets=filtered)
113
+
114
+ def extract(self, file_path: str | Path, *, mode: ExtractionMode | None = None) -> WorkbookData:
115
+ """Extract workbook semantic structure with the configured options."""
116
+ chosen_mode = mode or self.options.mode
117
+ if chosen_mode not in ("light", "standard", "verbose"):
118
+ raise ValueError(f"Unsupported mode: {chosen_mode}")
119
+ self._apply_table_params()
120
+ return extract_workbook(Path(file_path), mode=chosen_mode)
121
+
122
+ def serialize(
123
+ self,
124
+ data: WorkbookData,
125
+ *,
126
+ fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
127
+ pretty: Optional[bool] = None,
128
+ indent: int | None = None,
129
+ ) -> str:
130
+ """
131
+ Serialize WorkbookData using configured output defaults, applying include/exclude filters.
132
+ """
133
+ filtered = self._filter_workbook(data)
134
+ use_fmt = (fmt or self.output.fmt)
135
+ use_pretty = self.output.pretty if pretty is None else pretty
136
+ use_indent = self.output.indent if indent is None else indent
137
+ return serialize_workbook(filtered, fmt=use_fmt, pretty=use_pretty, indent=use_indent)
138
+
139
+ def export(
140
+ self,
141
+ data: WorkbookData,
142
+ output_path: Path | None = None,
143
+ *,
144
+ fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
145
+ pretty: Optional[bool] = None,
146
+ indent: int | None = None,
147
+ sheets_dir: Path | None = None,
148
+ stream: TextIO | None = None,
149
+ ) -> None:
150
+ """
151
+ Write WorkbookData to disk or stdout (when output_path is None).
152
+ Applies include/exclude filters before serialization.
153
+ """
154
+ text = self.serialize(data, fmt=fmt, pretty=pretty, indent=indent)
155
+ target_stream = stream or self.output.stream
156
+ chosen_fmt = (fmt or self.output.fmt)
157
+ chosen_sheets_dir = sheets_dir if sheets_dir is not None else self.output.sheets_dir
158
+
159
+ def _suffix_for(fmt_val: str) -> str:
160
+ if fmt_val in ("yaml", "yml"):
161
+ return ".yaml"
162
+ if fmt_val == "toon":
163
+ return ".toon"
164
+ if fmt_val == "json":
165
+ return ".json"
166
+ raise ValueError(f"Unsupported export format: {fmt_val}")
167
+
168
+ if output_path is not None:
169
+ output_path.write_text(text, encoding="utf-8")
170
+ else:
171
+ import sys
172
+
173
+ stream_target = target_stream or sys.stdout
174
+ stream_target.write(text)
175
+ if not text.endswith("\n"):
176
+ stream_target.write("\n")
177
+
178
+ if chosen_sheets_dir is not None:
179
+ filtered = self._filter_workbook(data)
180
+ save_sheets(
181
+ filtered,
182
+ chosen_sheets_dir,
183
+ fmt=chosen_fmt,
184
+ pretty=self.output.pretty if pretty is None else pretty,
185
+ indent=self.output.indent if indent is None else indent,
186
+ )
187
+
188
+ return None
189
+
190
+ def process(
191
+ self,
192
+ file_path: Path,
193
+ output_path: Path | None = None,
194
+ *,
195
+ out_fmt: Optional[str] = None,
196
+ image: bool = False,
197
+ pdf: bool = False,
198
+ dpi: int = 72,
199
+ mode: ExtractionMode | None = None,
200
+ pretty: bool | None = None,
201
+ indent: int | None = None,
202
+ sheets_dir: Path | None = None,
203
+ stream: TextIO | None = None,
204
+ ) -> None:
205
+ """
206
+ Convenience wrapper: extract, export (to file or stdout), and optionally render PDF/PNG.
207
+ """
208
+ wb = self.extract(file_path, mode=mode)
209
+ chosen_fmt = out_fmt or self.output.fmt
210
+ self.export(
211
+ wb,
212
+ output_path=output_path,
213
+ fmt=chosen_fmt, # type: ignore[arg-type]
214
+ pretty=pretty,
215
+ indent=indent,
216
+ sheets_dir=sheets_dir,
217
+ stream=stream,
218
+ )
219
+
220
+ if pdf or image:
221
+ base_target = output_path or file_path.with_suffix(
222
+ ".yaml" if chosen_fmt in ("yaml", "yml") else ".toon" if chosen_fmt == "toon" else ".json"
223
+ )
224
+ pdf_path = base_target.with_suffix(".pdf")
225
+ export_pdf(file_path, pdf_path)
226
+ if image:
227
+ images_dir = pdf_path.parent / f"{pdf_path.stem}_images"
228
+ export_sheet_images(file_path, images_dir, dpi=dpi)
@@ -163,9 +163,17 @@ class WorkbookData(BaseModel):
163
163
  case _:
164
164
  raise ValueError(f"Unsupported export format: {fmt}")
165
165
  return dest
166
+
167
+ def __getitem__(self, sheet_name: str) -> SheetData:
168
+ """Return the SheetData for the given sheet name."""
169
+ return self.sheets[sheet_name]
170
+
171
+ def __iter__(self):
172
+ """Iterate over (sheet_name, SheetData) pairs in order."""
173
+ return iter(self.sheets.items())
166
174
 
167
175
  def __iter__(self):
168
176
  return iter(self.sheets.items())
169
177
 
170
178
  def __getitem__(self, sheet_name: str) -> SheetData:
171
- return self.sheets[sheet_name]
179
+ return self.sheets[sheet_name]
File without changes
File without changes