exstruct 0.2.3__tar.gz → 0.2.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {exstruct-0.2.3 → exstruct-0.2.11}/PKG-INFO +34 -37
- {exstruct-0.2.3 → exstruct-0.2.11}/README.md +33 -36
- {exstruct-0.2.3 → exstruct-0.2.11}/pyproject.toml +1 -1
- exstruct-0.2.11/src/exstruct/__init__.py +121 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/cli/main.py +0 -6
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/core/cells.py +19 -58
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/core/charts.py +2 -12
- exstruct-0.2.11/src/exstruct/core/integrate.py +132 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/engine.py +134 -249
- exstruct-0.2.11/src/exstruct/io/__init__.py +187 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/models/__init__.py +169 -244
- exstruct-0.2.3/src/exstruct/__init__.py +0 -215
- exstruct-0.2.3/src/exstruct/core/integrate.py +0 -252
- exstruct-0.2.3/src/exstruct/io/__init__.py +0 -418
- {exstruct-0.2.3 → exstruct-0.2.11}/LICENSE +0 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/core/__init__.py +0 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/core/shapes.py +0 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/models/maps.py +0 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/py.typed +0 -0
- {exstruct-0.2.3 → exstruct-0.2.11}/src/exstruct/render/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: exstruct
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.11
|
|
4
4
|
Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
|
|
5
5
|
Keywords: excel,structure,data,exstruct
|
|
6
6
|
Author: harumiWeb
|
|
@@ -56,16 +56,16 @@ Description-Content-Type: text/markdown
|
|
|
56
56
|
|
|
57
57
|
# ExStruct — Excel Structured Extraction Engine
|
|
58
58
|
|
|
59
|
-
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/ci.yml)
|
|
60
60
|
|
|
61
61
|

|
|
62
62
|
|
|
63
|
-
ExStruct reads Excel workbooks and outputs structured data (
|
|
63
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
64
64
|
|
|
65
65
|
## Features
|
|
66
66
|
|
|
67
|
-
- **Excel → Structured JSON**: cells, shapes, charts, table candidates
|
|
68
|
-
- **Output modes**: `light` (cells + table candidates
|
|
67
|
+
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
68
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
69
69
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
70
70
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
71
71
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -77,18 +77,17 @@ ExStruct reads Excel workbooks and outputs structured data (cells, table candida
|
|
|
77
77
|
pip install exstruct
|
|
78
78
|
```
|
|
79
79
|
|
|
80
|
-
Optional extras:
|
|
81
|
-
|
|
82
|
-
- YAML: `pip install pyyaml`
|
|
83
|
-
- TOON: `pip install python-toon`
|
|
84
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
85
|
-
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
86
|
-
|
|
87
|
-
Platform note:
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
## Quick Start (CLI)
|
|
80
|
+
Optional extras:
|
|
81
|
+
|
|
82
|
+
- YAML: `pip install pyyaml`
|
|
83
|
+
- TOON: `pip install python-toon`
|
|
84
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
85
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
86
|
+
|
|
87
|
+
Platform note:
|
|
88
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
89
|
+
|
|
90
|
+
## Quick Start (CLI)
|
|
92
91
|
|
|
93
92
|
```bash
|
|
94
93
|
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
@@ -96,7 +95,6 @@ exstruct input.xlsx -o out.json --pretty # pretty JSON to a file
|
|
|
96
95
|
exstruct input.xlsx --format yaml # YAML (needs pyyaml)
|
|
97
96
|
exstruct input.xlsx --format toon # TOON (needs python-toon)
|
|
98
97
|
exstruct input.xlsx --sheets-dir sheets/ # split per sheet in chosen format
|
|
99
|
-
exstruct input.xlsx --print-areas-dir areas/ # split per print area (if any)
|
|
100
98
|
exstruct input.xlsx --mode light # cells + table candidates only
|
|
101
99
|
exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
|
|
102
100
|
```
|
|
@@ -114,7 +112,7 @@ set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
|
114
112
|
wb = extract("input.xlsx", mode="standard")
|
|
115
113
|
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
116
114
|
|
|
117
|
-
# Model helpers: iterate, index, and serialize directly
|
|
115
|
+
# Model helpers: iterate, index, and serialize directly from the models
|
|
118
116
|
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
119
117
|
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
120
118
|
print(name, len(sheet.rows))
|
|
@@ -126,19 +124,11 @@ print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
|
126
124
|
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
127
125
|
|
|
128
126
|
engine = ExStructEngine(
|
|
129
|
-
options=StructOptions(mode="
|
|
127
|
+
options=StructOptions(mode="standard"),
|
|
130
128
|
output=OutputOptions(include_shapes=False, pretty=True),
|
|
131
129
|
)
|
|
132
130
|
wb2 = engine.extract("input.xlsx")
|
|
133
131
|
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
134
|
-
|
|
135
|
-
# Enable hyperlinks in other modes
|
|
136
|
-
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
137
|
-
with_links = engine_links.extract("input.xlsx")
|
|
138
|
-
|
|
139
|
-
# Export per print area (if print areas exist)
|
|
140
|
-
from exstruct import export_print_areas_as
|
|
141
|
-
export_print_areas_as(wb, "areas", fmt="json", pretty=True)
|
|
142
132
|
```
|
|
143
133
|
|
|
144
134
|
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
@@ -161,8 +151,8 @@ Use higher thresholds to reduce false positives; lower them if true tables are m
|
|
|
161
151
|
## Output Modes
|
|
162
152
|
|
|
163
153
|
- **light**: cells + table candidates (no COM needed).
|
|
164
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
165
|
-
- **verbose**: all shapes (with width/height), charts, table candidates
|
|
154
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
155
|
+
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
166
156
|
|
|
167
157
|
## Error Handling / Fallbacks
|
|
168
158
|
|
|
@@ -191,7 +181,6 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
191
181
|
(Screenshot below is the actual sample Excel sheet)
|
|
192
182
|

|
|
193
183
|
Sample workbook: `sample/sample.xlsx`
|
|
194
|
-
Sample workbook: `sample/sample.xlsx`
|
|
195
184
|
|
|
196
185
|
### 1. Input: Excel Sheet Overview
|
|
197
186
|
|
|
@@ -378,12 +367,6 @@ In short, **exstruct = “an engine that converts Excel into a format AI can und
|
|
|
378
367
|
- Default JSON is compact to reduce tokens; use `--pretty` or `pretty=True` when readability matters.
|
|
379
368
|
- Field `table_candidates` replaces `tables`; adjust downstream consumers accordingly.
|
|
380
369
|
|
|
381
|
-
## Print Areas (PrintArea / PrintAreaView)
|
|
382
|
-
|
|
383
|
-
- `SheetData.print_areas` holds print areas (cell coordinates) in light/standard/verbose.
|
|
384
|
-
- Use `export_print_areas_as(...)` or CLI `--print-areas-dir` to write one file per print area (nothing is written if none exist).
|
|
385
|
-
- `PrintAreaView` includes rows and table candidates inside the area, plus shapes/charts that overlap the area (size-less shapes are treated as points). `normalize=True` rebases row/col indices to the area origin.
|
|
386
|
-
|
|
387
370
|
## License
|
|
388
371
|
|
|
389
372
|
BSD-3-Clause. See `LICENSE` for details.
|
|
@@ -391,3 +374,17 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
391
374
|
## Documentation
|
|
392
375
|
|
|
393
376
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
377
|
+
# Engine option cheat sheet
|
|
378
|
+
|
|
379
|
+
| Option class | Field | Meaning |
|
|
380
|
+
| -------------- | ------------------- | ------- |
|
|
381
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
382
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
383
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
384
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
385
|
+
| | include_rows | Include rows (False to drop) |
|
|
386
|
+
| | include_shapes | Include shapes |
|
|
387
|
+
| | include_charts | Include charts |
|
|
388
|
+
| | include_tables | Include table_candidates |
|
|
389
|
+
| | sheets_dir | Optional directory for per-sheet exports |
|
|
390
|
+
| | stream | Default stream when output_path is None |
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
# ExStruct — Excel Structured Extraction Engine
|
|
2
2
|
|
|
3
|
-
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/ci.yml)
|
|
4
4
|
|
|
5
5
|

|
|
6
6
|
|
|
7
|
-
ExStruct reads Excel workbooks and outputs structured data (
|
|
7
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
8
8
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
11
|
-
- **Excel → Structured JSON**: cells, shapes, charts, table candidates
|
|
12
|
-
- **Output modes**: `light` (cells + table candidates
|
|
11
|
+
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
12
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
13
13
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
14
14
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
15
15
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -21,18 +21,17 @@ ExStruct reads Excel workbooks and outputs structured data (cells, table candida
|
|
|
21
21
|
pip install exstruct
|
|
22
22
|
```
|
|
23
23
|
|
|
24
|
-
Optional extras:
|
|
25
|
-
|
|
26
|
-
- YAML: `pip install pyyaml`
|
|
27
|
-
- TOON: `pip install python-toon`
|
|
28
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
29
|
-
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
30
|
-
|
|
31
|
-
Platform note:
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
## Quick Start (CLI)
|
|
24
|
+
Optional extras:
|
|
25
|
+
|
|
26
|
+
- YAML: `pip install pyyaml`
|
|
27
|
+
- TOON: `pip install python-toon`
|
|
28
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
29
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
30
|
+
|
|
31
|
+
Platform note:
|
|
32
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
33
|
+
|
|
34
|
+
## Quick Start (CLI)
|
|
36
35
|
|
|
37
36
|
```bash
|
|
38
37
|
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
@@ -40,7 +39,6 @@ exstruct input.xlsx -o out.json --pretty # pretty JSON to a file
|
|
|
40
39
|
exstruct input.xlsx --format yaml # YAML (needs pyyaml)
|
|
41
40
|
exstruct input.xlsx --format toon # TOON (needs python-toon)
|
|
42
41
|
exstruct input.xlsx --sheets-dir sheets/ # split per sheet in chosen format
|
|
43
|
-
exstruct input.xlsx --print-areas-dir areas/ # split per print area (if any)
|
|
44
42
|
exstruct input.xlsx --mode light # cells + table candidates only
|
|
45
43
|
exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
|
|
46
44
|
```
|
|
@@ -58,7 +56,7 @@ set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
|
58
56
|
wb = extract("input.xlsx", mode="standard")
|
|
59
57
|
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
60
58
|
|
|
61
|
-
# Model helpers: iterate, index, and serialize directly
|
|
59
|
+
# Model helpers: iterate, index, and serialize directly from the models
|
|
62
60
|
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
63
61
|
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
64
62
|
print(name, len(sheet.rows))
|
|
@@ -70,19 +68,11 @@ print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
|
70
68
|
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
71
69
|
|
|
72
70
|
engine = ExStructEngine(
|
|
73
|
-
options=StructOptions(mode="
|
|
71
|
+
options=StructOptions(mode="standard"),
|
|
74
72
|
output=OutputOptions(include_shapes=False, pretty=True),
|
|
75
73
|
)
|
|
76
74
|
wb2 = engine.extract("input.xlsx")
|
|
77
75
|
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
78
|
-
|
|
79
|
-
# Enable hyperlinks in other modes
|
|
80
|
-
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
81
|
-
with_links = engine_links.extract("input.xlsx")
|
|
82
|
-
|
|
83
|
-
# Export per print area (if print areas exist)
|
|
84
|
-
from exstruct import export_print_areas_as
|
|
85
|
-
export_print_areas_as(wb, "areas", fmt="json", pretty=True)
|
|
86
76
|
```
|
|
87
77
|
|
|
88
78
|
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
@@ -105,8 +95,8 @@ Use higher thresholds to reduce false positives; lower them if true tables are m
|
|
|
105
95
|
## Output Modes
|
|
106
96
|
|
|
107
97
|
- **light**: cells + table candidates (no COM needed).
|
|
108
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
109
|
-
- **verbose**: all shapes (with width/height), charts, table candidates
|
|
98
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
99
|
+
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
110
100
|
|
|
111
101
|
## Error Handling / Fallbacks
|
|
112
102
|
|
|
@@ -135,7 +125,6 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
135
125
|
(Screenshot below is the actual sample Excel sheet)
|
|
136
126
|

|
|
137
127
|
Sample workbook: `sample/sample.xlsx`
|
|
138
|
-
Sample workbook: `sample/sample.xlsx`
|
|
139
128
|
|
|
140
129
|
### 1. Input: Excel Sheet Overview
|
|
141
130
|
|
|
@@ -322,12 +311,6 @@ In short, **exstruct = “an engine that converts Excel into a format AI can und
|
|
|
322
311
|
- Default JSON is compact to reduce tokens; use `--pretty` or `pretty=True` when readability matters.
|
|
323
312
|
- Field `table_candidates` replaces `tables`; adjust downstream consumers accordingly.
|
|
324
313
|
|
|
325
|
-
## Print Areas (PrintArea / PrintAreaView)
|
|
326
|
-
|
|
327
|
-
- `SheetData.print_areas` holds print areas (cell coordinates) in light/standard/verbose.
|
|
328
|
-
- Use `export_print_areas_as(...)` or CLI `--print-areas-dir` to write one file per print area (nothing is written if none exist).
|
|
329
|
-
- `PrintAreaView` includes rows and table candidates inside the area, plus shapes/charts that overlap the area (size-less shapes are treated as points). `normalize=True` rebases row/col indices to the area origin.
|
|
330
|
-
|
|
331
314
|
## License
|
|
332
315
|
|
|
333
316
|
BSD-3-Clause. See `LICENSE` for details.
|
|
@@ -335,3 +318,17 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
335
318
|
## Documentation
|
|
336
319
|
|
|
337
320
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
321
|
+
# Engine option cheat sheet
|
|
322
|
+
|
|
323
|
+
| Option class | Field | Meaning |
|
|
324
|
+
| -------------- | ------------------- | ------- |
|
|
325
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
326
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
327
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
328
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
329
|
+
| | include_rows | Include rows (False to drop) |
|
|
330
|
+
| | include_shapes | Include shapes |
|
|
331
|
+
| | include_charts | Include charts |
|
|
332
|
+
| | include_tables | Include table_candidates |
|
|
333
|
+
| | sheets_dir | Optional directory for per-sheet exports |
|
|
334
|
+
| | stream | Default stream when output_path is None |
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Literal, Optional, TextIO
|
|
5
|
+
|
|
6
|
+
from .core.integrate import extract_workbook
|
|
7
|
+
from .core.cells import set_table_detection_params
|
|
8
|
+
from .engine import ExStructEngine, OutputOptions, StructOptions
|
|
9
|
+
from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
|
|
10
|
+
from .models import CellRow, Chart, ChartSeries, Shape, SheetData, WorkbookData
|
|
11
|
+
from .render import export_pdf, export_sheet_images
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"extract",
|
|
15
|
+
"export",
|
|
16
|
+
"export_sheets",
|
|
17
|
+
"export_pdf",
|
|
18
|
+
"export_sheet_images",
|
|
19
|
+
"process_excel",
|
|
20
|
+
"ExtractionMode",
|
|
21
|
+
"CellRow",
|
|
22
|
+
"Shape",
|
|
23
|
+
"ChartSeries",
|
|
24
|
+
"Chart",
|
|
25
|
+
"SheetData",
|
|
26
|
+
"WorkbookData",
|
|
27
|
+
"set_table_detection_params",
|
|
28
|
+
"ExStructEngine",
|
|
29
|
+
"StructOptions",
|
|
30
|
+
"OutputOptions",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
ExtractionMode = Literal["light", "standard", "verbose"]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> WorkbookData:
|
|
38
|
+
"""Extract workbook semantic structure and return WorkbookData."""
|
|
39
|
+
engine = ExStructEngine(options=StructOptions(mode=mode))
|
|
40
|
+
return engine.extract(file_path, mode=mode)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def export(
|
|
44
|
+
data: WorkbookData,
|
|
45
|
+
path: str | Path,
|
|
46
|
+
fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
|
|
47
|
+
*,
|
|
48
|
+
pretty: bool = False,
|
|
49
|
+
indent: int | None = None,
|
|
50
|
+
) -> None:
|
|
51
|
+
"""Export WorkbookData to supported file formats (json/yaml/toon)."""
|
|
52
|
+
dest = Path(path)
|
|
53
|
+
format_hint = (fmt or dest.suffix.lstrip(".") or "json").lower()
|
|
54
|
+
match format_hint:
|
|
55
|
+
case "json":
|
|
56
|
+
save_as_json(data, dest, pretty=pretty, indent=indent)
|
|
57
|
+
case "yaml" | "yml":
|
|
58
|
+
save_as_yaml(data, dest)
|
|
59
|
+
case "toon":
|
|
60
|
+
save_as_toon(data, dest)
|
|
61
|
+
case _:
|
|
62
|
+
raise ValueError(f"Unsupported export format: {format_hint}")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def export_sheets(data: WorkbookData, dir_path: str | Path) -> dict[str, Path]:
|
|
66
|
+
"""
|
|
67
|
+
Export each sheet as a JSON file (book_name + SheetData) into a directory.
|
|
68
|
+
Returns a mapping of sheet name to written path.
|
|
69
|
+
"""
|
|
70
|
+
return save_sheets(data, Path(dir_path), fmt="json")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def export_sheets_as(
|
|
74
|
+
data: WorkbookData,
|
|
75
|
+
dir_path: str | Path,
|
|
76
|
+
fmt: Literal["json", "yaml", "yml", "toon"] = "json",
|
|
77
|
+
*,
|
|
78
|
+
pretty: bool = False,
|
|
79
|
+
indent: int | None = None,
|
|
80
|
+
) -> dict[str, Path]:
|
|
81
|
+
"""
|
|
82
|
+
Export each sheet in the given format (json/yaml/toon), including book_name and SheetData; returns sheet name → path map.
|
|
83
|
+
"""
|
|
84
|
+
return save_sheets(data, Path(dir_path), fmt=fmt, pretty=pretty, indent=indent)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def process_excel(
|
|
88
|
+
file_path: Path,
|
|
89
|
+
output_path: Path | None = None,
|
|
90
|
+
out_fmt: str = "json",
|
|
91
|
+
image: bool = False,
|
|
92
|
+
pdf: bool = False,
|
|
93
|
+
dpi: int = 72,
|
|
94
|
+
mode: ExtractionMode = "standard",
|
|
95
|
+
pretty: bool = False,
|
|
96
|
+
indent: int | None = None,
|
|
97
|
+
sheets_dir: Path | None = None,
|
|
98
|
+
stream: TextIO | None = None,
|
|
99
|
+
) -> None:
|
|
100
|
+
"""
|
|
101
|
+
Convenience wrapper for CLI: export workbook and optionally PDF/PNG images (Excel required for rendering).
|
|
102
|
+
- If output_path is None, writes the serialized workbook to stdout (or provided stream).
|
|
103
|
+
- If sheets_dir is given, also writes per-sheet files into that directory.
|
|
104
|
+
"""
|
|
105
|
+
engine = ExStructEngine(
|
|
106
|
+
options=StructOptions(mode=mode),
|
|
107
|
+
output=OutputOptions(fmt=out_fmt, pretty=pretty, indent=indent, sheets_dir=sheets_dir, stream=stream),
|
|
108
|
+
)
|
|
109
|
+
engine.process(
|
|
110
|
+
file_path=file_path,
|
|
111
|
+
output_path=output_path,
|
|
112
|
+
out_fmt=out_fmt,
|
|
113
|
+
image=image,
|
|
114
|
+
pdf=pdf,
|
|
115
|
+
dpi=dpi,
|
|
116
|
+
mode=mode,
|
|
117
|
+
pretty=pretty,
|
|
118
|
+
indent=indent,
|
|
119
|
+
sheets_dir=sheets_dir,
|
|
120
|
+
stream=stream,
|
|
121
|
+
)
|
|
@@ -57,11 +57,6 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
57
57
|
type=Path,
|
|
58
58
|
help="Optional directory to write one file per sheet (format follows --format).",
|
|
59
59
|
)
|
|
60
|
-
parser.add_argument(
|
|
61
|
-
"--print-areas-dir",
|
|
62
|
-
type=Path,
|
|
63
|
-
help="Optional directory to write one file per print area (format follows --format).",
|
|
64
|
-
)
|
|
65
60
|
return parser
|
|
66
61
|
|
|
67
62
|
|
|
@@ -85,7 +80,6 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
85
80
|
mode=args.mode,
|
|
86
81
|
pretty=args.pretty,
|
|
87
82
|
sheets_dir=args.sheets_dir,
|
|
88
|
-
print_areas_dir=args.print_areas_dir,
|
|
89
83
|
)
|
|
90
84
|
return 0
|
|
91
85
|
except Exception as e:
|
|
@@ -33,64 +33,25 @@ def warn_once(key: str, message: str) -> None:
|
|
|
33
33
|
_warned_keys.add(key)
|
|
34
34
|
|
|
35
35
|
|
|
36
|
-
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
-
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
-
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
-
result: Dict[str, List[CellRow]] = {}
|
|
40
|
-
for sheet_name, df in dfs.items():
|
|
41
|
-
df = df.fillna("")
|
|
42
|
-
rows: List[CellRow] = []
|
|
43
|
-
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
-
filtered: Dict[str, int | float | str] = {}
|
|
45
|
-
for j, v in enumerate(row):
|
|
46
|
-
s = "" if v is None else str(v)
|
|
47
|
-
if s.strip() == "":
|
|
48
|
-
continue
|
|
49
|
-
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
-
if not filtered:
|
|
51
|
-
continue
|
|
52
|
-
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
-
result[sheet_name] = rows
|
|
54
|
-
return result
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
def extract_sheet_cells_with_links(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
58
|
-
"""
|
|
59
|
-
Extract cells and hyperlinks per sheet.
|
|
60
|
-
|
|
61
|
-
Returns:
|
|
62
|
-
{sheet_name: [CellRow(r=..., c=..., links={"col_index": url, ...}), ...]}
|
|
63
|
-
|
|
64
|
-
Notes:
|
|
65
|
-
- Uses pandas extraction for values (same filtering as extract_sheet_cells).
|
|
66
|
-
- Collects hyperlinks via openpyxl (requires read_only=False because border maps/hyperlinks need full objects).
|
|
67
|
-
- Links are mapped by column index string (e.g., "0") to hyperlink.target.
|
|
68
|
-
"""
|
|
69
|
-
cell_rows = extract_sheet_cells(file_path)
|
|
70
|
-
wb = load_workbook(file_path, data_only=True, read_only=False)
|
|
71
|
-
links_by_sheet: Dict[str, Dict[int, Dict[str, str]]] = {}
|
|
72
|
-
for ws in wb.worksheets:
|
|
73
|
-
sheet_links: Dict[int, Dict[str, str]] = {}
|
|
74
|
-
for row in ws.iter_rows():
|
|
75
|
-
for cell in row:
|
|
76
|
-
link = getattr(cell, "hyperlink", None)
|
|
77
|
-
target = getattr(link, "target", None) if link else None
|
|
78
|
-
if not target:
|
|
79
|
-
continue
|
|
80
|
-
col_str = str(cell.col_idx - 1) # zero-based to align with extract_sheet_cells
|
|
81
|
-
sheet_links.setdefault(cell.row, {})[col_str] = target
|
|
82
|
-
links_by_sheet[ws.title] = sheet_links
|
|
83
|
-
|
|
84
|
-
merged: Dict[str, List[CellRow]] = {}
|
|
85
|
-
for sheet_name, rows in cell_rows.items():
|
|
86
|
-
sheet_links = links_by_sheet.get(sheet_name, {})
|
|
87
|
-
merged_rows: List[CellRow] = []
|
|
88
|
-
for row in rows:
|
|
89
|
-
links = sheet_links.get(row.r, {})
|
|
90
|
-
merged_rows.append(CellRow(r=row.r, c=row.c, links=links or None))
|
|
91
|
-
merged[sheet_name] = merged_rows
|
|
92
|
-
wb.close()
|
|
93
|
-
return merged
|
|
36
|
+
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
+
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
+
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
+
result: Dict[str, List[CellRow]] = {}
|
|
40
|
+
for sheet_name, df in dfs.items():
|
|
41
|
+
df = df.fillna("")
|
|
42
|
+
rows: List[CellRow] = []
|
|
43
|
+
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
+
filtered: Dict[str, int | float | str] = {}
|
|
45
|
+
for j, v in enumerate(row):
|
|
46
|
+
s = "" if v is None else str(v)
|
|
47
|
+
if s.strip() == "":
|
|
48
|
+
continue
|
|
49
|
+
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
+
if not filtered:
|
|
51
|
+
continue
|
|
52
|
+
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
+
result[sheet_name] = rows
|
|
54
|
+
return result
|
|
94
55
|
|
|
95
56
|
|
|
96
57
|
def shrink_to_content(
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
3
|
import logging
|
|
4
|
-
from typing import Dict, List, Optional
|
|
4
|
+
from typing import Dict, List, Optional
|
|
5
5
|
|
|
6
6
|
import xlwings as xw
|
|
7
7
|
|
|
@@ -166,7 +166,7 @@ def parse_series_formula(formula: str) -> Optional[Dict[str, Optional[str]]]:
|
|
|
166
166
|
}
|
|
167
167
|
|
|
168
168
|
|
|
169
|
-
def get_charts(sheet: xw.Sheet
|
|
169
|
+
def get_charts(sheet: xw.Sheet) -> List[Chart]:
|
|
170
170
|
"""Parse charts in a sheet into Chart models; failed charts carry an error field."""
|
|
171
171
|
charts: List[Chart] = []
|
|
172
172
|
for ch in sheet.charts:
|
|
@@ -182,14 +182,6 @@ def get_charts(sheet: xw.Sheet, mode: Literal["light", "standard", "verbose"] =
|
|
|
182
182
|
chart_type_label = XL_CHART_TYPE_MAP.get(
|
|
183
183
|
chart_type_num, f"unknown_{chart_type_num}"
|
|
184
184
|
)
|
|
185
|
-
chart_width: Optional[int] = None
|
|
186
|
-
chart_height: Optional[int] = None
|
|
187
|
-
try:
|
|
188
|
-
chart_width = int(ch.width)
|
|
189
|
-
chart_height = int(ch.height)
|
|
190
|
-
except Exception:
|
|
191
|
-
chart_width = None
|
|
192
|
-
chart_height = None
|
|
193
185
|
|
|
194
186
|
for s in chart_com.SeriesCollection():
|
|
195
187
|
parsed = parse_series_formula(getattr(s, "Formula", ""))
|
|
@@ -228,8 +220,6 @@ def get_charts(sheet: xw.Sheet, mode: Literal["light", "standard", "verbose"] =
|
|
|
228
220
|
title=title,
|
|
229
221
|
y_axis_title=y_axis_title,
|
|
230
222
|
y_axis_range=y_axis_range, # type: ignore
|
|
231
|
-
w=chart_width,
|
|
232
|
-
h=chart_height,
|
|
233
223
|
series=series_list,
|
|
234
224
|
l=int(ch.left),
|
|
235
225
|
t=int(ch.top),
|