exstruct 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {exstruct-0.2.0 → exstruct-0.2.2}/PKG-INFO +76 -37
- {exstruct-0.2.0 → exstruct-0.2.2}/README.md +74 -36
- {exstruct-0.2.0 → exstruct-0.2.2}/pyproject.toml +9 -2
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/__init__.py +25 -46
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/core/cells.py +58 -19
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/core/integrate.py +15 -7
- exstruct-0.2.2/src/exstruct/engine.py +251 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/models/__init__.py +10 -1
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/render/__init__.py +5 -3
- {exstruct-0.2.0 → exstruct-0.2.2}/LICENSE +0 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/cli/main.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/core/__init__.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/core/charts.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/core/shapes.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/io/__init__.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/models/maps.py +0 -0
- {exstruct-0.2.0 → exstruct-0.2.2}/src/exstruct/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: exstruct
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
|
|
5
5
|
Keywords: excel,structure,data,exstruct
|
|
6
6
|
Author: harumiWeb
|
|
@@ -41,6 +41,7 @@ Requires-Dist: pydantic>=2.12.5
|
|
|
41
41
|
Requires-Dist: scipy>=1.16.3
|
|
42
42
|
Requires-Dist: xlwings>=0.33.16
|
|
43
43
|
Requires-Dist: pypdfium2>=5.1.0 ; extra == 'render'
|
|
44
|
+
Requires-Dist: pillow>=12.0.0 ; extra == 'render'
|
|
44
45
|
Requires-Dist: python-toon>=0.1.3 ; extra == 'toon'
|
|
45
46
|
Requires-Dist: pyyaml>=6.0.3 ; extra == 'yaml'
|
|
46
47
|
Requires-Python: >=3.11
|
|
@@ -55,14 +56,16 @@ Description-Content-Type: text/markdown
|
|
|
55
56
|
|
|
56
57
|
# ExStruct — Excel Structured Extraction Engine
|
|
57
58
|
|
|
58
|
-
|
|
59
|
+
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml)
|
|
59
60
|
|
|
60
|
-
ExStruct
|
|
61
|
+

|
|
62
|
+
|
|
63
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
61
64
|
|
|
62
65
|
## Features
|
|
63
66
|
|
|
64
67
|
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
65
|
-
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
68
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height). Verbose also emits cell hyperlinks.
|
|
66
69
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
67
70
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
68
71
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -78,41 +81,61 @@ Optional extras:
|
|
|
78
81
|
|
|
79
82
|
- YAML: `pip install pyyaml`
|
|
80
83
|
- TOON: `pip install python-toon`
|
|
81
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2`
|
|
84
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
85
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
82
86
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
exstruct input.xlsx
|
|
90
|
-
exstruct input.xlsx
|
|
91
|
-
exstruct input.xlsx --
|
|
92
|
-
exstruct input.xlsx --
|
|
93
|
-
|
|
87
|
+
Platform note:
|
|
88
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
89
|
+
|
|
90
|
+
## Quick Start (CLI)
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
94
|
+
exstruct input.xlsx -o out.json --pretty # pretty JSON to a file
|
|
95
|
+
exstruct input.xlsx --format yaml # YAML (needs pyyaml)
|
|
96
|
+
exstruct input.xlsx --format toon # TOON (needs python-toon)
|
|
97
|
+
exstruct input.xlsx --sheets-dir sheets/ # split per sheet in chosen format
|
|
98
|
+
exstruct input.xlsx --mode light # cells + table candidates only
|
|
99
|
+
exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
|
|
100
|
+
```
|
|
94
101
|
|
|
95
102
|
## Quick Start (Python)
|
|
96
103
|
|
|
97
104
|
```python
|
|
98
105
|
from pathlib import Path
|
|
99
|
-
from exstruct import extract, export, set_table_detection_params
|
|
100
|
-
|
|
101
|
-
# Tune table detection (optional)
|
|
102
|
-
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
106
|
+
from exstruct import extract, export, set_table_detection_params
|
|
107
|
+
|
|
108
|
+
# Tune table detection (optional)
|
|
109
|
+
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
110
|
+
|
|
111
|
+
# Extract with modes: "light", "standard", "verbose"
|
|
112
|
+
wb = extract("input.xlsx", mode="standard")
|
|
113
|
+
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
114
|
+
|
|
115
|
+
# Model helpers: iterate, index, and serialize directly from the models
|
|
116
|
+
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
117
|
+
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
118
|
+
print(name, len(sheet.rows))
|
|
119
|
+
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
120
|
+
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
121
|
+
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
122
|
+
|
|
123
|
+
# ExStructEngine: per-instance options for extraction/output
|
|
124
|
+
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
103
125
|
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
126
|
+
engine = ExStructEngine(
|
|
127
|
+
options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
|
|
128
|
+
output=OutputOptions(include_shapes=False, pretty=True),
|
|
129
|
+
)
|
|
130
|
+
wb2 = engine.extract("input.xlsx")
|
|
131
|
+
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
107
132
|
|
|
108
|
-
#
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
115
|
-
```
|
|
133
|
+
# Enable hyperlinks in other modes
|
|
134
|
+
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
135
|
+
with_links = engine_links.extract("input.xlsx")
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
116
139
|
|
|
117
140
|
## Table Detection Tuning
|
|
118
141
|
|
|
@@ -129,11 +152,11 @@ set_table_detection_params(
|
|
|
129
152
|
|
|
130
153
|
Use higher thresholds to reduce false positives; lower them if true tables are missed.
|
|
131
154
|
|
|
132
|
-
## Output Modes
|
|
133
|
-
|
|
134
|
-
- **light**: cells + table candidates (no COM needed).
|
|
135
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
136
|
-
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
155
|
+
## Output Modes
|
|
156
|
+
|
|
157
|
+
- **light**: cells + table candidates (no COM needed).
|
|
158
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates. Hyperlinks are off unless `include_cell_links=True`.
|
|
159
|
+
- **verbose**: all shapes (with width/height), charts, table candidates, and cell hyperlinks.
|
|
137
160
|
|
|
138
161
|
## Error Handling / Fallbacks
|
|
139
162
|
|
|
@@ -160,7 +183,8 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
160
183
|
- Flowchart built only with shapes
|
|
161
184
|
|
|
162
185
|
(Screenshot below is the actual sample Excel sheet)
|
|
163
|
-
|
|
186
|
+

|
|
187
|
+
Sample workbook: `sample/sample.xlsx`
|
|
164
188
|
|
|
165
189
|
### 1. Input: Excel Sheet Overview
|
|
166
190
|
|
|
@@ -354,3 +378,18 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
354
378
|
## Documentation
|
|
355
379
|
|
|
356
380
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
381
|
+
# Engine option cheat sheet
|
|
382
|
+
|
|
383
|
+
| Option class | Field | Meaning |
|
|
384
|
+
| -------------- | ------------------- | ------- |
|
|
385
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
386
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
387
|
+
| | include_cell_links | Include cell hyperlinks in `rows[*].links` (None -> auto: verbose=True, others=False) |
|
|
388
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
389
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
390
|
+
| | include_rows | Include rows (False to drop) |
|
|
391
|
+
| | include_shapes | Include shapes |
|
|
392
|
+
| | include_charts | Include charts |
|
|
393
|
+
| | include_tables | Include table_candidates |
|
|
394
|
+
| | sheets_dir | Optional directory for per-sheet exports |
|
|
395
|
+
| | stream | Default stream when output_path is None |
|
|
@@ -1,13 +1,15 @@
|
|
|
1
1
|
# ExStruct — Excel Structured Extraction Engine
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml)
|
|
4
4
|
|
|
5
|
-
ExStruct
|
|
5
|
+

|
|
6
|
+
|
|
7
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
6
8
|
|
|
7
9
|
## Features
|
|
8
10
|
|
|
9
11
|
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
10
|
-
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
12
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height). Verbose also emits cell hyperlinks.
|
|
11
13
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
12
14
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
13
15
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -23,41 +25,61 @@ Optional extras:
|
|
|
23
25
|
|
|
24
26
|
- YAML: `pip install pyyaml`
|
|
25
27
|
- TOON: `pip install python-toon`
|
|
26
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2`
|
|
28
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
29
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
27
30
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
exstruct input.xlsx
|
|
35
|
-
exstruct input.xlsx
|
|
36
|
-
exstruct input.xlsx --
|
|
37
|
-
exstruct input.xlsx --
|
|
38
|
-
|
|
31
|
+
Platform note:
|
|
32
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
33
|
+
|
|
34
|
+
## Quick Start (CLI)
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
38
|
+
exstruct input.xlsx -o out.json --pretty # pretty JSON to a file
|
|
39
|
+
exstruct input.xlsx --format yaml # YAML (needs pyyaml)
|
|
40
|
+
exstruct input.xlsx --format toon # TOON (needs python-toon)
|
|
41
|
+
exstruct input.xlsx --sheets-dir sheets/ # split per sheet in chosen format
|
|
42
|
+
exstruct input.xlsx --mode light # cells + table candidates only
|
|
43
|
+
exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
|
|
44
|
+
```
|
|
39
45
|
|
|
40
46
|
## Quick Start (Python)
|
|
41
47
|
|
|
42
48
|
```python
|
|
43
49
|
from pathlib import Path
|
|
44
|
-
from exstruct import extract, export, set_table_detection_params
|
|
45
|
-
|
|
46
|
-
# Tune table detection (optional)
|
|
47
|
-
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
50
|
+
from exstruct import extract, export, set_table_detection_params
|
|
51
|
+
|
|
52
|
+
# Tune table detection (optional)
|
|
53
|
+
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
54
|
+
|
|
55
|
+
# Extract with modes: "light", "standard", "verbose"
|
|
56
|
+
wb = extract("input.xlsx", mode="standard")
|
|
57
|
+
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
58
|
+
|
|
59
|
+
# Model helpers: iterate, index, and serialize directly from the models
|
|
60
|
+
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
61
|
+
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
62
|
+
print(name, len(sheet.rows))
|
|
63
|
+
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
64
|
+
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
65
|
+
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
66
|
+
|
|
67
|
+
# ExStructEngine: per-instance options for extraction/output
|
|
68
|
+
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
48
69
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
70
|
+
engine = ExStructEngine(
|
|
71
|
+
options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
|
|
72
|
+
output=OutputOptions(include_shapes=False, pretty=True),
|
|
73
|
+
)
|
|
74
|
+
wb2 = engine.extract("input.xlsx")
|
|
75
|
+
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
52
76
|
|
|
53
|
-
#
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
60
|
-
```
|
|
77
|
+
# Enable hyperlinks in other modes
|
|
78
|
+
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
79
|
+
with_links = engine_links.extract("input.xlsx")
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
61
83
|
|
|
62
84
|
## Table Detection Tuning
|
|
63
85
|
|
|
@@ -74,11 +96,11 @@ set_table_detection_params(
|
|
|
74
96
|
|
|
75
97
|
Use higher thresholds to reduce false positives; lower them if true tables are missed.
|
|
76
98
|
|
|
77
|
-
## Output Modes
|
|
78
|
-
|
|
79
|
-
- **light**: cells + table candidates (no COM needed).
|
|
80
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
81
|
-
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
99
|
+
## Output Modes
|
|
100
|
+
|
|
101
|
+
- **light**: cells + table candidates (no COM needed).
|
|
102
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates. Hyperlinks are off unless `include_cell_links=True`.
|
|
103
|
+
- **verbose**: all shapes (with width/height), charts, table candidates, and cell hyperlinks.
|
|
82
104
|
|
|
83
105
|
## Error Handling / Fallbacks
|
|
84
106
|
|
|
@@ -105,7 +127,8 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
105
127
|
- Flowchart built only with shapes
|
|
106
128
|
|
|
107
129
|
(Screenshot below is the actual sample Excel sheet)
|
|
108
|
-
|
|
130
|
+

|
|
131
|
+
Sample workbook: `sample/sample.xlsx`
|
|
109
132
|
|
|
110
133
|
### 1. Input: Excel Sheet Overview
|
|
111
134
|
|
|
@@ -299,3 +322,18 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
299
322
|
## Documentation
|
|
300
323
|
|
|
301
324
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
325
|
+
# Engine option cheat sheet
|
|
326
|
+
|
|
327
|
+
| Option class | Field | Meaning |
|
|
328
|
+
| -------------- | ------------------- | ------- |
|
|
329
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
330
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
331
|
+
| | include_cell_links | Include cell hyperlinks in `rows[*].links` (None -> auto: verbose=True, others=False) |
|
|
332
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
333
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
334
|
+
| | include_rows | Include rows (False to drop) |
|
|
335
|
+
| | include_shapes | Include shapes |
|
|
336
|
+
| | include_charts | Include charts |
|
|
337
|
+
| | include_tables | Include table_candidates |
|
|
338
|
+
| | sheets_dir | Optional directory for per-sheet exports |
|
|
339
|
+
| | stream | Default stream when output_path is None |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "exstruct"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.2"
|
|
4
4
|
description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = { file = "LICENSE" }
|
|
@@ -33,7 +33,7 @@ dev = [
|
|
|
33
33
|
[project.optional-dependencies]
|
|
34
34
|
yaml = ["pyyaml>=6.0.3"]
|
|
35
35
|
toon = ["python-toon>=0.1.3"]
|
|
36
|
-
render = ["pypdfium2>=5.1.0"]
|
|
36
|
+
render = ["pypdfium2>=5.1.0", "Pillow>=12.0.0"]
|
|
37
37
|
|
|
38
38
|
[project.scripts]
|
|
39
39
|
exstruct = "exstruct.cli.main:main"
|
|
@@ -43,3 +43,10 @@ Homepage = "https://harumiweb.github.io/exstruct/"
|
|
|
43
43
|
Repository = "https://github.com/harumiWeb/exstruct"
|
|
44
44
|
Issues = "https://github.com/harumiWeb/exstruct/issues"
|
|
45
45
|
Documentation = "https://harumiweb.github.io/exstruct/"
|
|
46
|
+
|
|
47
|
+
[tool.coverage.run]
|
|
48
|
+
omit = [
|
|
49
|
+
"tests/*",
|
|
50
|
+
"*/test_*.py",
|
|
51
|
+
"*/gen_py/*",
|
|
52
|
+
]
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
-
from pathlib import Path
|
|
3
|
+
from pathlib import Path
|
|
4
4
|
from typing import Literal, Optional, TextIO
|
|
5
5
|
|
|
6
6
|
from .core.integrate import extract_workbook
|
|
7
7
|
from .core.cells import set_table_detection_params
|
|
8
|
+
from .engine import ExStructEngine, OutputOptions, StructOptions
|
|
8
9
|
from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
|
|
9
10
|
from .models import CellRow, Chart, ChartSeries, Shape, SheetData, WorkbookData
|
|
10
11
|
from .render import export_pdf, export_sheet_images
|
|
@@ -24,6 +25,9 @@ __all__ = [
|
|
|
24
25
|
"SheetData",
|
|
25
26
|
"WorkbookData",
|
|
26
27
|
"set_table_detection_params",
|
|
28
|
+
"ExStructEngine",
|
|
29
|
+
"StructOptions",
|
|
30
|
+
"OutputOptions",
|
|
27
31
|
]
|
|
28
32
|
|
|
29
33
|
|
|
@@ -32,9 +36,9 @@ ExtractionMode = Literal["light", "standard", "verbose"]
|
|
|
32
36
|
|
|
33
37
|
def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> WorkbookData:
|
|
34
38
|
"""Extract workbook semantic structure and return WorkbookData."""
|
|
35
|
-
if mode
|
|
36
|
-
|
|
37
|
-
return
|
|
39
|
+
include_links = True if mode == "verbose" else False
|
|
40
|
+
engine = ExStructEngine(options=StructOptions(mode=mode, include_cell_links=include_links))
|
|
41
|
+
return engine.extract(file_path, mode=mode)
|
|
38
42
|
|
|
39
43
|
|
|
40
44
|
def export(
|
|
@@ -99,45 +103,20 @@ def process_excel(
|
|
|
99
103
|
- If output_path is None, writes the serialized workbook to stdout (or provided stream).
|
|
100
104
|
- If sheets_dir is given, also writes per-sheet files into that directory.
|
|
101
105
|
"""
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
else:
|
|
120
|
-
if target_stream is None:
|
|
121
|
-
import sys
|
|
122
|
-
|
|
123
|
-
target_stream = sys.stdout
|
|
124
|
-
target_stream.write(text)
|
|
125
|
-
if not text.endswith("\n"):
|
|
126
|
-
target_stream.write("\n")
|
|
127
|
-
|
|
128
|
-
if sheets_dir is not None:
|
|
129
|
-
save_sheets(
|
|
130
|
-
workbook_model,
|
|
131
|
-
sheets_dir,
|
|
132
|
-
fmt=out_fmt,
|
|
133
|
-
pretty=pretty,
|
|
134
|
-
indent=indent,
|
|
135
|
-
)
|
|
136
|
-
|
|
137
|
-
if pdf or image:
|
|
138
|
-
base_target = output_path or file_path.with_suffix(_suffix_for(out_fmt))
|
|
139
|
-
pdf_path = base_target.with_suffix(".pdf")
|
|
140
|
-
export_pdf(file_path, pdf_path)
|
|
141
|
-
if image:
|
|
142
|
-
images_dir = pdf_path.parent / f"{pdf_path.stem}_images"
|
|
143
|
-
export_sheet_images(file_path, images_dir, dpi=dpi)
|
|
106
|
+
engine = ExStructEngine(
|
|
107
|
+
options=StructOptions(mode=mode),
|
|
108
|
+
output=OutputOptions(fmt=out_fmt, pretty=pretty, indent=indent, sheets_dir=sheets_dir, stream=stream),
|
|
109
|
+
)
|
|
110
|
+
engine.process(
|
|
111
|
+
file_path=file_path,
|
|
112
|
+
output_path=output_path,
|
|
113
|
+
out_fmt=out_fmt,
|
|
114
|
+
image=image,
|
|
115
|
+
pdf=pdf,
|
|
116
|
+
dpi=dpi,
|
|
117
|
+
mode=mode,
|
|
118
|
+
pretty=pretty,
|
|
119
|
+
indent=indent,
|
|
120
|
+
sheets_dir=sheets_dir,
|
|
121
|
+
stream=stream,
|
|
122
|
+
)
|
|
@@ -33,25 +33,64 @@ def warn_once(key: str, message: str) -> None:
|
|
|
33
33
|
_warned_keys.add(key)
|
|
34
34
|
|
|
35
35
|
|
|
36
|
-
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
-
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
-
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
-
result: Dict[str, List[CellRow]] = {}
|
|
40
|
-
for sheet_name, df in dfs.items():
|
|
41
|
-
df = df.fillna("")
|
|
42
|
-
rows: List[CellRow] = []
|
|
43
|
-
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
-
filtered: Dict[str, int | float | str] = {}
|
|
45
|
-
for j, v in enumerate(row):
|
|
46
|
-
s = "" if v is None else str(v)
|
|
47
|
-
if s.strip() == "":
|
|
48
|
-
continue
|
|
49
|
-
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
-
if not filtered:
|
|
51
|
-
continue
|
|
52
|
-
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
-
result[sheet_name] = rows
|
|
54
|
-
return result
|
|
36
|
+
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
+
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
+
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
+
result: Dict[str, List[CellRow]] = {}
|
|
40
|
+
for sheet_name, df in dfs.items():
|
|
41
|
+
df = df.fillna("")
|
|
42
|
+
rows: List[CellRow] = []
|
|
43
|
+
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
+
filtered: Dict[str, int | float | str] = {}
|
|
45
|
+
for j, v in enumerate(row):
|
|
46
|
+
s = "" if v is None else str(v)
|
|
47
|
+
if s.strip() == "":
|
|
48
|
+
continue
|
|
49
|
+
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
+
if not filtered:
|
|
51
|
+
continue
|
|
52
|
+
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
+
result[sheet_name] = rows
|
|
54
|
+
return result
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def extract_sheet_cells_with_links(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
58
|
+
"""
|
|
59
|
+
Extract cells and hyperlinks per sheet.
|
|
60
|
+
|
|
61
|
+
Returns:
|
|
62
|
+
{sheet_name: [CellRow(r=..., c=..., links={"col_index": url, ...}), ...]}
|
|
63
|
+
|
|
64
|
+
Notes:
|
|
65
|
+
- Uses pandas extraction for values (same filtering as extract_sheet_cells).
|
|
66
|
+
- Collects hyperlinks via openpyxl (requires read_only=False because border maps/hyperlinks need full objects).
|
|
67
|
+
- Links are mapped by column index string (e.g., "0") to hyperlink.target.
|
|
68
|
+
"""
|
|
69
|
+
cell_rows = extract_sheet_cells(file_path)
|
|
70
|
+
wb = load_workbook(file_path, data_only=True, read_only=False)
|
|
71
|
+
links_by_sheet: Dict[str, Dict[int, Dict[str, str]]] = {}
|
|
72
|
+
for ws in wb.worksheets:
|
|
73
|
+
sheet_links: Dict[int, Dict[str, str]] = {}
|
|
74
|
+
for row in ws.iter_rows():
|
|
75
|
+
for cell in row:
|
|
76
|
+
link = getattr(cell, "hyperlink", None)
|
|
77
|
+
target = getattr(link, "target", None) if link else None
|
|
78
|
+
if not target:
|
|
79
|
+
continue
|
|
80
|
+
col_str = str(cell.col_idx - 1) # zero-based to align with extract_sheet_cells
|
|
81
|
+
sheet_links.setdefault(cell.row, {})[col_str] = target
|
|
82
|
+
links_by_sheet[ws.title] = sheet_links
|
|
83
|
+
|
|
84
|
+
merged: Dict[str, List[CellRow]] = {}
|
|
85
|
+
for sheet_name, rows in cell_rows.items():
|
|
86
|
+
sheet_links = links_by_sheet.get(sheet_name, {})
|
|
87
|
+
merged_rows: List[CellRow] = []
|
|
88
|
+
for row in rows:
|
|
89
|
+
links = sheet_links.get(row.r, {})
|
|
90
|
+
merged_rows.append(CellRow(r=row.r, c=row.c, links=links or None))
|
|
91
|
+
merged[sheet_name] = merged_rows
|
|
92
|
+
wb.close()
|
|
93
|
+
return merged
|
|
55
94
|
|
|
56
95
|
|
|
57
96
|
def shrink_to_content(
|
|
@@ -4,15 +4,17 @@ from pathlib import Path
|
|
|
4
4
|
from typing import Dict, List, Literal
|
|
5
5
|
|
|
6
6
|
import logging
|
|
7
|
+
import os
|
|
7
8
|
|
|
8
9
|
import xlwings as xw
|
|
9
10
|
|
|
10
11
|
from ..models import CellRow, Shape, SheetData, WorkbookData
|
|
11
|
-
from .cells import (
|
|
12
|
-
detect_tables,
|
|
13
|
-
detect_tables_openpyxl,
|
|
14
|
-
extract_sheet_cells,
|
|
15
|
-
|
|
12
|
+
from .cells import (
|
|
13
|
+
detect_tables,
|
|
14
|
+
detect_tables_openpyxl,
|
|
15
|
+
extract_sheet_cells,
|
|
16
|
+
extract_sheet_cells_with_links,
|
|
17
|
+
)
|
|
16
18
|
from .charts import get_charts
|
|
17
19
|
from .shapes import get_shapes_with_position
|
|
18
20
|
|
|
@@ -74,13 +76,16 @@ def integrate_sheet_content(
|
|
|
74
76
|
|
|
75
77
|
|
|
76
78
|
def extract_workbook(
|
|
77
|
-
file_path: Path,
|
|
79
|
+
file_path: Path,
|
|
80
|
+
mode: Literal["light", "standard", "verbose"] = "standard",
|
|
81
|
+
*,
|
|
82
|
+
include_cell_links: bool = False,
|
|
78
83
|
) -> WorkbookData:
|
|
79
84
|
"""Extract workbook and return WorkbookData; fallback to cells+tables if Excel COM is unavailable."""
|
|
80
85
|
if mode not in _ALLOWED_MODES:
|
|
81
86
|
raise ValueError(f"Unsupported mode: {mode}")
|
|
82
87
|
|
|
83
|
-
cell_data = extract_sheet_cells(file_path)
|
|
88
|
+
cell_data = extract_sheet_cells_with_links(file_path) if include_cell_links else extract_sheet_cells(file_path)
|
|
84
89
|
|
|
85
90
|
def _cells_and_tables_only(reason: str) -> WorkbookData:
|
|
86
91
|
sheets: Dict[str, SheetData] = {}
|
|
@@ -104,6 +109,9 @@ def extract_workbook(
|
|
|
104
109
|
if mode == "light":
|
|
105
110
|
return _cells_and_tables_only("Light mode selected.")
|
|
106
111
|
|
|
112
|
+
if os.getenv("SKIP_COM_TESTS"):
|
|
113
|
+
return _cells_and_tables_only("SKIP_COM_TESTS is set; skipping COM/xlwings access.")
|
|
114
|
+
|
|
107
115
|
try:
|
|
108
116
|
wb, close_app = _open_workbook(file_path)
|
|
109
117
|
except Exception as e:
|
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Literal, Optional, TextIO
|
|
6
|
+
from contextlib import contextmanager
|
|
7
|
+
|
|
8
|
+
from .core.integrate import extract_workbook
|
|
9
|
+
from .core import cells as _cells
|
|
10
|
+
from .core.cells import set_table_detection_params
|
|
11
|
+
from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
|
|
12
|
+
from .models import SheetData, WorkbookData
|
|
13
|
+
from .render import export_pdf, export_sheet_images
|
|
14
|
+
|
|
15
|
+
ExtractionMode = Literal["light", "standard", "verbose"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class StructOptions:
|
|
20
|
+
"""
|
|
21
|
+
Extraction-time options for ExStructEngine.
|
|
22
|
+
|
|
23
|
+
Attributes:
|
|
24
|
+
mode: Extraction mode. One of "light", "standard", "verbose".
|
|
25
|
+
- light: cells + table candidates only (no COM, shapes/charts empty)
|
|
26
|
+
- standard: texted shapes + arrows + charts (if COM available)
|
|
27
|
+
- verbose: all shapes (width/height), charts, table candidates
|
|
28
|
+
table_params: Optional dict passed to `set_table_detection_params(**table_params)`
|
|
29
|
+
before extraction. Use this to tweak table detection heuristics
|
|
30
|
+
per engine instance without touching global state.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
mode: ExtractionMode = "standard"
|
|
34
|
+
table_params: Optional[dict] = None # forwarded to set_table_detection_params if provided
|
|
35
|
+
include_cell_links: Optional[bool] = None # None → auto: verbose=True, others=False
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class OutputOptions:
|
|
40
|
+
"""
|
|
41
|
+
Output-time options for ExStructEngine.
|
|
42
|
+
|
|
43
|
+
Attributes:
|
|
44
|
+
fmt: Default export format. One of "json", "yaml", "yml", "toon".
|
|
45
|
+
pretty: Whether to pretty-print JSON; default False (compact).
|
|
46
|
+
indent: Explicit indent size. If None and pretty=True, indent=2 for JSON.
|
|
47
|
+
include_rows: Include SheetData.rows in output (set False to drop).
|
|
48
|
+
include_shapes: Include SheetData.shapes in output.
|
|
49
|
+
include_charts: Include SheetData.charts in output.
|
|
50
|
+
include_tables: Include SheetData.table_candidates in output.
|
|
51
|
+
sheets_dir: Optional directory to write per-sheet files (in the chosen fmt).
|
|
52
|
+
stream: Optional default stream for stdout output when output_path is None.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
fmt: Literal["json", "yaml", "yml", "toon"] = "json"
|
|
56
|
+
pretty: bool = False
|
|
57
|
+
indent: int | None = None
|
|
58
|
+
include_rows: bool = True
|
|
59
|
+
include_shapes: bool = True
|
|
60
|
+
include_charts: bool = True
|
|
61
|
+
include_tables: bool = True
|
|
62
|
+
sheets_dir: Path | None = None
|
|
63
|
+
stream: TextIO | None = None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class ExStructEngine:
|
|
67
|
+
"""
|
|
68
|
+
Configurable engine for ExStruct extraction and export.
|
|
69
|
+
|
|
70
|
+
Instances are immutable; override options per call if needed.
|
|
71
|
+
|
|
72
|
+
Key behaviors:
|
|
73
|
+
- Uses StructOptions for extraction defaults (mode, table_params).
|
|
74
|
+
- Uses OutputOptions for serialization defaults (fmt, pretty/indent, include* filters).
|
|
75
|
+
- Methods:
|
|
76
|
+
extract(path, mode=None) -> WorkbookData
|
|
77
|
+
serialize(workbook, fmt=None, pretty=None, indent=None) -> str
|
|
78
|
+
export(workbook, output_path=None, fmt=None, pretty=None, indent=None,
|
|
79
|
+
sheets_dir=None, stream=None) -> None
|
|
80
|
+
process(file_path, output_path=None, out_fmt=None, image=False, pdf=False,
|
|
81
|
+
dpi=72, mode=None, pretty=None, indent=None, sheets_dir=None,
|
|
82
|
+
stream=None) -> None
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
def __init__(
|
|
86
|
+
self,
|
|
87
|
+
options: StructOptions | None = None,
|
|
88
|
+
output: OutputOptions | None = None,
|
|
89
|
+
) -> None:
|
|
90
|
+
self.options = options or StructOptions()
|
|
91
|
+
self.output = output or OutputOptions()
|
|
92
|
+
|
|
93
|
+
@staticmethod
|
|
94
|
+
def from_defaults() -> "ExStructEngine":
|
|
95
|
+
"""Factory to create an engine with default options."""
|
|
96
|
+
return ExStructEngine()
|
|
97
|
+
|
|
98
|
+
def _apply_table_params(self) -> None:
|
|
99
|
+
if self.options.table_params:
|
|
100
|
+
set_table_detection_params(**self.options.table_params)
|
|
101
|
+
|
|
102
|
+
@contextmanager
|
|
103
|
+
def _table_params_scope(self):
|
|
104
|
+
"""
|
|
105
|
+
Temporarily apply table_params and restore previous global config afterward.
|
|
106
|
+
"""
|
|
107
|
+
if not self.options.table_params:
|
|
108
|
+
yield
|
|
109
|
+
return
|
|
110
|
+
prev = dict(_cells._DETECTION_CONFIG) # type: ignore[attr-defined]
|
|
111
|
+
set_table_detection_params(**self.options.table_params)
|
|
112
|
+
try:
|
|
113
|
+
yield
|
|
114
|
+
finally:
|
|
115
|
+
set_table_detection_params(**prev)
|
|
116
|
+
|
|
117
|
+
def _filter_sheet(self, sheet: SheetData) -> SheetData:
|
|
118
|
+
return SheetData(
|
|
119
|
+
rows=sheet.rows if self.output.include_rows else [],
|
|
120
|
+
shapes=sheet.shapes if self.output.include_shapes else [],
|
|
121
|
+
charts=sheet.charts if self.output.include_charts else [],
|
|
122
|
+
table_candidates=sheet.table_candidates if self.output.include_tables else [],
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
def _filter_workbook(self, wb: WorkbookData) -> WorkbookData:
|
|
126
|
+
filtered = {
|
|
127
|
+
name: self._filter_sheet(sheet)
|
|
128
|
+
for name, sheet in wb.sheets.items()
|
|
129
|
+
}
|
|
130
|
+
return WorkbookData(book_name=wb.book_name, sheets=filtered)
|
|
131
|
+
|
|
132
|
+
def extract(self, file_path: str | Path, *, mode: ExtractionMode | None = None) -> WorkbookData:
|
|
133
|
+
"""Extract workbook semantic structure with the configured options."""
|
|
134
|
+
chosen_mode = mode or self.options.mode
|
|
135
|
+
if chosen_mode not in ("light", "standard", "verbose"):
|
|
136
|
+
raise ValueError(f"Unsupported mode: {chosen_mode}")
|
|
137
|
+
include_links = (
|
|
138
|
+
self.options.include_cell_links
|
|
139
|
+
if self.options.include_cell_links is not None
|
|
140
|
+
else chosen_mode == "verbose"
|
|
141
|
+
)
|
|
142
|
+
with self._table_params_scope():
|
|
143
|
+
return extract_workbook(Path(file_path), mode=chosen_mode, include_cell_links=include_links)
|
|
144
|
+
|
|
145
|
+
def serialize(
|
|
146
|
+
self,
|
|
147
|
+
data: WorkbookData,
|
|
148
|
+
*,
|
|
149
|
+
fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
|
|
150
|
+
pretty: Optional[bool] = None,
|
|
151
|
+
indent: int | None = None,
|
|
152
|
+
) -> str:
|
|
153
|
+
"""
|
|
154
|
+
Serialize WorkbookData using configured output defaults, applying include/exclude filters.
|
|
155
|
+
"""
|
|
156
|
+
filtered = self._filter_workbook(data)
|
|
157
|
+
use_fmt = (fmt or self.output.fmt)
|
|
158
|
+
use_pretty = self.output.pretty if pretty is None else pretty
|
|
159
|
+
use_indent = self.output.indent if indent is None else indent
|
|
160
|
+
return serialize_workbook(filtered, fmt=use_fmt, pretty=use_pretty, indent=use_indent)
|
|
161
|
+
|
|
162
|
+
def export(
|
|
163
|
+
self,
|
|
164
|
+
data: WorkbookData,
|
|
165
|
+
output_path: Path | None = None,
|
|
166
|
+
*,
|
|
167
|
+
fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
|
|
168
|
+
pretty: Optional[bool] = None,
|
|
169
|
+
indent: int | None = None,
|
|
170
|
+
sheets_dir: Path | None = None,
|
|
171
|
+
stream: TextIO | None = None,
|
|
172
|
+
) -> None:
|
|
173
|
+
"""
|
|
174
|
+
Write WorkbookData to disk or stdout (when output_path is None).
|
|
175
|
+
Applies include/exclude filters before serialization.
|
|
176
|
+
"""
|
|
177
|
+
text = self.serialize(data, fmt=fmt, pretty=pretty, indent=indent)
|
|
178
|
+
target_stream = stream or self.output.stream
|
|
179
|
+
chosen_fmt = (fmt or self.output.fmt)
|
|
180
|
+
chosen_sheets_dir = sheets_dir if sheets_dir is not None else self.output.sheets_dir
|
|
181
|
+
|
|
182
|
+
def _suffix_for(fmt_val: str) -> str:
|
|
183
|
+
if fmt_val in ("yaml", "yml"):
|
|
184
|
+
return ".yaml"
|
|
185
|
+
if fmt_val == "toon":
|
|
186
|
+
return ".toon"
|
|
187
|
+
if fmt_val == "json":
|
|
188
|
+
return ".json"
|
|
189
|
+
raise ValueError(f"Unsupported export format: {fmt_val}")
|
|
190
|
+
|
|
191
|
+
if output_path is not None:
|
|
192
|
+
output_path.write_text(text, encoding="utf-8")
|
|
193
|
+
else:
|
|
194
|
+
import sys
|
|
195
|
+
|
|
196
|
+
stream_target = target_stream or sys.stdout
|
|
197
|
+
stream_target.write(text)
|
|
198
|
+
if not text.endswith("\n"):
|
|
199
|
+
stream_target.write("\n")
|
|
200
|
+
|
|
201
|
+
if chosen_sheets_dir is not None:
|
|
202
|
+
filtered = self._filter_workbook(data)
|
|
203
|
+
save_sheets(
|
|
204
|
+
filtered,
|
|
205
|
+
chosen_sheets_dir,
|
|
206
|
+
fmt=chosen_fmt,
|
|
207
|
+
pretty=self.output.pretty if pretty is None else pretty,
|
|
208
|
+
indent=self.output.indent if indent is None else indent,
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
return None
|
|
212
|
+
|
|
213
|
+
def process(
|
|
214
|
+
self,
|
|
215
|
+
file_path: Path,
|
|
216
|
+
output_path: Path | None = None,
|
|
217
|
+
*,
|
|
218
|
+
out_fmt: Optional[str] = None,
|
|
219
|
+
image: bool = False,
|
|
220
|
+
pdf: bool = False,
|
|
221
|
+
dpi: int = 72,
|
|
222
|
+
mode: ExtractionMode | None = None,
|
|
223
|
+
pretty: bool | None = None,
|
|
224
|
+
indent: int | None = None,
|
|
225
|
+
sheets_dir: Path | None = None,
|
|
226
|
+
stream: TextIO | None = None,
|
|
227
|
+
) -> None:
|
|
228
|
+
"""
|
|
229
|
+
Convenience wrapper: extract, export (to file or stdout), and optionally render PDF/PNG.
|
|
230
|
+
"""
|
|
231
|
+
wb = self.extract(file_path, mode=mode)
|
|
232
|
+
chosen_fmt = out_fmt or self.output.fmt
|
|
233
|
+
self.export(
|
|
234
|
+
wb,
|
|
235
|
+
output_path=output_path,
|
|
236
|
+
fmt=chosen_fmt, # type: ignore[arg-type]
|
|
237
|
+
pretty=pretty,
|
|
238
|
+
indent=indent,
|
|
239
|
+
sheets_dir=sheets_dir,
|
|
240
|
+
stream=stream,
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
if pdf or image:
|
|
244
|
+
base_target = output_path or file_path.with_suffix(
|
|
245
|
+
".yaml" if chosen_fmt in ("yaml", "yml") else ".toon" if chosen_fmt == "toon" else ".json"
|
|
246
|
+
)
|
|
247
|
+
pdf_path = base_target.with_suffix(".pdf")
|
|
248
|
+
export_pdf(file_path, pdf_path)
|
|
249
|
+
if image:
|
|
250
|
+
images_dir = pdf_path.parent / f"{pdf_path.stem}_images"
|
|
251
|
+
export_sheet_images(file_path, images_dir, dpi=dpi)
|
|
@@ -23,6 +23,7 @@ class Shape(BaseModel):
|
|
|
23
23
|
class CellRow(BaseModel):
|
|
24
24
|
r: int
|
|
25
25
|
c: Dict[str, int | float | str]
|
|
26
|
+
links: Optional[Dict[str, str]] = None
|
|
26
27
|
|
|
27
28
|
|
|
28
29
|
class ChartSeries(BaseModel):
|
|
@@ -163,9 +164,17 @@ class WorkbookData(BaseModel):
|
|
|
163
164
|
case _:
|
|
164
165
|
raise ValueError(f"Unsupported export format: {fmt}")
|
|
165
166
|
return dest
|
|
167
|
+
|
|
168
|
+
def __getitem__(self, sheet_name: str) -> SheetData:
|
|
169
|
+
"""Return the SheetData for the given sheet name."""
|
|
170
|
+
return self.sheets[sheet_name]
|
|
171
|
+
|
|
172
|
+
def __iter__(self):
|
|
173
|
+
"""Iterate over (sheet_name, SheetData) pairs in order."""
|
|
174
|
+
return iter(self.sheets.items())
|
|
166
175
|
|
|
167
176
|
def __iter__(self):
|
|
168
177
|
return iter(self.sheets.items())
|
|
169
178
|
|
|
170
179
|
def __getitem__(self, sheet_name: str) -> SheetData:
|
|
171
|
-
return self.sheets[sheet_name]
|
|
180
|
+
return self.sheets[sheet_name]
|
|
@@ -54,12 +54,14 @@ def _require_pdfium():
|
|
|
54
54
|
import pypdfium2 as pdfium # type: ignore
|
|
55
55
|
except ImportError as e:
|
|
56
56
|
raise RuntimeError(
|
|
57
|
-
"Image rendering requires pypdfium2. Install it via `pip install pypdfium2` or add the 'render' extra."
|
|
57
|
+
"Image rendering requires pypdfium2. Install it via `pip install pypdfium2 pillow` or add the 'render' extra."
|
|
58
58
|
) from e
|
|
59
59
|
return pdfium
|
|
60
60
|
|
|
61
61
|
|
|
62
|
-
def export_sheet_images(
|
|
62
|
+
def export_sheet_images(
|
|
63
|
+
excel_path: Path, output_dir: Path, dpi: int = 144
|
|
64
|
+
) -> List[Path]:
|
|
63
65
|
"""Export each sheet as PNG (via PDF then pypdfium2 rasterization) and return paths in sheet order."""
|
|
64
66
|
pdfium = _require_pdfium()
|
|
65
67
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
@@ -73,7 +75,7 @@ def export_sheet_images(excel_path: Path, output_dir: Path, dpi: int = 144) -> L
|
|
|
73
75
|
with pdfium.PdfDocument(str(tmp_pdf)) as pdf: # type: ignore
|
|
74
76
|
for i, sheet_name in enumerate(sheet_names):
|
|
75
77
|
page = pdf[i]
|
|
76
|
-
bitmap = page.render(scale=scale)
|
|
78
|
+
bitmap = page.render(scale=scale) # type: ignore
|
|
77
79
|
pil_image = bitmap.to_pil() # type: ignore
|
|
78
80
|
safe_name = _sanitize_sheet_filename(sheet_name)
|
|
79
81
|
img_path = output_dir / f"{i+1:02d}_{safe_name}.png"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|