exstruct 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {exstruct-0.2.1 → exstruct-0.2.2}/PKG-INFO +70 -41
- {exstruct-0.2.1 → exstruct-0.2.2}/README.md +68 -40
- {exstruct-0.2.1 → exstruct-0.2.2}/pyproject.toml +9 -2
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/__init__.py +2 -1
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/core/cells.py +58 -19
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/core/integrate.py +15 -7
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/engine.py +25 -2
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/models/__init__.py +1 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/render/__init__.py +5 -3
- {exstruct-0.2.1 → exstruct-0.2.2}/LICENSE +0 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/cli/main.py +0 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/core/__init__.py +0 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/core/charts.py +0 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/core/shapes.py +0 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/io/__init__.py +0 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/models/maps.py +0 -0
- {exstruct-0.2.1 → exstruct-0.2.2}/src/exstruct/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: exstruct
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
|
|
5
5
|
Keywords: excel,structure,data,exstruct
|
|
6
6
|
Author: harumiWeb
|
|
@@ -41,6 +41,7 @@ Requires-Dist: pydantic>=2.12.5
|
|
|
41
41
|
Requires-Dist: scipy>=1.16.3
|
|
42
42
|
Requires-Dist: xlwings>=0.33.16
|
|
43
43
|
Requires-Dist: pypdfium2>=5.1.0 ; extra == 'render'
|
|
44
|
+
Requires-Dist: pillow>=12.0.0 ; extra == 'render'
|
|
44
45
|
Requires-Dist: python-toon>=0.1.3 ; extra == 'toon'
|
|
45
46
|
Requires-Dist: pyyaml>=6.0.3 ; extra == 'yaml'
|
|
46
47
|
Requires-Python: >=3.11
|
|
@@ -55,14 +56,16 @@ Description-Content-Type: text/markdown
|
|
|
55
56
|
|
|
56
57
|
# ExStruct — Excel Structured Extraction Engine
|
|
57
58
|
|
|
58
|
-
|
|
59
|
+
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml)
|
|
59
60
|
|
|
60
|
-
ExStruct
|
|
61
|
+

|
|
62
|
+
|
|
63
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
61
64
|
|
|
62
65
|
## Features
|
|
63
66
|
|
|
64
67
|
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
65
|
-
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
68
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height). Verbose also emits cell hyperlinks.
|
|
66
69
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
67
70
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
68
71
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -78,51 +81,61 @@ Optional extras:
|
|
|
78
81
|
|
|
79
82
|
- YAML: `pip install pyyaml`
|
|
80
83
|
- TOON: `pip install python-toon`
|
|
81
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2`
|
|
84
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
85
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
82
86
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
exstruct input.xlsx
|
|
90
|
-
exstruct input.xlsx
|
|
91
|
-
exstruct input.xlsx --
|
|
92
|
-
exstruct input.xlsx --
|
|
93
|
-
|
|
87
|
+
Platform note:
|
|
88
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
89
|
+
|
|
90
|
+
## Quick Start (CLI)
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
94
|
+
exstruct input.xlsx -o out.json --pretty # pretty JSON to a file
|
|
95
|
+
exstruct input.xlsx --format yaml # YAML (needs pyyaml)
|
|
96
|
+
exstruct input.xlsx --format toon # TOON (needs python-toon)
|
|
97
|
+
exstruct input.xlsx --sheets-dir sheets/ # split per sheet in chosen format
|
|
98
|
+
exstruct input.xlsx --mode light # cells + table candidates only
|
|
99
|
+
exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
|
|
100
|
+
```
|
|
94
101
|
|
|
95
102
|
## Quick Start (Python)
|
|
96
103
|
|
|
97
104
|
```python
|
|
98
105
|
from pathlib import Path
|
|
99
|
-
from exstruct import extract, export, set_table_detection_params
|
|
100
|
-
|
|
101
|
-
# Tune table detection (optional)
|
|
102
|
-
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
103
|
-
|
|
104
|
-
# Extract with modes: "light", "standard", "verbose"
|
|
105
|
-
wb = extract("input.xlsx", mode="standard")
|
|
106
|
-
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
107
|
-
|
|
108
|
-
# Model helpers: iterate, index, and serialize directly from the models
|
|
109
|
-
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
110
|
-
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
111
|
-
print(name, len(sheet.rows))
|
|
112
|
-
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
113
|
-
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
114
|
-
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
115
|
-
|
|
106
|
+
from exstruct import extract, export, set_table_detection_params
|
|
107
|
+
|
|
108
|
+
# Tune table detection (optional)
|
|
109
|
+
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
110
|
+
|
|
111
|
+
# Extract with modes: "light", "standard", "verbose"
|
|
112
|
+
wb = extract("input.xlsx", mode="standard")
|
|
113
|
+
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
114
|
+
|
|
115
|
+
# Model helpers: iterate, index, and serialize directly from the models
|
|
116
|
+
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
117
|
+
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
118
|
+
print(name, len(sheet.rows))
|
|
119
|
+
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
120
|
+
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
121
|
+
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
122
|
+
|
|
116
123
|
# ExStructEngine: per-instance options for extraction/output
|
|
117
124
|
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
118
125
|
|
|
119
126
|
engine = ExStructEngine(
|
|
120
|
-
options=StructOptions(mode="
|
|
127
|
+
options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
|
|
121
128
|
output=OutputOptions(include_shapes=False, pretty=True),
|
|
122
129
|
)
|
|
123
130
|
wb2 = engine.extract("input.xlsx")
|
|
124
131
|
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
125
|
-
|
|
132
|
+
|
|
133
|
+
# Enable hyperlinks in other modes
|
|
134
|
+
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
135
|
+
with_links = engine_links.extract("input.xlsx")
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
126
139
|
|
|
127
140
|
## Table Detection Tuning
|
|
128
141
|
|
|
@@ -139,11 +152,11 @@ set_table_detection_params(
|
|
|
139
152
|
|
|
140
153
|
Use higher thresholds to reduce false positives; lower them if true tables are missed.
|
|
141
154
|
|
|
142
|
-
## Output Modes
|
|
143
|
-
|
|
144
|
-
- **light**: cells + table candidates (no COM needed).
|
|
145
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
146
|
-
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
155
|
+
## Output Modes
|
|
156
|
+
|
|
157
|
+
- **light**: cells + table candidates (no COM needed).
|
|
158
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates. Hyperlinks are off unless `include_cell_links=True`.
|
|
159
|
+
- **verbose**: all shapes (with width/height), charts, table candidates, and cell hyperlinks.
|
|
147
160
|
|
|
148
161
|
## Error Handling / Fallbacks
|
|
149
162
|
|
|
@@ -170,7 +183,8 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
170
183
|
- Flowchart built only with shapes
|
|
171
184
|
|
|
172
185
|
(Screenshot below is the actual sample Excel sheet)
|
|
173
|
-
|
|
186
|
+

|
|
187
|
+
Sample workbook: `sample/sample.xlsx`
|
|
174
188
|
|
|
175
189
|
### 1. Input: Excel Sheet Overview
|
|
176
190
|
|
|
@@ -364,3 +378,18 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
364
378
|
## Documentation
|
|
365
379
|
|
|
366
380
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
381
|
+
# Engine option cheat sheet
|
|
382
|
+
|
|
383
|
+
| Option class | Field | Meaning |
|
|
384
|
+
| -------------- | ------------------- | ------- |
|
|
385
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
386
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
387
|
+
| | include_cell_links | Include cell hyperlinks in `rows[*].links` (None -> auto: verbose=True, others=False) |
|
|
388
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
389
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
390
|
+
| | include_rows | Include rows (False to drop) |
|
|
391
|
+
| | include_shapes | Include shapes |
|
|
392
|
+
| | include_charts | Include charts |
|
|
393
|
+
| | include_tables | Include table_candidates |
|
|
394
|
+
| | sheets_dir | Optional directory for per-sheet exports |
|
|
395
|
+
| | stream | Default stream when output_path is None |
|
|
@@ -1,13 +1,15 @@
|
|
|
1
1
|
# ExStruct — Excel Structured Extraction Engine
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml)
|
|
4
4
|
|
|
5
|
-
ExStruct
|
|
5
|
+

|
|
6
|
+
|
|
7
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
6
8
|
|
|
7
9
|
## Features
|
|
8
10
|
|
|
9
11
|
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
10
|
-
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
12
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height). Verbose also emits cell hyperlinks.
|
|
11
13
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
12
14
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
13
15
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -23,51 +25,61 @@ Optional extras:
|
|
|
23
25
|
|
|
24
26
|
- YAML: `pip install pyyaml`
|
|
25
27
|
- TOON: `pip install python-toon`
|
|
26
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2`
|
|
28
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
29
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
27
30
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
exstruct input.xlsx
|
|
35
|
-
exstruct input.xlsx
|
|
36
|
-
exstruct input.xlsx --
|
|
37
|
-
exstruct input.xlsx --
|
|
38
|
-
|
|
31
|
+
Platform note:
|
|
32
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
33
|
+
|
|
34
|
+
## Quick Start (CLI)
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
38
|
+
exstruct input.xlsx -o out.json --pretty # pretty JSON to a file
|
|
39
|
+
exstruct input.xlsx --format yaml # YAML (needs pyyaml)
|
|
40
|
+
exstruct input.xlsx --format toon # TOON (needs python-toon)
|
|
41
|
+
exstruct input.xlsx --sheets-dir sheets/ # split per sheet in chosen format
|
|
42
|
+
exstruct input.xlsx --mode light # cells + table candidates only
|
|
43
|
+
exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
|
|
44
|
+
```
|
|
39
45
|
|
|
40
46
|
## Quick Start (Python)
|
|
41
47
|
|
|
42
48
|
```python
|
|
43
49
|
from pathlib import Path
|
|
44
|
-
from exstruct import extract, export, set_table_detection_params
|
|
45
|
-
|
|
46
|
-
# Tune table detection (optional)
|
|
47
|
-
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
48
|
-
|
|
49
|
-
# Extract with modes: "light", "standard", "verbose"
|
|
50
|
-
wb = extract("input.xlsx", mode="standard")
|
|
51
|
-
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
52
|
-
|
|
53
|
-
# Model helpers: iterate, index, and serialize directly from the models
|
|
54
|
-
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
55
|
-
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
56
|
-
print(name, len(sheet.rows))
|
|
57
|
-
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
58
|
-
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
59
|
-
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
60
|
-
|
|
50
|
+
from exstruct import extract, export, set_table_detection_params
|
|
51
|
+
|
|
52
|
+
# Tune table detection (optional)
|
|
53
|
+
set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
|
|
54
|
+
|
|
55
|
+
# Extract with modes: "light", "standard", "verbose"
|
|
56
|
+
wb = extract("input.xlsx", mode="standard")
|
|
57
|
+
export(wb, Path("out.json"), pretty=False) # compact JSON
|
|
58
|
+
|
|
59
|
+
# Model helpers: iterate, index, and serialize directly from the models
|
|
60
|
+
first_sheet = wb["Sheet1"] # __getitem__ access
|
|
61
|
+
for name, sheet in wb: # __iter__ yields (name, SheetData)
|
|
62
|
+
print(name, len(sheet.rows))
|
|
63
|
+
wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
64
|
+
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
65
|
+
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
66
|
+
|
|
61
67
|
# ExStructEngine: per-instance options for extraction/output
|
|
62
68
|
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
63
69
|
|
|
64
70
|
engine = ExStructEngine(
|
|
65
|
-
options=StructOptions(mode="
|
|
71
|
+
options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
|
|
66
72
|
output=OutputOptions(include_shapes=False, pretty=True),
|
|
67
73
|
)
|
|
68
74
|
wb2 = engine.extract("input.xlsx")
|
|
69
75
|
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
70
|
-
|
|
76
|
+
|
|
77
|
+
# Enable hyperlinks in other modes
|
|
78
|
+
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
79
|
+
with_links = engine_links.extract("input.xlsx")
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
71
83
|
|
|
72
84
|
## Table Detection Tuning
|
|
73
85
|
|
|
@@ -84,11 +96,11 @@ set_table_detection_params(
|
|
|
84
96
|
|
|
85
97
|
Use higher thresholds to reduce false positives; lower them if true tables are missed.
|
|
86
98
|
|
|
87
|
-
## Output Modes
|
|
88
|
-
|
|
89
|
-
- **light**: cells + table candidates (no COM needed).
|
|
90
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
91
|
-
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
99
|
+
## Output Modes
|
|
100
|
+
|
|
101
|
+
- **light**: cells + table candidates (no COM needed).
|
|
102
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates. Hyperlinks are off unless `include_cell_links=True`.
|
|
103
|
+
- **verbose**: all shapes (with width/height), charts, table candidates, and cell hyperlinks.
|
|
92
104
|
|
|
93
105
|
## Error Handling / Fallbacks
|
|
94
106
|
|
|
@@ -115,7 +127,8 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
115
127
|
- Flowchart built only with shapes
|
|
116
128
|
|
|
117
129
|
(Screenshot below is the actual sample Excel sheet)
|
|
118
|
-
|
|
130
|
+

|
|
131
|
+
Sample workbook: `sample/sample.xlsx`
|
|
119
132
|
|
|
120
133
|
### 1. Input: Excel Sheet Overview
|
|
121
134
|
|
|
@@ -309,3 +322,18 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
309
322
|
## Documentation
|
|
310
323
|
|
|
311
324
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
325
|
+
# Engine option cheat sheet
|
|
326
|
+
|
|
327
|
+
| Option class | Field | Meaning |
|
|
328
|
+
| -------------- | ------------------- | ------- |
|
|
329
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
330
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
331
|
+
| | include_cell_links | Include cell hyperlinks in `rows[*].links` (None -> auto: verbose=True, others=False) |
|
|
332
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
333
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
334
|
+
| | include_rows | Include rows (False to drop) |
|
|
335
|
+
| | include_shapes | Include shapes |
|
|
336
|
+
| | include_charts | Include charts |
|
|
337
|
+
| | include_tables | Include table_candidates |
|
|
338
|
+
| | sheets_dir | Optional directory for per-sheet exports |
|
|
339
|
+
| | stream | Default stream when output_path is None |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "exstruct"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.2"
|
|
4
4
|
description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = { file = "LICENSE" }
|
|
@@ -33,7 +33,7 @@ dev = [
|
|
|
33
33
|
[project.optional-dependencies]
|
|
34
34
|
yaml = ["pyyaml>=6.0.3"]
|
|
35
35
|
toon = ["python-toon>=0.1.3"]
|
|
36
|
-
render = ["pypdfium2>=5.1.0"]
|
|
36
|
+
render = ["pypdfium2>=5.1.0", "Pillow>=12.0.0"]
|
|
37
37
|
|
|
38
38
|
[project.scripts]
|
|
39
39
|
exstruct = "exstruct.cli.main:main"
|
|
@@ -43,3 +43,10 @@ Homepage = "https://harumiweb.github.io/exstruct/"
|
|
|
43
43
|
Repository = "https://github.com/harumiWeb/exstruct"
|
|
44
44
|
Issues = "https://github.com/harumiWeb/exstruct/issues"
|
|
45
45
|
Documentation = "https://harumiweb.github.io/exstruct/"
|
|
46
|
+
|
|
47
|
+
[tool.coverage.run]
|
|
48
|
+
omit = [
|
|
49
|
+
"tests/*",
|
|
50
|
+
"*/test_*.py",
|
|
51
|
+
"*/gen_py/*",
|
|
52
|
+
]
|
|
@@ -36,7 +36,8 @@ ExtractionMode = Literal["light", "standard", "verbose"]
|
|
|
36
36
|
|
|
37
37
|
def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> WorkbookData:
|
|
38
38
|
"""Extract workbook semantic structure and return WorkbookData."""
|
|
39
|
-
|
|
39
|
+
include_links = True if mode == "verbose" else False
|
|
40
|
+
engine = ExStructEngine(options=StructOptions(mode=mode, include_cell_links=include_links))
|
|
40
41
|
return engine.extract(file_path, mode=mode)
|
|
41
42
|
|
|
42
43
|
|
|
@@ -33,25 +33,64 @@ def warn_once(key: str, message: str) -> None:
|
|
|
33
33
|
_warned_keys.add(key)
|
|
34
34
|
|
|
35
35
|
|
|
36
|
-
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
-
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
-
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
-
result: Dict[str, List[CellRow]] = {}
|
|
40
|
-
for sheet_name, df in dfs.items():
|
|
41
|
-
df = df.fillna("")
|
|
42
|
-
rows: List[CellRow] = []
|
|
43
|
-
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
-
filtered: Dict[str, int | float | str] = {}
|
|
45
|
-
for j, v in enumerate(row):
|
|
46
|
-
s = "" if v is None else str(v)
|
|
47
|
-
if s.strip() == "":
|
|
48
|
-
continue
|
|
49
|
-
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
-
if not filtered:
|
|
51
|
-
continue
|
|
52
|
-
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
-
result[sheet_name] = rows
|
|
54
|
-
return result
|
|
36
|
+
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
+
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
+
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
+
result: Dict[str, List[CellRow]] = {}
|
|
40
|
+
for sheet_name, df in dfs.items():
|
|
41
|
+
df = df.fillna("")
|
|
42
|
+
rows: List[CellRow] = []
|
|
43
|
+
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
+
filtered: Dict[str, int | float | str] = {}
|
|
45
|
+
for j, v in enumerate(row):
|
|
46
|
+
s = "" if v is None else str(v)
|
|
47
|
+
if s.strip() == "":
|
|
48
|
+
continue
|
|
49
|
+
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
+
if not filtered:
|
|
51
|
+
continue
|
|
52
|
+
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
+
result[sheet_name] = rows
|
|
54
|
+
return result
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def extract_sheet_cells_with_links(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
58
|
+
"""
|
|
59
|
+
Extract cells and hyperlinks per sheet.
|
|
60
|
+
|
|
61
|
+
Returns:
|
|
62
|
+
{sheet_name: [CellRow(r=..., c=..., links={"col_index": url, ...}), ...]}
|
|
63
|
+
|
|
64
|
+
Notes:
|
|
65
|
+
- Uses pandas extraction for values (same filtering as extract_sheet_cells).
|
|
66
|
+
- Collects hyperlinks via openpyxl (requires read_only=False because border maps/hyperlinks need full objects).
|
|
67
|
+
- Links are mapped by column index string (e.g., "0") to hyperlink.target.
|
|
68
|
+
"""
|
|
69
|
+
cell_rows = extract_sheet_cells(file_path)
|
|
70
|
+
wb = load_workbook(file_path, data_only=True, read_only=False)
|
|
71
|
+
links_by_sheet: Dict[str, Dict[int, Dict[str, str]]] = {}
|
|
72
|
+
for ws in wb.worksheets:
|
|
73
|
+
sheet_links: Dict[int, Dict[str, str]] = {}
|
|
74
|
+
for row in ws.iter_rows():
|
|
75
|
+
for cell in row:
|
|
76
|
+
link = getattr(cell, "hyperlink", None)
|
|
77
|
+
target = getattr(link, "target", None) if link else None
|
|
78
|
+
if not target:
|
|
79
|
+
continue
|
|
80
|
+
col_str = str(cell.col_idx - 1) # zero-based to align with extract_sheet_cells
|
|
81
|
+
sheet_links.setdefault(cell.row, {})[col_str] = target
|
|
82
|
+
links_by_sheet[ws.title] = sheet_links
|
|
83
|
+
|
|
84
|
+
merged: Dict[str, List[CellRow]] = {}
|
|
85
|
+
for sheet_name, rows in cell_rows.items():
|
|
86
|
+
sheet_links = links_by_sheet.get(sheet_name, {})
|
|
87
|
+
merged_rows: List[CellRow] = []
|
|
88
|
+
for row in rows:
|
|
89
|
+
links = sheet_links.get(row.r, {})
|
|
90
|
+
merged_rows.append(CellRow(r=row.r, c=row.c, links=links or None))
|
|
91
|
+
merged[sheet_name] = merged_rows
|
|
92
|
+
wb.close()
|
|
93
|
+
return merged
|
|
55
94
|
|
|
56
95
|
|
|
57
96
|
def shrink_to_content(
|
|
@@ -4,15 +4,17 @@ from pathlib import Path
|
|
|
4
4
|
from typing import Dict, List, Literal
|
|
5
5
|
|
|
6
6
|
import logging
|
|
7
|
+
import os
|
|
7
8
|
|
|
8
9
|
import xlwings as xw
|
|
9
10
|
|
|
10
11
|
from ..models import CellRow, Shape, SheetData, WorkbookData
|
|
11
|
-
from .cells import (
|
|
12
|
-
detect_tables,
|
|
13
|
-
detect_tables_openpyxl,
|
|
14
|
-
extract_sheet_cells,
|
|
15
|
-
|
|
12
|
+
from .cells import (
|
|
13
|
+
detect_tables,
|
|
14
|
+
detect_tables_openpyxl,
|
|
15
|
+
extract_sheet_cells,
|
|
16
|
+
extract_sheet_cells_with_links,
|
|
17
|
+
)
|
|
16
18
|
from .charts import get_charts
|
|
17
19
|
from .shapes import get_shapes_with_position
|
|
18
20
|
|
|
@@ -74,13 +76,16 @@ def integrate_sheet_content(
|
|
|
74
76
|
|
|
75
77
|
|
|
76
78
|
def extract_workbook(
|
|
77
|
-
file_path: Path,
|
|
79
|
+
file_path: Path,
|
|
80
|
+
mode: Literal["light", "standard", "verbose"] = "standard",
|
|
81
|
+
*,
|
|
82
|
+
include_cell_links: bool = False,
|
|
78
83
|
) -> WorkbookData:
|
|
79
84
|
"""Extract workbook and return WorkbookData; fallback to cells+tables if Excel COM is unavailable."""
|
|
80
85
|
if mode not in _ALLOWED_MODES:
|
|
81
86
|
raise ValueError(f"Unsupported mode: {mode}")
|
|
82
87
|
|
|
83
|
-
cell_data = extract_sheet_cells(file_path)
|
|
88
|
+
cell_data = extract_sheet_cells_with_links(file_path) if include_cell_links else extract_sheet_cells(file_path)
|
|
84
89
|
|
|
85
90
|
def _cells_and_tables_only(reason: str) -> WorkbookData:
|
|
86
91
|
sheets: Dict[str, SheetData] = {}
|
|
@@ -104,6 +109,9 @@ def extract_workbook(
|
|
|
104
109
|
if mode == "light":
|
|
105
110
|
return _cells_and_tables_only("Light mode selected.")
|
|
106
111
|
|
|
112
|
+
if os.getenv("SKIP_COM_TESTS"):
|
|
113
|
+
return _cells_and_tables_only("SKIP_COM_TESTS is set; skipping COM/xlwings access.")
|
|
114
|
+
|
|
107
115
|
try:
|
|
108
116
|
wb, close_app = _open_workbook(file_path)
|
|
109
117
|
except Exception as e:
|
|
@@ -3,8 +3,10 @@ from __future__ import annotations
|
|
|
3
3
|
from dataclasses import dataclass
|
|
4
4
|
from pathlib import Path
|
|
5
5
|
from typing import Literal, Optional, TextIO
|
|
6
|
+
from contextlib import contextmanager
|
|
6
7
|
|
|
7
8
|
from .core.integrate import extract_workbook
|
|
9
|
+
from .core import cells as _cells
|
|
8
10
|
from .core.cells import set_table_detection_params
|
|
9
11
|
from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets, serialize_workbook
|
|
10
12
|
from .models import SheetData, WorkbookData
|
|
@@ -30,6 +32,7 @@ class StructOptions:
|
|
|
30
32
|
|
|
31
33
|
mode: ExtractionMode = "standard"
|
|
32
34
|
table_params: Optional[dict] = None # forwarded to set_table_detection_params if provided
|
|
35
|
+
include_cell_links: Optional[bool] = None # None → auto: verbose=True, others=False
|
|
33
36
|
|
|
34
37
|
|
|
35
38
|
@dataclass(frozen=True)
|
|
@@ -96,6 +99,21 @@ class ExStructEngine:
|
|
|
96
99
|
if self.options.table_params:
|
|
97
100
|
set_table_detection_params(**self.options.table_params)
|
|
98
101
|
|
|
102
|
+
@contextmanager
|
|
103
|
+
def _table_params_scope(self):
|
|
104
|
+
"""
|
|
105
|
+
Temporarily apply table_params and restore previous global config afterward.
|
|
106
|
+
"""
|
|
107
|
+
if not self.options.table_params:
|
|
108
|
+
yield
|
|
109
|
+
return
|
|
110
|
+
prev = dict(_cells._DETECTION_CONFIG) # type: ignore[attr-defined]
|
|
111
|
+
set_table_detection_params(**self.options.table_params)
|
|
112
|
+
try:
|
|
113
|
+
yield
|
|
114
|
+
finally:
|
|
115
|
+
set_table_detection_params(**prev)
|
|
116
|
+
|
|
99
117
|
def _filter_sheet(self, sheet: SheetData) -> SheetData:
|
|
100
118
|
return SheetData(
|
|
101
119
|
rows=sheet.rows if self.output.include_rows else [],
|
|
@@ -116,8 +134,13 @@ class ExStructEngine:
|
|
|
116
134
|
chosen_mode = mode or self.options.mode
|
|
117
135
|
if chosen_mode not in ("light", "standard", "verbose"):
|
|
118
136
|
raise ValueError(f"Unsupported mode: {chosen_mode}")
|
|
119
|
-
|
|
120
|
-
|
|
137
|
+
include_links = (
|
|
138
|
+
self.options.include_cell_links
|
|
139
|
+
if self.options.include_cell_links is not None
|
|
140
|
+
else chosen_mode == "verbose"
|
|
141
|
+
)
|
|
142
|
+
with self._table_params_scope():
|
|
143
|
+
return extract_workbook(Path(file_path), mode=chosen_mode, include_cell_links=include_links)
|
|
121
144
|
|
|
122
145
|
def serialize(
|
|
123
146
|
self,
|
|
@@ -54,12 +54,14 @@ def _require_pdfium():
|
|
|
54
54
|
import pypdfium2 as pdfium # type: ignore
|
|
55
55
|
except ImportError as e:
|
|
56
56
|
raise RuntimeError(
|
|
57
|
-
"Image rendering requires pypdfium2. Install it via `pip install pypdfium2` or add the 'render' extra."
|
|
57
|
+
"Image rendering requires pypdfium2. Install it via `pip install pypdfium2 pillow` or add the 'render' extra."
|
|
58
58
|
) from e
|
|
59
59
|
return pdfium
|
|
60
60
|
|
|
61
61
|
|
|
62
|
-
def export_sheet_images(
|
|
62
|
+
def export_sheet_images(
|
|
63
|
+
excel_path: Path, output_dir: Path, dpi: int = 144
|
|
64
|
+
) -> List[Path]:
|
|
63
65
|
"""Export each sheet as PNG (via PDF then pypdfium2 rasterization) and return paths in sheet order."""
|
|
64
66
|
pdfium = _require_pdfium()
|
|
65
67
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
@@ -73,7 +75,7 @@ def export_sheet_images(excel_path: Path, output_dir: Path, dpi: int = 144) -> L
|
|
|
73
75
|
with pdfium.PdfDocument(str(tmp_pdf)) as pdf: # type: ignore
|
|
74
76
|
for i, sheet_name in enumerate(sheet_names):
|
|
75
77
|
page = pdf[i]
|
|
76
|
-
bitmap = page.render(scale=scale)
|
|
78
|
+
bitmap = page.render(scale=scale) # type: ignore
|
|
77
79
|
pil_image = bitmap.to_pil() # type: ignore
|
|
78
80
|
safe_name = _sanitize_sheet_filename(sheet_name)
|
|
79
81
|
img_path = output_dir / f"{i+1:02d}_{safe_name}.png"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|