exstruct 0.2.2__tar.gz → 0.2.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {exstruct-0.2.2 → exstruct-0.2.11}/PKG-INFO +38 -43
- {exstruct-0.2.2 → exstruct-0.2.11}/README.md +37 -42
- {exstruct-0.2.2 → exstruct-0.2.11}/pyproject.toml +1 -1
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/__init__.py +1 -2
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/core/cells.py +19 -58
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/core/integrate.py +7 -15
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/engine.py +1 -7
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/models/__init__.py +0 -1
- {exstruct-0.2.2 → exstruct-0.2.11}/LICENSE +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/cli/main.py +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/core/__init__.py +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/core/charts.py +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/core/shapes.py +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/io/__init__.py +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/models/maps.py +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/py.typed +0 -0
- {exstruct-0.2.2 → exstruct-0.2.11}/src/exstruct/render/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: exstruct
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.11
|
|
4
4
|
Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
|
|
5
5
|
Keywords: excel,structure,data,exstruct
|
|
6
6
|
Author: harumiWeb
|
|
@@ -56,16 +56,16 @@ Description-Content-Type: text/markdown
|
|
|
56
56
|
|
|
57
57
|
# ExStruct — Excel Structured Extraction Engine
|
|
58
58
|
|
|
59
|
-
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/ci.yml)
|
|
60
60
|
|
|
61
61
|

|
|
62
62
|
|
|
63
|
-
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts
|
|
63
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
64
64
|
|
|
65
65
|
## Features
|
|
66
66
|
|
|
67
67
|
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
68
|
-
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
68
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
69
69
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
70
70
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
71
71
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -77,17 +77,17 @@ ExStruct reads Excel workbooks and outputs structured data (tables, shapes, char
|
|
|
77
77
|
pip install exstruct
|
|
78
78
|
```
|
|
79
79
|
|
|
80
|
-
Optional extras:
|
|
81
|
-
|
|
82
|
-
- YAML: `pip install pyyaml`
|
|
83
|
-
- TOON: `pip install python-toon`
|
|
84
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
85
|
-
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
86
|
-
|
|
87
|
-
Platform note:
|
|
88
|
-
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
89
|
-
|
|
90
|
-
## Quick Start (CLI)
|
|
80
|
+
Optional extras:
|
|
81
|
+
|
|
82
|
+
- YAML: `pip install pyyaml`
|
|
83
|
+
- TOON: `pip install python-toon`
|
|
84
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
85
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
86
|
+
|
|
87
|
+
Platform note:
|
|
88
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
89
|
+
|
|
90
|
+
## Quick Start (CLI)
|
|
91
91
|
|
|
92
92
|
```bash
|
|
93
93
|
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
@@ -120,19 +120,15 @@ wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
|
120
120
|
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
121
121
|
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
122
122
|
|
|
123
|
-
# ExStructEngine: per-instance options for extraction/output
|
|
124
|
-
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
125
|
-
|
|
126
|
-
engine = ExStructEngine(
|
|
127
|
-
options=StructOptions(mode="
|
|
128
|
-
output=OutputOptions(include_shapes=False, pretty=True),
|
|
129
|
-
)
|
|
130
|
-
wb2 = engine.extract("input.xlsx")
|
|
131
|
-
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
132
|
-
|
|
133
|
-
# Enable hyperlinks in other modes
|
|
134
|
-
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
135
|
-
with_links = engine_links.extract("input.xlsx")
|
|
123
|
+
# ExStructEngine: per-instance options for extraction/output
|
|
124
|
+
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
125
|
+
|
|
126
|
+
engine = ExStructEngine(
|
|
127
|
+
options=StructOptions(mode="standard"),
|
|
128
|
+
output=OutputOptions(include_shapes=False, pretty=True),
|
|
129
|
+
)
|
|
130
|
+
wb2 = engine.extract("input.xlsx")
|
|
131
|
+
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
136
132
|
```
|
|
137
133
|
|
|
138
134
|
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
@@ -152,11 +148,11 @@ set_table_detection_params(
|
|
|
152
148
|
|
|
153
149
|
Use higher thresholds to reduce false positives; lower them if true tables are missed.
|
|
154
150
|
|
|
155
|
-
## Output Modes
|
|
156
|
-
|
|
157
|
-
- **light**: cells + table candidates (no COM needed).
|
|
158
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
159
|
-
- **verbose**: all shapes (with width/height), charts, table candidates
|
|
151
|
+
## Output Modes
|
|
152
|
+
|
|
153
|
+
- **light**: cells + table candidates (no COM needed).
|
|
154
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
155
|
+
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
160
156
|
|
|
161
157
|
## Error Handling / Fallbacks
|
|
162
158
|
|
|
@@ -380,16 +376,15 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
380
376
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
381
377
|
# Engine option cheat sheet
|
|
382
378
|
|
|
383
|
-
| Option class | Field | Meaning |
|
|
384
|
-
| -------------- | ------------------- | ------- |
|
|
385
|
-
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
386
|
-
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
387
|
-
|
|
|
388
|
-
|
|
|
389
|
-
| |
|
|
390
|
-
| |
|
|
391
|
-
| |
|
|
392
|
-
| | include_charts | Include charts |
|
|
379
|
+
| Option class | Field | Meaning |
|
|
380
|
+
| -------------- | ------------------- | ------- |
|
|
381
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
382
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
383
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
384
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
385
|
+
| | include_rows | Include rows (False to drop) |
|
|
386
|
+
| | include_shapes | Include shapes |
|
|
387
|
+
| | include_charts | Include charts |
|
|
393
388
|
| | include_tables | Include table_candidates |
|
|
394
389
|
| | sheets_dir | Optional directory for per-sheet exports |
|
|
395
390
|
| | stream | Default stream when output_path is None |
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
# ExStruct — Excel Structured Extraction Engine
|
|
2
2
|
|
|
3
|
-
[](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://pypi.org/project/exstruct/) [](https://pepy.tech/projects/exstruct)  [](https://github.com/harumiWeb/exstruct/actions/workflows/ci.yml)
|
|
4
4
|
|
|
5
5
|

|
|
6
6
|
|
|
7
|
-
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts
|
|
7
|
+
ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
|
|
8
8
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
11
11
|
- **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
|
|
12
|
-
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
12
|
+
- **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
|
|
13
13
|
- **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
|
|
14
14
|
- **Table detection tuning**: adjust heuristics at runtime via API.
|
|
15
15
|
- **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
|
|
@@ -21,17 +21,17 @@ ExStruct reads Excel workbooks and outputs structured data (tables, shapes, char
|
|
|
21
21
|
pip install exstruct
|
|
22
22
|
```
|
|
23
23
|
|
|
24
|
-
Optional extras:
|
|
25
|
-
|
|
26
|
-
- YAML: `pip install pyyaml`
|
|
27
|
-
- TOON: `pip install python-toon`
|
|
28
|
-
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
29
|
-
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
30
|
-
|
|
31
|
-
Platform note:
|
|
32
|
-
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
33
|
-
|
|
34
|
-
## Quick Start (CLI)
|
|
24
|
+
Optional extras:
|
|
25
|
+
|
|
26
|
+
- YAML: `pip install pyyaml`
|
|
27
|
+
- TOON: `pip install python-toon`
|
|
28
|
+
- Rendering (PDF/PNG): Excel + `pip install pypdfium2 pillow`
|
|
29
|
+
- All extras at once: `pip install exstruct[yaml,toon,render]`
|
|
30
|
+
|
|
31
|
+
Platform note:
|
|
32
|
+
- Full extraction (shapes/charts) targets Windows + Excel (COM via xlwings). On other platforms, use `mode=light` to get cells + `table_candidates` safely.
|
|
33
|
+
|
|
34
|
+
## Quick Start (CLI)
|
|
35
35
|
|
|
36
36
|
```bash
|
|
37
37
|
exstruct input.xlsx > output.json # compact JSON to stdout (default)
|
|
@@ -64,19 +64,15 @@ wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
|
|
|
64
64
|
first_sheet.save("sheet.json") # SheetData → file (by extension)
|
|
65
65
|
print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
|
|
66
66
|
|
|
67
|
-
# ExStructEngine: per-instance options for extraction/output
|
|
68
|
-
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
69
|
-
|
|
70
|
-
engine = ExStructEngine(
|
|
71
|
-
options=StructOptions(mode="
|
|
72
|
-
output=OutputOptions(include_shapes=False, pretty=True),
|
|
73
|
-
)
|
|
74
|
-
wb2 = engine.extract("input.xlsx")
|
|
75
|
-
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
76
|
-
|
|
77
|
-
# Enable hyperlinks in other modes
|
|
78
|
-
engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
|
|
79
|
-
with_links = engine_links.extract("input.xlsx")
|
|
67
|
+
# ExStructEngine: per-instance options for extraction/output
|
|
68
|
+
from exstruct import ExStructEngine, StructOptions, OutputOptions
|
|
69
|
+
|
|
70
|
+
engine = ExStructEngine(
|
|
71
|
+
options=StructOptions(mode="standard"),
|
|
72
|
+
output=OutputOptions(include_shapes=False, pretty=True),
|
|
73
|
+
)
|
|
74
|
+
wb2 = engine.extract("input.xlsx")
|
|
75
|
+
engine.export(wb2, Path("out_filtered.json")) # drops shapes via OutputOptions
|
|
80
76
|
```
|
|
81
77
|
|
|
82
78
|
**Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
|
|
@@ -96,11 +92,11 @@ set_table_detection_params(
|
|
|
96
92
|
|
|
97
93
|
Use higher thresholds to reduce false positives; lower them if true tables are missed.
|
|
98
94
|
|
|
99
|
-
## Output Modes
|
|
100
|
-
|
|
101
|
-
- **light**: cells + table candidates (no COM needed).
|
|
102
|
-
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
103
|
-
- **verbose**: all shapes (with width/height), charts, table candidates
|
|
95
|
+
## Output Modes
|
|
96
|
+
|
|
97
|
+
- **light**: cells + table candidates (no COM needed).
|
|
98
|
+
- **standard**: texted shapes + arrows, charts (COM if available), table candidates.
|
|
99
|
+
- **verbose**: all shapes (with width/height), charts, table candidates.
|
|
104
100
|
|
|
105
101
|
## Error Handling / Fallbacks
|
|
106
102
|
|
|
@@ -324,16 +320,15 @@ BSD-3-Clause. See `LICENSE` for details.
|
|
|
324
320
|
- API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
|
|
325
321
|
# Engine option cheat sheet
|
|
326
322
|
|
|
327
|
-
| Option class | Field | Meaning |
|
|
328
|
-
| -------------- | ------------------- | ------- |
|
|
329
|
-
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
330
|
-
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
331
|
-
|
|
|
332
|
-
|
|
|
333
|
-
| |
|
|
334
|
-
| |
|
|
335
|
-
| |
|
|
336
|
-
| | include_charts | Include charts |
|
|
323
|
+
| Option class | Field | Meaning |
|
|
324
|
+
| -------------- | ------------------- | ------- |
|
|
325
|
+
| StructOptions | mode | "light"/"standard"/"verbose" |
|
|
326
|
+
| | table_params | Dict passed to `set_table_detection_params` (table_score_threshold, density_min, coverage_min, min_nonempty_cells) |
|
|
327
|
+
| OutputOptions | fmt | Default format ("json"/"yaml"/"yml"/"toon") |
|
|
328
|
+
| | pretty / indent | Pretty-print JSON and control indent |
|
|
329
|
+
| | include_rows | Include rows (False to drop) |
|
|
330
|
+
| | include_shapes | Include shapes |
|
|
331
|
+
| | include_charts | Include charts |
|
|
337
332
|
| | include_tables | Include table_candidates |
|
|
338
333
|
| | sheets_dir | Optional directory for per-sheet exports |
|
|
339
334
|
| | stream | Default stream when output_path is None |
|
|
@@ -36,8 +36,7 @@ ExtractionMode = Literal["light", "standard", "verbose"]
|
|
|
36
36
|
|
|
37
37
|
def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> WorkbookData:
|
|
38
38
|
"""Extract workbook semantic structure and return WorkbookData."""
|
|
39
|
-
|
|
40
|
-
engine = ExStructEngine(options=StructOptions(mode=mode, include_cell_links=include_links))
|
|
39
|
+
engine = ExStructEngine(options=StructOptions(mode=mode))
|
|
41
40
|
return engine.extract(file_path, mode=mode)
|
|
42
41
|
|
|
43
42
|
|
|
@@ -33,64 +33,25 @@ def warn_once(key: str, message: str) -> None:
|
|
|
33
33
|
_warned_keys.add(key)
|
|
34
34
|
|
|
35
35
|
|
|
36
|
-
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
-
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
-
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
-
result: Dict[str, List[CellRow]] = {}
|
|
40
|
-
for sheet_name, df in dfs.items():
|
|
41
|
-
df = df.fillna("")
|
|
42
|
-
rows: List[CellRow] = []
|
|
43
|
-
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
-
filtered: Dict[str, int | float | str] = {}
|
|
45
|
-
for j, v in enumerate(row):
|
|
46
|
-
s = "" if v is None else str(v)
|
|
47
|
-
if s.strip() == "":
|
|
48
|
-
continue
|
|
49
|
-
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
-
if not filtered:
|
|
51
|
-
continue
|
|
52
|
-
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
-
result[sheet_name] = rows
|
|
54
|
-
return result
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
def extract_sheet_cells_with_links(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
58
|
-
"""
|
|
59
|
-
Extract cells and hyperlinks per sheet.
|
|
60
|
-
|
|
61
|
-
Returns:
|
|
62
|
-
{sheet_name: [CellRow(r=..., c=..., links={"col_index": url, ...}), ...]}
|
|
63
|
-
|
|
64
|
-
Notes:
|
|
65
|
-
- Uses pandas extraction for values (same filtering as extract_sheet_cells).
|
|
66
|
-
- Collects hyperlinks via openpyxl (requires read_only=False because border maps/hyperlinks need full objects).
|
|
67
|
-
- Links are mapped by column index string (e.g., "0") to hyperlink.target.
|
|
68
|
-
"""
|
|
69
|
-
cell_rows = extract_sheet_cells(file_path)
|
|
70
|
-
wb = load_workbook(file_path, data_only=True, read_only=False)
|
|
71
|
-
links_by_sheet: Dict[str, Dict[int, Dict[str, str]]] = {}
|
|
72
|
-
for ws in wb.worksheets:
|
|
73
|
-
sheet_links: Dict[int, Dict[str, str]] = {}
|
|
74
|
-
for row in ws.iter_rows():
|
|
75
|
-
for cell in row:
|
|
76
|
-
link = getattr(cell, "hyperlink", None)
|
|
77
|
-
target = getattr(link, "target", None) if link else None
|
|
78
|
-
if not target:
|
|
79
|
-
continue
|
|
80
|
-
col_str = str(cell.col_idx - 1) # zero-based to align with extract_sheet_cells
|
|
81
|
-
sheet_links.setdefault(cell.row, {})[col_str] = target
|
|
82
|
-
links_by_sheet[ws.title] = sheet_links
|
|
83
|
-
|
|
84
|
-
merged: Dict[str, List[CellRow]] = {}
|
|
85
|
-
for sheet_name, rows in cell_rows.items():
|
|
86
|
-
sheet_links = links_by_sheet.get(sheet_name, {})
|
|
87
|
-
merged_rows: List[CellRow] = []
|
|
88
|
-
for row in rows:
|
|
89
|
-
links = sheet_links.get(row.r, {})
|
|
90
|
-
merged_rows.append(CellRow(r=row.r, c=row.c, links=links or None))
|
|
91
|
-
merged[sheet_name] = merged_rows
|
|
92
|
-
wb.close()
|
|
93
|
-
return merged
|
|
36
|
+
def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
|
|
37
|
+
"""Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
|
|
38
|
+
dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
|
|
39
|
+
result: Dict[str, List[CellRow]] = {}
|
|
40
|
+
for sheet_name, df in dfs.items():
|
|
41
|
+
df = df.fillna("")
|
|
42
|
+
rows: List[CellRow] = []
|
|
43
|
+
for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
|
|
44
|
+
filtered: Dict[str, int | float | str] = {}
|
|
45
|
+
for j, v in enumerate(row):
|
|
46
|
+
s = "" if v is None else str(v)
|
|
47
|
+
if s.strip() == "":
|
|
48
|
+
continue
|
|
49
|
+
filtered[str(j)] = _coerce_numeric_preserve_format(s)
|
|
50
|
+
if not filtered:
|
|
51
|
+
continue
|
|
52
|
+
rows.append(CellRow(r=excel_row, c=filtered))
|
|
53
|
+
result[sheet_name] = rows
|
|
54
|
+
return result
|
|
94
55
|
|
|
95
56
|
|
|
96
57
|
def shrink_to_content(
|
|
@@ -4,17 +4,15 @@ from pathlib import Path
|
|
|
4
4
|
from typing import Dict, List, Literal
|
|
5
5
|
|
|
6
6
|
import logging
|
|
7
|
-
import os
|
|
8
7
|
|
|
9
8
|
import xlwings as xw
|
|
10
9
|
|
|
11
10
|
from ..models import CellRow, Shape, SheetData, WorkbookData
|
|
12
|
-
from .cells import (
|
|
13
|
-
detect_tables,
|
|
14
|
-
detect_tables_openpyxl,
|
|
15
|
-
extract_sheet_cells,
|
|
16
|
-
|
|
17
|
-
)
|
|
11
|
+
from .cells import (
|
|
12
|
+
detect_tables,
|
|
13
|
+
detect_tables_openpyxl,
|
|
14
|
+
extract_sheet_cells,
|
|
15
|
+
)
|
|
18
16
|
from .charts import get_charts
|
|
19
17
|
from .shapes import get_shapes_with_position
|
|
20
18
|
|
|
@@ -76,16 +74,13 @@ def integrate_sheet_content(
|
|
|
76
74
|
|
|
77
75
|
|
|
78
76
|
def extract_workbook(
|
|
79
|
-
file_path: Path,
|
|
80
|
-
mode: Literal["light", "standard", "verbose"] = "standard",
|
|
81
|
-
*,
|
|
82
|
-
include_cell_links: bool = False,
|
|
77
|
+
file_path: Path, mode: Literal["light", "standard", "verbose"] = "standard"
|
|
83
78
|
) -> WorkbookData:
|
|
84
79
|
"""Extract workbook and return WorkbookData; fallback to cells+tables if Excel COM is unavailable."""
|
|
85
80
|
if mode not in _ALLOWED_MODES:
|
|
86
81
|
raise ValueError(f"Unsupported mode: {mode}")
|
|
87
82
|
|
|
88
|
-
cell_data =
|
|
83
|
+
cell_data = extract_sheet_cells(file_path)
|
|
89
84
|
|
|
90
85
|
def _cells_and_tables_only(reason: str) -> WorkbookData:
|
|
91
86
|
sheets: Dict[str, SheetData] = {}
|
|
@@ -109,9 +104,6 @@ def extract_workbook(
|
|
|
109
104
|
if mode == "light":
|
|
110
105
|
return _cells_and_tables_only("Light mode selected.")
|
|
111
106
|
|
|
112
|
-
if os.getenv("SKIP_COM_TESTS"):
|
|
113
|
-
return _cells_and_tables_only("SKIP_COM_TESTS is set; skipping COM/xlwings access.")
|
|
114
|
-
|
|
115
107
|
try:
|
|
116
108
|
wb, close_app = _open_workbook(file_path)
|
|
117
109
|
except Exception as e:
|
|
@@ -32,7 +32,6 @@ class StructOptions:
|
|
|
32
32
|
|
|
33
33
|
mode: ExtractionMode = "standard"
|
|
34
34
|
table_params: Optional[dict] = None # forwarded to set_table_detection_params if provided
|
|
35
|
-
include_cell_links: Optional[bool] = None # None → auto: verbose=True, others=False
|
|
36
35
|
|
|
37
36
|
|
|
38
37
|
@dataclass(frozen=True)
|
|
@@ -134,13 +133,8 @@ class ExStructEngine:
|
|
|
134
133
|
chosen_mode = mode or self.options.mode
|
|
135
134
|
if chosen_mode not in ("light", "standard", "verbose"):
|
|
136
135
|
raise ValueError(f"Unsupported mode: {chosen_mode}")
|
|
137
|
-
include_links = (
|
|
138
|
-
self.options.include_cell_links
|
|
139
|
-
if self.options.include_cell_links is not None
|
|
140
|
-
else chosen_mode == "verbose"
|
|
141
|
-
)
|
|
142
136
|
with self._table_params_scope():
|
|
143
|
-
return extract_workbook(Path(file_path), mode=chosen_mode
|
|
137
|
+
return extract_workbook(Path(file_path), mode=chosen_mode)
|
|
144
138
|
|
|
145
139
|
def serialize(
|
|
146
140
|
self,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|