exstruct 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
exstruct-0.1.0/LICENSE ADDED
@@ -0,0 +1,29 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2025, ExStruct Contributors
4
+ All rights reserved.
5
+
6
+ Redistribution and use in source and binary forms, with or without
7
+ modification, are permitted provided that the following conditions are met:
8
+
9
+ 1. Redistributions of source code must retain the above copyright notice, this
10
+ list of conditions and the following disclaimer.
11
+
12
+ 2. Redistributions in binary form must reproduce the above copyright notice,
13
+ this list of conditions and the following disclaimer in the documentation
14
+ and/or other materials provided with the distribution.
15
+
16
+ 3. Neither the name of the copyright holder nor the names of its
17
+ contributors may be used to endorse or promote products derived from
18
+ this software without specific prior written permission.
19
+
20
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
21
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
22
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
23
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
24
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
25
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
26
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
27
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
28
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
29
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,154 @@
1
+ Metadata-Version: 2.3
2
+ Name: exstruct
3
+ Version: 0.1.0
4
+ Summary: Add your description here
5
+ Keywords: excel,structure,data,exstruct
6
+ Author: harumiWeb
7
+ Author-email: harumiWeb <ganaharumi@outlook.jp>
8
+ License: BSD 3-Clause License
9
+
10
+ Copyright (c) 2025, ExStruct Contributors
11
+ All rights reserved.
12
+
13
+ Redistribution and use in source and binary forms, with or without
14
+ modification, are permitted provided that the following conditions are met:
15
+
16
+ 1. Redistributions of source code must retain the above copyright notice, this
17
+ list of conditions and the following disclaimer.
18
+
19
+ 2. Redistributions in binary form must reproduce the above copyright notice,
20
+ this list of conditions and the following disclaimer in the documentation
21
+ and/or other materials provided with the distribution.
22
+
23
+ 3. Neither the name of the copyright holder nor the names of its
24
+ contributors may be used to endorse or promote products derived from
25
+ this software without specific prior written permission.
26
+
27
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
28
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
29
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
30
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
31
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
32
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
33
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
34
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
35
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
36
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
37
+ Requires-Dist: numpy>=2.3.5
38
+ Requires-Dist: openpyxl>=3.1.5
39
+ Requires-Dist: pandas>=2.3.3
40
+ Requires-Dist: pydantic>=2.12.5
41
+ Requires-Dist: scipy>=1.16.3
42
+ Requires-Dist: xlwings>=0.33.16
43
+ Requires-Dist: pypdfium2>=5.1.0 ; extra == 'render'
44
+ Requires-Dist: python-toon>=0.1.3 ; extra == 'toon'
45
+ Requires-Dist: pyyaml>=6.0.3 ; extra == 'yaml'
46
+ Requires-Python: >=3.12
47
+ Project-URL: Documentation, https://harumiweb.github.io/exstruct/
48
+ Project-URL: Homepage, https://harumiweb.github.io/exstruct/
49
+ Project-URL: Issues, https://github.com/harumiWeb/exstruct/issues
50
+ Project-URL: Repository, https://github.com/harumiWeb/exstruct
51
+ Provides-Extra: render
52
+ Provides-Extra: toon
53
+ Provides-Extra: yaml
54
+ Description-Content-Type: text/markdown
55
+
56
+ # ExStruct — Excel Structured Extraction Engine
57
+
58
+ ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
59
+
60
+ ## Features
61
+
62
+ - **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
63
+ - **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
64
+ - **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
65
+ - **Table detection tuning**: adjust heuristics at runtime via API.
66
+ - **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
67
+ - **Graceful fallback**: if Excel COM is unavailable, extraction falls back to cells + table candidates without crashing.
68
+
69
+ ## Installation
70
+
71
+ ```bash
72
+ pip install exstruct
73
+ ```
74
+
75
+ Optional extras:
76
+
77
+ - YAML: `pip install pyyaml`
78
+ - TOON: `pip install python-toon`
79
+ - Rendering (PDF/PNG): Excel + `pip install pypdfium2`
80
+
81
+ ## Quick Start (CLI)
82
+
83
+ ```bash
84
+ exstruct input.xlsx # compact JSON (default)
85
+ exstruct input.xlsx --pretty # pretty-printed JSON
86
+ exstruct input.xlsx --format yaml # YAML (needs pyyaml)
87
+ exstruct input.xlsx --format toon # TOON (needs python-toon)
88
+ exstruct input.xlsx --mode light # cells + table candidates only
89
+ exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
90
+ ```
91
+
92
+ ## Quick Start (Python)
93
+
94
+ ```python
95
+ from pathlib import Path
96
+ from exstruct import extract, export, set_table_detection_params
97
+
98
+ # Tune table detection (optional)
99
+ set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
100
+
101
+ # Extract with modes: "light", "standard", "verbose"
102
+ wb = extract("input.xlsx", mode="standard")
103
+ export(wb, Path("out.json"), pretty=False) # compact JSON
104
+ ```
105
+
106
+ ## Table Detection Tuning
107
+
108
+ ```python
109
+ from exstruct import set_table_detection_params
110
+
111
+ set_table_detection_params(
112
+ table_score_threshold=0.35, # increase to be stricter
113
+ density_min=0.05,
114
+ coverage_min=0.2,
115
+ min_nonempty_cells=3,
116
+ )
117
+ ```
118
+
119
+ Use higher thresholds to reduce false positives; lower them if true tables are missed.
120
+
121
+ ## Output Modes
122
+
123
+ - **light**: cells + table candidates (no COM needed).
124
+ - **standard**: texted shapes + arrows, charts (COM if available), table candidates.
125
+ - **verbose**: all shapes (with width/height), charts, table candidates.
126
+
127
+ ## Error Handling / Fallbacks
128
+
129
+ - Excel COM unavailable → falls back to cells + table candidates; shapes/charts empty.
130
+ - Shape extraction failure → logs warning, still returns cells + table candidates.
131
+ - CLI prints errors to stdout/stderr and returns non-zero on failures.
132
+
133
+ ## Optional Rendering
134
+
135
+ Requires Excel and `pypdfium2`.
136
+
137
+ ```bash
138
+ exstruct input.xlsx --pdf --image --dpi 144
139
+ ```
140
+
141
+ Creates `<output>.pdf` and `<output>_images/` PNGs per sheet.
142
+
143
+ ## Notes
144
+
145
+ - Default JSON is compact to reduce tokens; use `--pretty` or `pretty=True` when readability matters.
146
+ - Field `table_candidates` replaces `tables`; adjust downstream consumers accordingly.
147
+
148
+ ## License
149
+
150
+ BSD-3-Clause. See `LICENSE` for details.
151
+
152
+ ## Documentation
153
+
154
+ - API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
@@ -0,0 +1,99 @@
1
+ # ExStruct — Excel Structured Extraction Engine
2
+
3
+ ExStruct reads Excel workbooks and outputs structured data (tables, shapes, charts) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
4
+
5
+ ## Features
6
+
7
+ - **Excel → Structured JSON**: cells, shapes, charts, and table candidates per sheet.
8
+ - **Output modes**: `light` (cells + table candidates only), `standard` (texted shapes + arrows, charts), `verbose` (all shapes with width/height).
9
+ - **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
10
+ - **Table detection tuning**: adjust heuristics at runtime via API.
11
+ - **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
12
+ - **Graceful fallback**: if Excel COM is unavailable, extraction falls back to cells + table candidates without crashing.
13
+
14
+ ## Installation
15
+
16
+ ```bash
17
+ pip install exstruct
18
+ ```
19
+
20
+ Optional extras:
21
+
22
+ - YAML: `pip install pyyaml`
23
+ - TOON: `pip install python-toon`
24
+ - Rendering (PDF/PNG): Excel + `pip install pypdfium2`
25
+
26
+ ## Quick Start (CLI)
27
+
28
+ ```bash
29
+ exstruct input.xlsx # compact JSON (default)
30
+ exstruct input.xlsx --pretty # pretty-printed JSON
31
+ exstruct input.xlsx --format yaml # YAML (needs pyyaml)
32
+ exstruct input.xlsx --format toon # TOON (needs python-toon)
33
+ exstruct input.xlsx --mode light # cells + table candidates only
34
+ exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
35
+ ```
36
+
37
+ ## Quick Start (Python)
38
+
39
+ ```python
40
+ from pathlib import Path
41
+ from exstruct import extract, export, set_table_detection_params
42
+
43
+ # Tune table detection (optional)
44
+ set_table_detection_params(table_score_threshold=0.3, density_min=0.04)
45
+
46
+ # Extract with modes: "light", "standard", "verbose"
47
+ wb = extract("input.xlsx", mode="standard")
48
+ export(wb, Path("out.json"), pretty=False) # compact JSON
49
+ ```
50
+
51
+ ## Table Detection Tuning
52
+
53
+ ```python
54
+ from exstruct import set_table_detection_params
55
+
56
+ set_table_detection_params(
57
+ table_score_threshold=0.35, # increase to be stricter
58
+ density_min=0.05,
59
+ coverage_min=0.2,
60
+ min_nonempty_cells=3,
61
+ )
62
+ ```
63
+
64
+ Use higher thresholds to reduce false positives; lower them if true tables are missed.
65
+
66
+ ## Output Modes
67
+
68
+ - **light**: cells + table candidates (no COM needed).
69
+ - **standard**: texted shapes + arrows, charts (COM if available), table candidates.
70
+ - **verbose**: all shapes (with width/height), charts, table candidates.
71
+
72
+ ## Error Handling / Fallbacks
73
+
74
+ - Excel COM unavailable → falls back to cells + table candidates; shapes/charts empty.
75
+ - Shape extraction failure → logs warning, still returns cells + table candidates.
76
+ - CLI prints errors to stdout/stderr and returns non-zero on failures.
77
+
78
+ ## Optional Rendering
79
+
80
+ Requires Excel and `pypdfium2`.
81
+
82
+ ```bash
83
+ exstruct input.xlsx --pdf --image --dpi 144
84
+ ```
85
+
86
+ Creates `<output>.pdf` and `<output>_images/` PNGs per sheet.
87
+
88
+ ## Notes
89
+
90
+ - Default JSON is compact to reduce tokens; use `--pretty` or `pretty=True` when readability matters.
91
+ - Field `table_candidates` replaces `tables`; adjust downstream consumers accordingly.
92
+
93
+ ## License
94
+
95
+ BSD-3-Clause. See `LICENSE` for details.
96
+
97
+ ## Documentation
98
+
99
+ - API Reference (GitHub Pages): https://harumiweb.github.io/exstruct/
@@ -0,0 +1,45 @@
1
+ [project]
2
+ name = "exstruct"
3
+ version = "0.1.0"
4
+ description = "Add your description here"
5
+ readme = "README.md"
6
+ license = { file = "LICENSE" }
7
+ keywords = ["excel", "structure", "data", "exstruct"]
8
+ authors = [
9
+ { name = "harumiWeb", email = "ganaharumi@outlook.jp" }
10
+ ]
11
+ requires-python = ">=3.12"
12
+ dependencies = [
13
+ "numpy>=2.3.5",
14
+ "openpyxl>=3.1.5",
15
+ "pandas>=2.3.3",
16
+ "pydantic>=2.12.5",
17
+ "scipy>=1.16.3",
18
+ "xlwings>=0.33.16",
19
+ ]
20
+
21
+ [build-system]
22
+ requires = ["uv_build>=0.8.4,<0.9.0"]
23
+ build-backend = "uv_build"
24
+
25
+ [dependency-groups]
26
+ dev = [
27
+ "mkdocs-material>=9.7.0",
28
+ "pytest>=9.0.1",
29
+ "pytest-cov>=7.0.0",
30
+ "pytest-mock>=3.15.1",
31
+ ]
32
+
33
+ [project.optional-dependencies]
34
+ yaml = ["pyyaml>=6.0.3"]
35
+ toon = ["python-toon>=0.1.3"]
36
+ render = ["pypdfium2>=5.1.0"]
37
+
38
+ [project.scripts]
39
+ exstruct = "exstruct.cli.main:main"
40
+
41
+ [project.urls]
42
+ Homepage = "https://harumiweb.github.io/exstruct/"
43
+ Repository = "https://github.com/harumiWeb/exstruct"
44
+ Issues = "https://github.com/harumiWeb/exstruct/issues"
45
+ Documentation = "https://harumiweb.github.io/exstruct/"
@@ -0,0 +1,116 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+ from typing import Literal, Optional
5
+
6
+ from .core.integrate import extract_workbook
7
+ from .core.cells import set_table_detection_params
8
+ from .io import save_as_json, save_as_toon, save_as_yaml, save_sheets
9
+ from .models import CellRow, Chart, ChartSeries, Shape, SheetData, WorkbookData
10
+ from .render import export_pdf, export_sheet_images
11
+
12
+ __all__ = [
13
+ "extract",
14
+ "export",
15
+ "export_sheets",
16
+ "export_pdf",
17
+ "export_sheet_images",
18
+ "process_excel",
19
+ "ExtractionMode",
20
+ "CellRow",
21
+ "Shape",
22
+ "ChartSeries",
23
+ "Chart",
24
+ "SheetData",
25
+ "WorkbookData",
26
+ "set_table_detection_params",
27
+ ]
28
+
29
+
30
+ ExtractionMode = Literal["light", "standard", "verbose"]
31
+
32
+
33
+ def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> WorkbookData:
34
+ """Extract workbook semantic structure and return WorkbookData."""
35
+ if mode not in ("light", "standard", "verbose"):
36
+ raise ValueError(f"Unsupported mode: {mode}")
37
+ return extract_workbook(Path(file_path), mode=mode)
38
+
39
+
40
+ def export(
41
+ data: WorkbookData,
42
+ path: str | Path,
43
+ fmt: Optional[Literal["json", "yaml", "yml", "toon"]] = None,
44
+ *,
45
+ pretty: bool = False,
46
+ indent: int | None = None,
47
+ ) -> None:
48
+ """Export WorkbookData to supported file formats (json/yaml/toon)."""
49
+ dest = Path(path)
50
+ format_hint = (fmt or dest.suffix.lstrip(".") or "json").lower()
51
+ match format_hint:
52
+ case "json":
53
+ save_as_json(data, dest, pretty=pretty, indent=indent)
54
+ case "yaml" | "yml":
55
+ save_as_yaml(data, dest)
56
+ case "toon":
57
+ save_as_toon(data, dest)
58
+ case _:
59
+ raise ValueError(f"Unsupported export format: {format_hint}")
60
+
61
+
62
+ def export_sheets(data: WorkbookData, dir_path: str | Path) -> dict[str, Path]:
63
+ """
64
+ Export each sheet as a JSON file (book_name + SheetData) into a directory.
65
+ Returns a mapping of sheet name to written path.
66
+ """
67
+ return save_sheets(data, Path(dir_path), fmt="json")
68
+
69
+
70
+ def export_sheets_as(
71
+ data: WorkbookData,
72
+ dir_path: str | Path,
73
+ fmt: Literal["json", "yaml", "yml", "toon"] = "json",
74
+ *,
75
+ pretty: bool = False,
76
+ indent: int | None = None,
77
+ ) -> dict[str, Path]:
78
+ """
79
+ Export each sheet in the given format (json/yaml/toon), including book_name and SheetData; returns sheet name → path map.
80
+ """
81
+ return save_sheets(data, Path(dir_path), fmt=fmt, pretty=pretty, indent=indent)
82
+
83
+
84
+ def process_excel(
85
+ file_path: Path,
86
+ output_path: Path,
87
+ out_fmt: str = "json",
88
+ image: bool = False,
89
+ pdf: bool = False,
90
+ dpi: int = 72,
91
+ mode: ExtractionMode = "standard",
92
+ pretty: bool = False,
93
+ indent: int | None = None,
94
+ ) -> None:
95
+ """Convenience wrapper for CLI: export workbook and optionally PDF/PNG images (Excel required for rendering)."""
96
+ if mode not in ("light", "standard", "verbose"):
97
+ raise ValueError(f"Unsupported mode: {mode}")
98
+ workbook_model = extract(file_path, mode=mode)
99
+ match out_fmt:
100
+ case "json":
101
+ save_as_json(workbook_model, output_path, pretty=pretty, indent=indent)
102
+ case "yaml" | "yml":
103
+ save_as_yaml(workbook_model, output_path)
104
+ case "toon":
105
+ save_as_toon(workbook_model, output_path)
106
+ case _:
107
+ raise ValueError(f"Unsupported export format: {out_fmt}")
108
+
109
+ if pdf or image:
110
+ pdf_path = output_path.with_suffix(".pdf")
111
+ export_pdf(file_path, pdf_path)
112
+ if image:
113
+ images_dir = output_path.parent / f"{output_path.stem}_images"
114
+ export_sheet_images(file_path, images_dir, dpi=dpi)
115
+
116
+ print(f"{file_path.name} -> {output_path} completed.")
@@ -0,0 +1,93 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ from pathlib import Path
5
+
6
+ from exstruct import process_excel
7
+
8
+
9
+ def build_parser() -> argparse.ArgumentParser:
10
+ parser = argparse.ArgumentParser(
11
+ description="Dev-only CLI stub for ExStruct extraction."
12
+ )
13
+ parser.add_argument("input", type=Path, help="Excel file (.xlsx/.xlsm/.xls)")
14
+ parser.add_argument(
15
+ "-o",
16
+ "--output",
17
+ type=Path,
18
+ help="Output path (defaults to <input>.json)",
19
+ )
20
+ parser.add_argument(
21
+ "-f",
22
+ "--format",
23
+ default="json",
24
+ choices=["json", "yaml", "yml", "toon"],
25
+ help="Export format",
26
+ )
27
+ parser.add_argument(
28
+ "--image",
29
+ action="store_true",
30
+ help="(placeholder) Render PNG alongside JSON",
31
+ )
32
+ parser.add_argument(
33
+ "--pdf",
34
+ action="store_true",
35
+ help="(placeholder) Render PDF alongside JSON",
36
+ )
37
+ parser.add_argument(
38
+ "--dpi",
39
+ type=int,
40
+ default=144,
41
+ help="DPI for image rendering (placeholder)",
42
+ )
43
+ parser.add_argument(
44
+ "-m",
45
+ "--mode",
46
+ default="standard",
47
+ choices=["light", "standard", "verbose"],
48
+ help="Extraction detail level",
49
+ )
50
+ parser.add_argument(
51
+ "--pretty",
52
+ action="store_true",
53
+ help="Pretty-print JSON output (indent=2). Default is compact JSON.",
54
+ )
55
+ return parser
56
+
57
+
58
+ def main(argv: list[str] | None = None) -> int:
59
+ parser = build_parser()
60
+ args = parser.parse_args(argv)
61
+
62
+ input_path: Path = args.input
63
+ if not input_path.exists():
64
+ print(f"File not found: {input_path}")
65
+ return 0
66
+
67
+ suffix = ".json"
68
+ if args.format in ("yaml", "yml"):
69
+ suffix = ".yaml"
70
+ elif args.format == "toon":
71
+ suffix = ".toon"
72
+
73
+ output_path: Path = args.output or input_path.with_suffix(suffix)
74
+
75
+ try:
76
+ process_excel(
77
+ file_path=input_path,
78
+ output_path=output_path,
79
+ out_fmt=args.format,
80
+ image=args.image,
81
+ pdf=args.pdf,
82
+ dpi=args.dpi,
83
+ mode=args.mode,
84
+ pretty=args.pretty,
85
+ )
86
+ return 0
87
+ except Exception as e:
88
+ print(f"Error: {e}")
89
+ return 1
90
+
91
+
92
+ if __name__ == "__main__":
93
+ raise SystemExit(main())
File without changes