exstruct 0.2.51__tar.gz → 0.2.61__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,10 +1,9 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: exstruct
3
- Version: 0.2.51
3
+ Version: 0.2.61
4
4
  Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
5
5
  Keywords: excel,structure,data,exstruct
6
6
  Author: harumiWeb
7
- Author-email: harumiWeb <ganaharumi@outlook.jp>
8
7
  License: BSD 3-Clause License
9
8
 
10
9
  Copyright (c) 2025, ExStruct Contributors
@@ -60,12 +59,15 @@ Description-Content-Type: text/markdown
60
59
 
61
60
  ![ExStruct Image](/docs/assets/icon.webp)
62
61
 
63
- ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, print areas/views, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
62
+ ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, print areas/views, auto page-break areas, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
63
+
64
+ [日本版README](README.ja.md)
64
65
 
65
66
  ## Features
66
67
 
67
- - **Excel → Structured JSON**: cells, shapes, charts, table candidates, and print areas/views per sheet.
68
+ - **Excel → Structured JSON**: cells, shapes, charts, table candidates, print areas/views, and auto page-break areas per sheet.
68
69
  - **Output modes**: `light` (cells + table candidates + print areas; no COM, shapes/charts empty), `standard` (texted shapes + arrows, charts, print areas), `verbose` (all shapes with width/height, charts with size, print areas). Verbose also emits cell hyperlinks. Size output is flag-controlled.
70
+ - **Auto page-break export (COM only)**: capture Excel-computed auto page breaks and write per-area JSON/YAML/TOON when requested.
69
71
  - **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
70
72
  - **Table detection tuning**: adjust heuristics at runtime via API.
71
73
  - **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
@@ -101,6 +103,8 @@ exstruct input.xlsx --mode light # cells + table candidates only
101
103
  exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
102
104
  ```
103
105
 
106
+ Auto page-break exports are API-only (Excel/COM): set `DestinationOptions.auto_page_breaks_dir` or call `export_auto_page_breaks(...)`.
107
+
104
108
  ## Quick Start (Python)
105
109
 
106
110
  ```python
@@ -122,27 +126,45 @@ wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
122
126
  first_sheet.save("sheet.json") # SheetData → file (by extension)
123
127
  print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
124
128
 
125
- # ExStructEngine: per-instance options (nested configs)
126
- from exstruct import ExStructEngine, StructOptions, OutputOptions, FormatOptions, FilterOptions, DestinationOptions
127
-
128
- engine = ExStructEngine(
129
- options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
130
- output=OutputOptions(
131
- format=FormatOptions(pretty=True),
132
- filters=FilterOptions(include_shapes=False), # drop shapes in output
133
- destinations=DestinationOptions(sheets_dir=Path("out_sheets")), # also write per-sheet files
134
- ),
135
- )
136
- wb2 = engine.extract("input.xlsx")
137
- engine.export(wb2, Path("out_filtered.json")) # drops shapes via filters
138
-
139
- # Enable hyperlinks in other modes
140
- engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
141
- with_links = engine_links.extract("input.xlsx")
129
+ # ExStructEngine: per-instance options (nested configs)
130
+ from exstruct import (
131
+ DestinationOptions,
132
+ ExStructEngine,
133
+ FilterOptions,
134
+ FormatOptions,
135
+ OutputOptions,
136
+ StructOptions,
137
+ export_auto_page_breaks,
138
+ )
139
+
140
+ engine = ExStructEngine(
141
+ options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
142
+ output=OutputOptions(
143
+ format=FormatOptions(pretty=True),
144
+ filters=FilterOptions(include_shapes=False), # drop shapes in output
145
+ destinations=DestinationOptions(sheets_dir=Path("out_sheets")), # also write per-sheet files
146
+ ),
147
+ )
148
+ wb2 = engine.extract("input.xlsx")
149
+ engine.export(wb2, Path("out_filtered.json")) # drops shapes via filters
150
+
151
+ # Enable hyperlinks in other modes
152
+ engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
153
+ with_links = engine_links.extract("input.xlsx")
142
154
 
143
155
  # Export per print area (if print areas exist)
144
156
  from exstruct import export_print_areas_as
145
157
  export_print_areas_as(wb, "areas", fmt="json", pretty=True)
158
+
159
+ # Auto page-break extraction/output (COM only; raises if no auto breaks exist)
160
+ engine_auto = ExStructEngine(
161
+ output=OutputOptions(
162
+ destinations=DestinationOptions(auto_page_breaks_dir=Path("auto_areas"))
163
+ )
164
+ )
165
+ wb_auto = engine_auto.extract("input.xlsx") # includes SheetData.auto_print_areas
166
+ engine_auto.export(wb_auto, Path("out_with_auto.json")) # also writes auto_areas/*
167
+ export_auto_page_breaks(wb_auto, "auto_areas", fmt="json", pretty=True) # manual writer
146
168
  ```
147
169
 
148
170
  **Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
@@ -382,10 +404,12 @@ In short, **exstruct = “an engine that converts Excel into a format AI can und
382
404
  - Default JSON is compact to reduce tokens; use `--pretty` or `pretty=True` when readability matters.
383
405
  - Field `table_candidates` replaces `tables`; adjust downstream consumers accordingly.
384
406
 
385
- ## Print Areas (PrintArea / PrintAreaView)
407
+ ## Print Areas and Auto Page Breaks (PrintArea / PrintAreaView)
386
408
 
387
409
  - `SheetData.print_areas` holds print areas (cell coordinates) in light/standard/verbose.
410
+ - `SheetData.auto_print_areas` holds Excel COM-computed auto page-break areas when auto page-break extraction is enabled (COM only).
388
411
  - Use `export_print_areas_as(...)` or CLI `--print-areas-dir` to write one file per print area (nothing is written if none exist).
412
+ - Use `DestinationOptions.auto_page_breaks_dir` (preferred) or `export_auto_page_breaks(...)` to write per-auto-page-break files; the API raises `ValueError` if no auto page breaks exist.
389
413
  - `PrintAreaView` includes rows and table candidates inside the area, plus shapes/charts that overlap the area (size-less shapes are treated as points). `normalize=True` rebases row/col indices to the area origin.
390
414
 
391
415
  ## License
@@ -4,12 +4,15 @@
4
4
 
5
5
  ![ExStruct Image](/docs/assets/icon.webp)
6
6
 
7
- ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, print areas/views, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
7
+ ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, print areas/views, auto page-break areas, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
8
+
9
+ [日本版README](README.ja.md)
8
10
 
9
11
  ## Features
10
12
 
11
- - **Excel → Structured JSON**: cells, shapes, charts, table candidates, and print areas/views per sheet.
13
+ - **Excel → Structured JSON**: cells, shapes, charts, table candidates, print areas/views, and auto page-break areas per sheet.
12
14
  - **Output modes**: `light` (cells + table candidates + print areas; no COM, shapes/charts empty), `standard` (texted shapes + arrows, charts, print areas), `verbose` (all shapes with width/height, charts with size, print areas). Verbose also emits cell hyperlinks. Size output is flag-controlled.
15
+ - **Auto page-break export (COM only)**: capture Excel-computed auto page breaks and write per-area JSON/YAML/TOON when requested.
13
16
  - **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
14
17
  - **Table detection tuning**: adjust heuristics at runtime via API.
15
18
  - **CLI rendering** (Excel required): optional PDF and per-sheet PNGs.
@@ -45,6 +48,8 @@ exstruct input.xlsx --mode light # cells + table candidates only
45
48
  exstruct input.xlsx --pdf --image # PDF and PNGs (Excel required)
46
49
  ```
47
50
 
51
+ Auto page-break exports are API-only (Excel/COM): set `DestinationOptions.auto_page_breaks_dir` or call `export_auto_page_breaks(...)`.
52
+
48
53
  ## Quick Start (Python)
49
54
 
50
55
  ```python
@@ -66,27 +71,45 @@ wb.save("out.json", pretty=True) # WorkbookData → file (by extension)
66
71
  first_sheet.save("sheet.json") # SheetData → file (by extension)
67
72
  print(first_sheet.to_yaml()) # YAML text (requires pyyaml)
68
73
 
69
- # ExStructEngine: per-instance options (nested configs)
70
- from exstruct import ExStructEngine, StructOptions, OutputOptions, FormatOptions, FilterOptions, DestinationOptions
71
-
72
- engine = ExStructEngine(
73
- options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
74
- output=OutputOptions(
75
- format=FormatOptions(pretty=True),
76
- filters=FilterOptions(include_shapes=False), # drop shapes in output
77
- destinations=DestinationOptions(sheets_dir=Path("out_sheets")), # also write per-sheet files
78
- ),
79
- )
80
- wb2 = engine.extract("input.xlsx")
81
- engine.export(wb2, Path("out_filtered.json")) # drops shapes via filters
82
-
83
- # Enable hyperlinks in other modes
84
- engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
85
- with_links = engine_links.extract("input.xlsx")
74
+ # ExStructEngine: per-instance options (nested configs)
75
+ from exstruct import (
76
+ DestinationOptions,
77
+ ExStructEngine,
78
+ FilterOptions,
79
+ FormatOptions,
80
+ OutputOptions,
81
+ StructOptions,
82
+ export_auto_page_breaks,
83
+ )
84
+
85
+ engine = ExStructEngine(
86
+ options=StructOptions(mode="verbose"), # verbose includes hyperlinks by default
87
+ output=OutputOptions(
88
+ format=FormatOptions(pretty=True),
89
+ filters=FilterOptions(include_shapes=False), # drop shapes in output
90
+ destinations=DestinationOptions(sheets_dir=Path("out_sheets")), # also write per-sheet files
91
+ ),
92
+ )
93
+ wb2 = engine.extract("input.xlsx")
94
+ engine.export(wb2, Path("out_filtered.json")) # drops shapes via filters
95
+
96
+ # Enable hyperlinks in other modes
97
+ engine_links = ExStructEngine(options=StructOptions(mode="standard", include_cell_links=True))
98
+ with_links = engine_links.extract("input.xlsx")
86
99
 
87
100
  # Export per print area (if print areas exist)
88
101
  from exstruct import export_print_areas_as
89
102
  export_print_areas_as(wb, "areas", fmt="json", pretty=True)
103
+
104
+ # Auto page-break extraction/output (COM only; raises if no auto breaks exist)
105
+ engine_auto = ExStructEngine(
106
+ output=OutputOptions(
107
+ destinations=DestinationOptions(auto_page_breaks_dir=Path("auto_areas"))
108
+ )
109
+ )
110
+ wb_auto = engine_auto.extract("input.xlsx") # includes SheetData.auto_print_areas
111
+ engine_auto.export(wb_auto, Path("out_with_auto.json")) # also writes auto_areas/*
112
+ export_auto_page_breaks(wb_auto, "auto_areas", fmt="json", pretty=True) # manual writer
90
113
  ```
91
114
 
92
115
  **Note (non-COM environments):** If Excel COM is unavailable, extraction still runs and returns cells + `table_candidates`; `shapes`/`charts` will be empty.
@@ -326,10 +349,12 @@ In short, **exstruct = “an engine that converts Excel into a format AI can und
326
349
  - Default JSON is compact to reduce tokens; use `--pretty` or `pretty=True` when readability matters.
327
350
  - Field `table_candidates` replaces `tables`; adjust downstream consumers accordingly.
328
351
 
329
- ## Print Areas (PrintArea / PrintAreaView)
352
+ ## Print Areas and Auto Page Breaks (PrintArea / PrintAreaView)
330
353
 
331
354
  - `SheetData.print_areas` holds print areas (cell coordinates) in light/standard/verbose.
355
+ - `SheetData.auto_print_areas` holds Excel COM-computed auto page-break areas when auto page-break extraction is enabled (COM only).
332
356
  - Use `export_print_areas_as(...)` or CLI `--print-areas-dir` to write one file per print area (nothing is written if none exist).
357
+ - Use `DestinationOptions.auto_page_breaks_dir` (preferred) or `export_auto_page_breaks(...)` to write per-auto-page-break files; the API raises `ValueError` if no auto page breaks exist.
333
358
  - `PrintAreaView` includes rows and table candidates inside the area, plus shapes/charts that overlap the area (size-less shapes are treated as points). `normalize=True` rebases row/col indices to the area origin.
334
359
 
335
360
  ## License
@@ -1,111 +1,118 @@
1
- [project]
2
- name = "exstruct"
3
- version = "0.2.51"
4
- description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
5
- readme = "README.md"
6
- license = { file = "LICENSE" }
7
- keywords = ["excel", "structure", "data", "exstruct"]
8
- authors = [
9
- { name = "harumiWeb", email = "ganaharumi@outlook.jp" }
10
- ]
11
- requires-python = ">=3.11"
12
- dependencies = [
13
- "numpy>=2.3.5",
14
- "openpyxl>=3.1.5",
15
- "pandas>=2.3.3",
16
- "pydantic>=2.12.5",
17
- "scipy>=1.16.3",
18
- "xlwings>=0.33.16",
19
- ]
20
-
21
- [build-system]
22
- requires = ["uv_build>=0.8.4,<0.9.0"]
23
- build-backend = "uv_build"
24
-
25
- [dependency-groups]
26
- dev = [
27
- "mkdocs-material>=9.7.0",
28
- "mypy>=1.19.0",
29
- "pytest>=9.0.1",
30
- "pytest-cov>=7.0.0",
31
- "pytest-mock>=3.15.1",
32
- "ruff>=0.14.8",
33
- ]
34
-
35
- [project.optional-dependencies]
36
- yaml = ["pyyaml>=6.0.3"]
37
- toon = ["python-toon>=0.1.3"]
38
- render = ["pypdfium2>=5.1.0", "Pillow>=12.0.0"]
39
-
40
- [project.scripts]
41
- exstruct = "exstruct.cli.main:main"
42
-
43
- [project.urls]
44
- Homepage = "https://harumiweb.github.io/exstruct/"
45
- Repository = "https://github.com/harumiWeb/exstruct"
46
- Issues = "https://github.com/harumiWeb/exstruct/issues"
47
- Documentation = "https://harumiweb.github.io/exstruct/"
48
-
49
- [tool.coverage.run]
50
- omit = [
51
- "tests/*",
52
- "*/test_*.py",
53
- "*/gen_py/*",
54
- ]
55
-
56
- [tool.ruff]
57
- target-version = "py311"
58
- src = ["exstruct"]
59
-
60
- select = [
61
- "E", # pycodestyle errors
62
- "W", # pycodestyle warnings
63
- "F", # pyflakes
64
- "I", # import sorting
65
- "UP", # pyupgrade
66
- "B", # flake8-bugbear
67
- "N", # naming
68
- "C90", # complexity
69
- "A", # flake8-builtins
70
- "ANN", # type annotations
71
- ]
72
-
73
- ignore = [
74
- "E501", # 行長は許容(Excel JSON は長くなりがち)
75
- "B008", # Pydantic の default_factory を誤検知するため
76
- "ANN101", # self に型を要求されてしまうため
77
- "ANN102", # cls も同様
78
- ]
79
-
80
- fix = true
81
-
82
- # 型ヒントのスタイル
83
- [tool.ruff.lint]
84
- extend-select = ["ANN"]
85
-
86
- # import の並び替え設定
87
- [tool.ruff.isort]
88
- combine-as-imports = true
89
- known-first-party = ["exstruct"]
90
- force-sort-within-sections = true
91
-
92
- # 複雑度チェック(関数の最大複雑度)
93
- [tool.ruff.mccabe]
94
- max-complexity = 12
95
-
96
- [tool.ruff.per-file-ignores]
97
- "tests/**/*.py" = ["N802", "N803", "N806"]
98
-
99
-
100
- [tool.mypy]
101
- packages = ["exstruct"]
102
- python_version = "3.11"
103
-
104
- # 外部ライブラリは一切チェックしない
105
- ignore_missing_imports = true
106
-
107
- # 自作コードは厳密にチェックする
108
- strict = true
109
-
110
- # Pydantic v2 向け
111
- plugins = ["pydantic.mypy"]
1
+ [project]
2
+ name = "exstruct"
3
+ version = "0.2.61"
4
+ description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
5
+ readme = "README.md"
6
+ license = { file = "LICENSE" }
7
+ keywords = ["excel", "structure", "data", "exstruct"]
8
+ authors = [
9
+ { name = "harumiWeb"}
10
+ ]
11
+ requires-python = ">=3.11"
12
+ dependencies = [
13
+ "numpy>=2.3.5",
14
+ "openpyxl>=3.1.5",
15
+ "pandas>=2.3.3",
16
+ "pydantic>=2.12.5",
17
+ "scipy>=1.16.3",
18
+ "xlwings>=0.33.16",
19
+ ]
20
+
21
+ [build-system]
22
+ requires = ["uv_build>=0.8.4,<0.9.0"]
23
+ build-backend = "uv_build"
24
+
25
+ [dependency-groups]
26
+ dev = [
27
+ "mkdocs-material>=9.7.0",
28
+ "mypy>=1.19.0",
29
+ "pre-commit>=4.5.0",
30
+ "pytest>=9.0.1",
31
+ "pytest-cov>=7.0.0",
32
+ "pytest-mock>=3.15.1",
33
+ "ruff>=0.14.8",
34
+ ]
35
+
36
+ [project.optional-dependencies]
37
+ yaml = ["pyyaml>=6.0.3"]
38
+ toon = ["python-toon>=0.1.3"]
39
+ render = ["pypdfium2>=5.1.0", "Pillow>=12.0.0"]
40
+
41
+ [project.scripts]
42
+ exstruct = "exstruct.cli.main:main"
43
+
44
+ [project.urls]
45
+ Homepage = "https://harumiweb.github.io/exstruct/"
46
+ Repository = "https://github.com/harumiWeb/exstruct"
47
+ Issues = "https://github.com/harumiWeb/exstruct/issues"
48
+ Documentation = "https://harumiweb.github.io/exstruct/"
49
+
50
+ [tool.coverage.run]
51
+ omit = [
52
+ "tests/*",
53
+ "*/test_*.py",
54
+ "*/gen_py/*",
55
+ ]
56
+
57
+ [tool.ruff]
58
+ target-version = "py311"
59
+ src = ["exstruct"]
60
+
61
+ select = [
62
+ "E", # pycodestyle errors
63
+ "W", # pycodestyle warnings
64
+ "F", # pyflakes
65
+ "I", # import sorting
66
+ "UP", # pyupgrade
67
+ "B", # flake8-bugbear
68
+ "N", # naming
69
+ "C90", # complexity
70
+ "A", # flake8-builtins
71
+ "ANN", # type annotations
72
+ ]
73
+
74
+ ignore = [
75
+ "E501", # 行長は許容(Excel JSON は長くなりがち)
76
+ "B008", # Pydantic の default_factory を誤検知するため
77
+ "ANN101", # self に型を要求されてしまうため
78
+ "ANN102", # cls も同様
79
+ ]
80
+
81
+ fix = true
82
+
83
+ # 型ヒントのスタイル
84
+ [tool.ruff.lint]
85
+ extend-select = ["ANN"]
86
+
87
+ # import の並び替え設定
88
+ [tool.ruff.isort]
89
+ combine-as-imports = true
90
+ known-first-party = ["exstruct"]
91
+ force-sort-within-sections = true
92
+
93
+ # 複雑度チェック(関数の最大複雑度)
94
+ [tool.ruff.mccabe]
95
+ max-complexity = 12
96
+
97
+ [tool.ruff.per-file-ignores]
98
+ "tests/**/*.py" = ["N802", "N803", "N806"]
99
+
100
+
101
+ [tool.mypy]
102
+ packages = ["exstruct"]
103
+ python_version = "3.11"
104
+
105
+ # 外部ライブラリは一切チェックしない
106
+ ignore_missing_imports = true
107
+
108
+ # 自作コードは厳密にチェックする
109
+ strict = true
110
+
111
+ # Pydantic v2 向け
112
+ plugins = ["pydantic.mypy"]
113
+
114
+ [tool.pytest.ini_options]
115
+ markers = [
116
+ "com: requires Excel COM (Windows + Excel)",
117
+ "render: requires Excel COM and pypdfium2; set RUN_RENDER_SMOKE=1 to enable",
118
+ ]