exstruct 0.2.90__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {exstruct-0.2.90 → exstruct-0.3.0}/PKG-INFO +1 -2
- {exstruct-0.2.90 → exstruct-0.3.0}/README.md +0 -1
- {exstruct-0.2.90 → exstruct-0.3.0}/pyproject.toml +128 -119
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/__init__.py +15 -11
- exstruct-0.3.0/src/exstruct/core/backends/__init__.py +7 -0
- exstruct-0.3.0/src/exstruct/core/backends/base.py +38 -0
- exstruct-0.3.0/src/exstruct/core/backends/com_backend.py +226 -0
- exstruct-0.3.0/src/exstruct/core/backends/openpyxl_backend.py +179 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/core/cells.py +254 -200
- exstruct-0.3.0/src/exstruct/core/integrate.py +52 -0
- exstruct-0.3.0/src/exstruct/core/logging_utils.py +16 -0
- exstruct-0.3.0/src/exstruct/core/modeling.py +74 -0
- exstruct-0.3.0/src/exstruct/core/pipeline.py +696 -0
- exstruct-0.3.0/src/exstruct/core/ranges.py +48 -0
- exstruct-0.3.0/src/exstruct/core/workbook.py +114 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/engine.py +10 -143
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/errors.py +46 -35
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/io/__init__.py +72 -132
- exstruct-0.3.0/src/exstruct/io/serialize.py +112 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/models/__init__.py +8 -4
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/render/__init__.py +3 -7
- exstruct-0.2.90/src/exstruct/core/integrate.py +0 -454
- {exstruct-0.2.90 → exstruct-0.3.0}/LICENSE +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/cli/availability.py +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/cli/main.py +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/core/__init__.py +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/core/charts.py +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/core/shapes.py +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/models/maps.py +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/models/types.py +0 -0
- {exstruct-0.2.90 → exstruct-0.3.0}/src/exstruct/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: exstruct
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
|
|
5
5
|
Keywords: excel,structure,data,exstruct
|
|
6
6
|
Author: harumiWeb
|
|
@@ -218,7 +218,6 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
218
218
|
(Screenshot below is the actual sample Excel sheet)
|
|
219
219
|

|
|
220
220
|
Sample workbook: `sample/sample.xlsx`
|
|
221
|
-
Sample workbook: `sample/sample.xlsx`
|
|
222
221
|
|
|
223
222
|
### 1. Input: Excel Sheet Overview
|
|
224
223
|
|
|
@@ -163,7 +163,6 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
|
|
|
163
163
|
(Screenshot below is the actual sample Excel sheet)
|
|
164
164
|

|
|
165
165
|
Sample workbook: `sample/sample.xlsx`
|
|
166
|
-
Sample workbook: `sample/sample.xlsx`
|
|
167
166
|
|
|
168
167
|
### 1. Input: Excel Sheet Overview
|
|
169
168
|
|
|
@@ -1,119 +1,128 @@
|
|
|
1
|
-
[project]
|
|
2
|
-
name = "exstruct"
|
|
3
|
-
version = "0.
|
|
4
|
-
description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
|
|
5
|
-
readme = "README.md"
|
|
6
|
-
license = { file = "LICENSE" }
|
|
7
|
-
keywords = ["excel", "structure", "data", "exstruct"]
|
|
8
|
-
authors = [
|
|
9
|
-
{ name = "harumiWeb"}
|
|
10
|
-
]
|
|
11
|
-
requires-python = ">=3.11"
|
|
12
|
-
dependencies = [
|
|
13
|
-
"numpy>=2.3.5",
|
|
14
|
-
"openpyxl>=3.1.5",
|
|
15
|
-
"pandas>=2.3.3",
|
|
16
|
-
"pydantic>=2.12.5",
|
|
17
|
-
"scipy>=1.16.3",
|
|
18
|
-
"xlwings>=0.33.16",
|
|
19
|
-
]
|
|
20
|
-
|
|
21
|
-
[build-system]
|
|
22
|
-
requires = ["uv_build>=0.8.4,<0.9.0"]
|
|
23
|
-
build-backend = "uv_build"
|
|
24
|
-
|
|
25
|
-
[dependency-groups]
|
|
26
|
-
dev = [
|
|
27
|
-
"mkdocs-material>=9.7.0",
|
|
28
|
-
"mkdocstrings-python>=2.0.1",
|
|
29
|
-
"mypy>=1.19.0",
|
|
30
|
-
"pre-commit>=4.5.0",
|
|
31
|
-
"pytest>=9.0.1",
|
|
32
|
-
"pytest-cov>=7.0.0",
|
|
33
|
-
"pytest-mock>=3.15.1",
|
|
34
|
-
"ruff>=0.14.8",
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
"
|
|
55
|
-
"*/
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
"
|
|
65
|
-
"
|
|
66
|
-
"
|
|
67
|
-
"
|
|
68
|
-
"
|
|
69
|
-
"
|
|
70
|
-
"
|
|
71
|
-
"
|
|
72
|
-
"
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
"
|
|
78
|
-
"
|
|
79
|
-
"
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
"
|
|
119
|
-
|
|
1
|
+
[project]
|
|
2
|
+
name = "exstruct"
|
|
3
|
+
version = "0.3.0"
|
|
4
|
+
description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = { file = "LICENSE" }
|
|
7
|
+
keywords = ["excel", "structure", "data", "exstruct"]
|
|
8
|
+
authors = [
|
|
9
|
+
{ name = "harumiWeb"}
|
|
10
|
+
]
|
|
11
|
+
requires-python = ">=3.11"
|
|
12
|
+
dependencies = [
|
|
13
|
+
"numpy>=2.3.5",
|
|
14
|
+
"openpyxl>=3.1.5",
|
|
15
|
+
"pandas>=2.3.3",
|
|
16
|
+
"pydantic>=2.12.5",
|
|
17
|
+
"scipy>=1.16.3",
|
|
18
|
+
"xlwings>=0.33.16",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
[build-system]
|
|
22
|
+
requires = ["uv_build>=0.8.4,<0.9.0"]
|
|
23
|
+
build-backend = "uv_build"
|
|
24
|
+
|
|
25
|
+
[dependency-groups]
|
|
26
|
+
dev = [
|
|
27
|
+
"mkdocs-material>=9.7.0",
|
|
28
|
+
"mkdocstrings-python>=2.0.1",
|
|
29
|
+
"mypy>=1.19.0",
|
|
30
|
+
"pre-commit>=4.5.0",
|
|
31
|
+
"pytest>=9.0.1",
|
|
32
|
+
"pytest-cov>=7.0.0",
|
|
33
|
+
"pytest-mock>=3.15.1",
|
|
34
|
+
"ruff>=0.14.8",
|
|
35
|
+
"taskipy>=1.14.1",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
yaml = ["pyyaml>=6.0.3"]
|
|
40
|
+
toon = ["python-toon>=0.1.3"]
|
|
41
|
+
render = ["pypdfium2>=5.1.0", "Pillow>=12.0.0"]
|
|
42
|
+
|
|
43
|
+
[project.scripts]
|
|
44
|
+
exstruct = "exstruct.cli.main:main"
|
|
45
|
+
|
|
46
|
+
[project.urls]
|
|
47
|
+
Homepage = "https://harumiweb.github.io/exstruct/"
|
|
48
|
+
Repository = "https://github.com/harumiWeb/exstruct"
|
|
49
|
+
Issues = "https://github.com/harumiWeb/exstruct/issues"
|
|
50
|
+
Documentation = "https://harumiweb.github.io/exstruct/"
|
|
51
|
+
|
|
52
|
+
[tool.coverage.run]
|
|
53
|
+
omit = [
|
|
54
|
+
"tests/*",
|
|
55
|
+
"*/test_*.py",
|
|
56
|
+
"*/gen_py/*",
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
[tool.ruff]
|
|
60
|
+
target-version = "py311"
|
|
61
|
+
src = ["exstruct"]
|
|
62
|
+
|
|
63
|
+
select = [
|
|
64
|
+
"E", # pycodestyle errors
|
|
65
|
+
"W", # pycodestyle warnings
|
|
66
|
+
"F", # pyflakes
|
|
67
|
+
"I", # import sorting
|
|
68
|
+
"UP", # pyupgrade
|
|
69
|
+
"B", # flake8-bugbear
|
|
70
|
+
"N", # naming
|
|
71
|
+
"C90", # complexity
|
|
72
|
+
"A", # flake8-builtins
|
|
73
|
+
"ANN", # type annotations
|
|
74
|
+
]
|
|
75
|
+
|
|
76
|
+
ignore = [
|
|
77
|
+
"E501", # 行長は許容(Excel JSON は長くなりがち)
|
|
78
|
+
"B008", # Pydantic の default_factory を誤検知するため
|
|
79
|
+
"ANN101", # self に型を要求されてしまうため
|
|
80
|
+
"ANN102", # cls も同様
|
|
81
|
+
]
|
|
82
|
+
|
|
83
|
+
fix = true
|
|
84
|
+
|
|
85
|
+
# 型ヒントのスタイル
|
|
86
|
+
[tool.ruff.lint]
|
|
87
|
+
extend-select = ["ANN"]
|
|
88
|
+
|
|
89
|
+
# import の並び替え設定
|
|
90
|
+
[tool.ruff.isort]
|
|
91
|
+
combine-as-imports = true
|
|
92
|
+
known-first-party = ["exstruct"]
|
|
93
|
+
force-sort-within-sections = true
|
|
94
|
+
|
|
95
|
+
# 複雑度チェック(関数の最大複雑度)
|
|
96
|
+
[tool.ruff.mccabe]
|
|
97
|
+
max-complexity = 12
|
|
98
|
+
|
|
99
|
+
[tool.ruff.per-file-ignores]
|
|
100
|
+
"tests/**/*.py" = ["N802", "N803", "N806"]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
[tool.mypy]
|
|
104
|
+
packages = ["exstruct"]
|
|
105
|
+
python_version = "3.11"
|
|
106
|
+
|
|
107
|
+
# 外部ライブラリは一切チェックしない
|
|
108
|
+
ignore_missing_imports = true
|
|
109
|
+
|
|
110
|
+
# 自作コードは厳密にチェックする
|
|
111
|
+
strict = true
|
|
112
|
+
|
|
113
|
+
# Pydantic v2 向け
|
|
114
|
+
plugins = ["pydantic.mypy"]
|
|
115
|
+
|
|
116
|
+
[tool.pytest.ini_options]
|
|
117
|
+
markers = [
|
|
118
|
+
"com: requires Excel COM (Windows + Excel)",
|
|
119
|
+
"render: requires Excel COM and pypdfium2; set RUN_RENDER_SMOKE=1 to enable",
|
|
120
|
+
]
|
|
121
|
+
|
|
122
|
+
[tool.taskipy.tasks]
|
|
123
|
+
ruff = "ruff check ."
|
|
124
|
+
ruff-fix = "ruff check . --fix"
|
|
125
|
+
mypy = "mypy src/exstruct --strict"
|
|
126
|
+
test = "pytest -vv --cov=exstruct --cov-report=term-missing --cov-report=xml"
|
|
127
|
+
docs = "mkdocs serve"
|
|
128
|
+
build-docs = "mkdocs build && python scripts/gen_json_schema.py && python scripts/gen_model_docs.py"
|
|
@@ -11,6 +11,7 @@ from .engine import (
|
|
|
11
11
|
DestinationOptions,
|
|
12
12
|
ExStructEngine,
|
|
13
13
|
FilterOptions,
|
|
14
|
+
FormatOptions,
|
|
14
15
|
OutputOptions,
|
|
15
16
|
StructOptions,
|
|
16
17
|
)
|
|
@@ -76,6 +77,7 @@ __all__ = [
|
|
|
76
77
|
"StructOptions",
|
|
77
78
|
"OutputOptions",
|
|
78
79
|
"FilterOptions",
|
|
80
|
+
"FormatOptions",
|
|
79
81
|
"DestinationOptions",
|
|
80
82
|
"ColorsOptions",
|
|
81
83
|
"serialize_workbook",
|
|
@@ -95,7 +97,7 @@ def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> Workboo
|
|
|
95
97
|
mode: "light" / "standard" / "verbose"
|
|
96
98
|
- light: cells + table detection only (no COM, shapes/charts empty). Print areas via openpyxl.
|
|
97
99
|
- standard: texted shapes + arrows + charts (COM if available), print areas included. Shape/chart size is kept but hidden by default in output.
|
|
98
|
-
- verbose: all shapes (including textless) with size, charts with size.
|
|
100
|
+
- verbose: all shapes (including textless) with size, charts with size, and colors_map.
|
|
99
101
|
|
|
100
102
|
Returns:
|
|
101
103
|
WorkbookData containing sheets, rows, shapes, charts, and print areas.
|
|
@@ -365,16 +367,18 @@ def process_excel(
|
|
|
365
367
|
engine = ExStructEngine(
|
|
366
368
|
options=StructOptions(mode=mode),
|
|
367
369
|
output=OutputOptions(
|
|
368
|
-
fmt=out_fmt,
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
370
|
+
format=FormatOptions(fmt=out_fmt, pretty=pretty, indent=indent),
|
|
371
|
+
filters=FilterOptions(
|
|
372
|
+
include_print_areas=None if mode == "light" else True,
|
|
373
|
+
include_shape_size=True if mode == "verbose" else False,
|
|
374
|
+
include_chart_size=True if mode == "verbose" else False,
|
|
375
|
+
),
|
|
376
|
+
destinations=DestinationOptions(
|
|
377
|
+
sheets_dir=sheets_dir,
|
|
378
|
+
print_areas_dir=print_areas_dir,
|
|
379
|
+
auto_page_breaks_dir=auto_page_breaks_dir,
|
|
380
|
+
stream=stream,
|
|
381
|
+
),
|
|
378
382
|
),
|
|
379
383
|
)
|
|
380
384
|
engine.process(
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Protocol
|
|
5
|
+
|
|
6
|
+
from ...models import CellRow, PrintArea
|
|
7
|
+
from ..cells import WorkbookColorsMap
|
|
8
|
+
|
|
9
|
+
CellData = dict[str, list[CellRow]]
|
|
10
|
+
PrintAreaData = dict[str, list[PrintArea]]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class BackendConfig:
|
|
15
|
+
"""Configuration options shared across backends.
|
|
16
|
+
|
|
17
|
+
Attributes:
|
|
18
|
+
include_default_background: Whether to include default background colors.
|
|
19
|
+
ignore_colors: Optional set of color keys to ignore.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
include_default_background: bool
|
|
23
|
+
ignore_colors: set[str] | None
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class Backend(Protocol):
|
|
27
|
+
"""Protocol for backend implementations."""
|
|
28
|
+
|
|
29
|
+
def extract_cells(self, *, include_links: bool) -> CellData:
|
|
30
|
+
"""Extract cell rows from the workbook."""
|
|
31
|
+
|
|
32
|
+
def extract_print_areas(self) -> PrintAreaData:
|
|
33
|
+
"""Extract print areas from the workbook."""
|
|
34
|
+
|
|
35
|
+
def extract_colors_map(
|
|
36
|
+
self, *, include_default_background: bool, ignore_colors: set[str] | None
|
|
37
|
+
) -> WorkbookColorsMap | None:
|
|
38
|
+
"""Extract colors map from the workbook."""
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"""COM backend for Excel workbook extraction via xlwings."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import logging
|
|
7
|
+
from typing import Any, cast
|
|
8
|
+
|
|
9
|
+
import xlwings as xw
|
|
10
|
+
|
|
11
|
+
from ...models import PrintArea
|
|
12
|
+
from ..cells import WorkbookColorsMap, extract_sheet_colors_map_com
|
|
13
|
+
from ..ranges import parse_range_zero_based
|
|
14
|
+
from .base import PrintAreaData
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class ComBackend:
|
|
21
|
+
"""COM-based backend for extraction tasks.
|
|
22
|
+
|
|
23
|
+
Attributes:
|
|
24
|
+
workbook: xlwings workbook instance.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
workbook: xw.Book
|
|
28
|
+
|
|
29
|
+
def extract_print_areas(self) -> PrintAreaData:
|
|
30
|
+
"""Extract print areas per sheet via xlwings/COM.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
Mapping of sheet name to print area list.
|
|
34
|
+
"""
|
|
35
|
+
areas: PrintAreaData = {}
|
|
36
|
+
for sheet in self.workbook.sheets:
|
|
37
|
+
raw = ""
|
|
38
|
+
try:
|
|
39
|
+
raw = sheet.api.PageSetup.PrintArea or ""
|
|
40
|
+
except Exception as exc:
|
|
41
|
+
logger.warning(
|
|
42
|
+
"Failed to read print area via COM for sheet '%s'. (%r)",
|
|
43
|
+
sheet.name,
|
|
44
|
+
exc,
|
|
45
|
+
)
|
|
46
|
+
if not raw:
|
|
47
|
+
continue
|
|
48
|
+
for part in str(raw).split(","):
|
|
49
|
+
parsed = _parse_print_area_range(part)
|
|
50
|
+
if not parsed:
|
|
51
|
+
continue
|
|
52
|
+
r1, c1, r2, c2 = parsed
|
|
53
|
+
areas.setdefault(sheet.name, []).append(
|
|
54
|
+
PrintArea(r1=r1 + 1, c1=c1, r2=r2 + 1, c2=c2)
|
|
55
|
+
)
|
|
56
|
+
return areas
|
|
57
|
+
|
|
58
|
+
def extract_colors_map(
|
|
59
|
+
self, *, include_default_background: bool, ignore_colors: set[str] | None
|
|
60
|
+
) -> WorkbookColorsMap | None:
|
|
61
|
+
"""Extract colors_map via COM; logs and skips on failure.
|
|
62
|
+
|
|
63
|
+
Args:
|
|
64
|
+
include_default_background: Whether to include default backgrounds.
|
|
65
|
+
ignore_colors: Optional set of color keys to ignore.
|
|
66
|
+
|
|
67
|
+
Returns:
|
|
68
|
+
WorkbookColorsMap or None when extraction fails.
|
|
69
|
+
"""
|
|
70
|
+
try:
|
|
71
|
+
return extract_sheet_colors_map_com(
|
|
72
|
+
self.workbook,
|
|
73
|
+
include_default_background=include_default_background,
|
|
74
|
+
ignore_colors=ignore_colors,
|
|
75
|
+
)
|
|
76
|
+
except Exception as exc:
|
|
77
|
+
logger.warning(
|
|
78
|
+
"COM color map extraction failed; falling back to openpyxl. (%r)",
|
|
79
|
+
exc,
|
|
80
|
+
)
|
|
81
|
+
return None
|
|
82
|
+
|
|
83
|
+
def extract_auto_page_breaks(self) -> PrintAreaData:
|
|
84
|
+
"""Compute auto page-break rectangles per sheet using Excel COM.
|
|
85
|
+
|
|
86
|
+
Returns:
|
|
87
|
+
Mapping of sheet name to auto page-break areas.
|
|
88
|
+
"""
|
|
89
|
+
results: PrintAreaData = {}
|
|
90
|
+
for sheet in self.workbook.sheets:
|
|
91
|
+
ws_api: Any | None = None
|
|
92
|
+
original_display: bool | None = None
|
|
93
|
+
failed = False
|
|
94
|
+
try:
|
|
95
|
+
ws_api = cast(Any, sheet.api)
|
|
96
|
+
original_display = ws_api.DisplayPageBreaks
|
|
97
|
+
ws_api.DisplayPageBreaks = True
|
|
98
|
+
print_area = ws_api.PageSetup.PrintArea or ws_api.UsedRange.Address
|
|
99
|
+
parts_raw = _split_csv_respecting_quotes(str(print_area))
|
|
100
|
+
area_parts: list[str] = []
|
|
101
|
+
for part in parts_raw:
|
|
102
|
+
rng = _normalize_area_for_sheet(part, sheet.name)
|
|
103
|
+
if rng:
|
|
104
|
+
area_parts.append(rng)
|
|
105
|
+
hpb = cast(Any, ws_api.HPageBreaks)
|
|
106
|
+
vpb = cast(Any, ws_api.VPageBreaks)
|
|
107
|
+
h_break_rows = [
|
|
108
|
+
hpb.Item(i).Location.Row for i in range(1, int(hpb.Count) + 1)
|
|
109
|
+
]
|
|
110
|
+
v_break_cols = [
|
|
111
|
+
vpb.Item(i).Location.Column for i in range(1, int(vpb.Count) + 1)
|
|
112
|
+
]
|
|
113
|
+
for addr in area_parts:
|
|
114
|
+
range_obj = cast(Any, ws_api.Range(addr))
|
|
115
|
+
min_row = int(range_obj.Row)
|
|
116
|
+
max_row = min_row + int(range_obj.Rows.Count) - 1
|
|
117
|
+
min_col = int(range_obj.Column)
|
|
118
|
+
max_col = min_col + int(range_obj.Columns.Count) - 1
|
|
119
|
+
rows = (
|
|
120
|
+
[min_row]
|
|
121
|
+
+ [r for r in h_break_rows if min_row < r <= max_row]
|
|
122
|
+
+ [max_row + 1]
|
|
123
|
+
)
|
|
124
|
+
cols = (
|
|
125
|
+
[min_col]
|
|
126
|
+
+ [c for c in v_break_cols if min_col < c <= max_col]
|
|
127
|
+
+ [max_col + 1]
|
|
128
|
+
)
|
|
129
|
+
for i in range(len(rows) - 1):
|
|
130
|
+
r1, r2 = rows[i], rows[i + 1] - 1
|
|
131
|
+
for j in range(len(cols) - 1):
|
|
132
|
+
c1, c2 = cols[j], cols[j + 1] - 1
|
|
133
|
+
results.setdefault(sheet.name, []).append(
|
|
134
|
+
PrintArea(r1=r1, c1=c1 - 1, r2=r2, c2=c2 - 1)
|
|
135
|
+
)
|
|
136
|
+
except Exception as exc:
|
|
137
|
+
logger.warning(
|
|
138
|
+
"Failed to extract auto page breaks via COM for sheet '%s'. (%r)",
|
|
139
|
+
sheet.name,
|
|
140
|
+
exc,
|
|
141
|
+
)
|
|
142
|
+
failed = True
|
|
143
|
+
finally:
|
|
144
|
+
if ws_api is not None and original_display is not None:
|
|
145
|
+
try:
|
|
146
|
+
ws_api.DisplayPageBreaks = original_display
|
|
147
|
+
except Exception as exc:
|
|
148
|
+
logger.debug(
|
|
149
|
+
"Failed to restore DisplayPageBreaks for sheet '%s'. (%r)",
|
|
150
|
+
sheet.name,
|
|
151
|
+
exc,
|
|
152
|
+
)
|
|
153
|
+
if failed:
|
|
154
|
+
continue
|
|
155
|
+
return results
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _parse_print_area_range(range_str: str) -> tuple[int, int, int, int] | None:
|
|
159
|
+
"""Parse an Excel range string into zero-based coordinates.
|
|
160
|
+
|
|
161
|
+
Args:
|
|
162
|
+
range_str: Excel range string.
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
Zero-based (r1, c1, r2, c2) tuple or None on failure.
|
|
166
|
+
"""
|
|
167
|
+
bounds = parse_range_zero_based(range_str)
|
|
168
|
+
if bounds is None:
|
|
169
|
+
return None
|
|
170
|
+
return (bounds.r1, bounds.c1, bounds.r2, bounds.c2)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _normalize_area_for_sheet(part: str, ws_name: str) -> str | None:
|
|
174
|
+
"""Strip sheet name from a range part when it matches the target sheet.
|
|
175
|
+
|
|
176
|
+
Args:
|
|
177
|
+
part: Raw range string part.
|
|
178
|
+
ws_name: Target worksheet name.
|
|
179
|
+
|
|
180
|
+
Returns:
|
|
181
|
+
Range without sheet prefix, or None if not matching.
|
|
182
|
+
"""
|
|
183
|
+
s = part.strip()
|
|
184
|
+
if "!" not in s:
|
|
185
|
+
return s
|
|
186
|
+
sheet, rng = s.rsplit("!", 1)
|
|
187
|
+
sheet = sheet.strip()
|
|
188
|
+
if sheet.startswith("'") and sheet.endswith("'"):
|
|
189
|
+
sheet = sheet[1:-1].replace("''", "'")
|
|
190
|
+
return rng if sheet == ws_name else None
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _split_csv_respecting_quotes(raw: str) -> list[str]:
|
|
194
|
+
"""Split a CSV-like string while keeping commas inside single quotes intact.
|
|
195
|
+
|
|
196
|
+
Args:
|
|
197
|
+
raw: Raw CSV-like string.
|
|
198
|
+
|
|
199
|
+
Returns:
|
|
200
|
+
List of split parts.
|
|
201
|
+
"""
|
|
202
|
+
parts: list[str] = []
|
|
203
|
+
buf: list[str] = []
|
|
204
|
+
in_quote = False
|
|
205
|
+
i = 0
|
|
206
|
+
while i < len(raw):
|
|
207
|
+
ch = raw[i]
|
|
208
|
+
if ch == "'":
|
|
209
|
+
if in_quote and i + 1 < len(raw) and raw[i + 1] == "'":
|
|
210
|
+
buf.append("''")
|
|
211
|
+
i += 2
|
|
212
|
+
continue
|
|
213
|
+
in_quote = not in_quote
|
|
214
|
+
buf.append(ch)
|
|
215
|
+
i += 1
|
|
216
|
+
continue
|
|
217
|
+
if ch == "," and not in_quote:
|
|
218
|
+
parts.append("".join(buf).strip())
|
|
219
|
+
buf = []
|
|
220
|
+
i += 1
|
|
221
|
+
continue
|
|
222
|
+
buf.append(ch)
|
|
223
|
+
i += 1
|
|
224
|
+
if buf:
|
|
225
|
+
parts.append("".join(buf).strip())
|
|
226
|
+
return [p for p in parts if p]
|