exstruct 0.2.90__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {exstruct-0.2.90 → exstruct-0.3.1}/PKG-INFO +26 -6
  2. {exstruct-0.2.90 → exstruct-0.3.1}/README.md +25 -5
  3. {exstruct-0.2.90 → exstruct-0.3.1}/pyproject.toml +15 -1
  4. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/__init__.py +15 -11
  5. exstruct-0.3.1/src/exstruct/core/backends/__init__.py +7 -0
  6. exstruct-0.3.1/src/exstruct/core/backends/base.py +38 -0
  7. exstruct-0.3.1/src/exstruct/core/backends/com_backend.py +226 -0
  8. exstruct-0.3.1/src/exstruct/core/backends/openpyxl_backend.py +179 -0
  9. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/core/cells.py +254 -200
  10. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/core/charts.py +243 -241
  11. exstruct-0.3.1/src/exstruct/core/integrate.py +52 -0
  12. exstruct-0.3.1/src/exstruct/core/logging_utils.py +16 -0
  13. exstruct-0.3.1/src/exstruct/core/modeling.py +83 -0
  14. exstruct-0.3.1/src/exstruct/core/pipeline.py +696 -0
  15. exstruct-0.3.1/src/exstruct/core/ranges.py +48 -0
  16. exstruct-0.3.1/src/exstruct/core/shapes.py +521 -0
  17. exstruct-0.3.1/src/exstruct/core/workbook.py +114 -0
  18. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/engine.py +10 -143
  19. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/errors.py +12 -1
  20. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/io/__init__.py +130 -138
  21. exstruct-0.3.1/src/exstruct/io/serialize.py +112 -0
  22. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/models/__init__.py +42 -9
  23. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/render/__init__.py +3 -7
  24. exstruct-0.2.90/src/exstruct/core/integrate.py +0 -454
  25. exstruct-0.2.90/src/exstruct/core/shapes.py +0 -275
  26. {exstruct-0.2.90 → exstruct-0.3.1}/LICENSE +0 -0
  27. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/cli/availability.py +0 -0
  28. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/cli/main.py +0 -0
  29. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/core/__init__.py +0 -0
  30. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/models/maps.py +0 -0
  31. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/models/types.py +0 -0
  32. {exstruct-0.2.90 → exstruct-0.3.1}/src/exstruct/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: exstruct
3
- Version: 0.2.90
3
+ Version: 0.3.1
4
4
  Summary: Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines
5
5
  Keywords: excel,structure,data,exstruct
6
6
  Author: harumiWeb
@@ -55,18 +55,18 @@ Description-Content-Type: text/markdown
55
55
 
56
56
  # ExStruct — Excel Structured Extraction Engine
57
57
 
58
- [![PyPI version](https://badge.fury.io/py/exstruct.svg)](https://pypi.org/project/exstruct/) [![PyPI Downloads](https://static.pepy.tech/personalized-badge/exstruct?period=total&units=INTERNATIONAL_SYSTEM&left_color=BLACK&right_color=GREEN&left_text=downloads)](https://pepy.tech/projects/exstruct) ![Licence: BSD-3-Clause](https://img.shields.io/badge/license-BSD--3--Clause-blue?style=flat-square) [![pytest](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml/badge.svg)](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml) [![Codacy Badge](https://app.codacy.com/project/badge/Grade/e081cb4f634e4175b259eb7c34f54f60)](https://app.codacy.com/gh/harumiWeb/exstruct/dashboard?utm_source=gh&utm_medium=referral&utm_content=&utm_campaign=Badge_grade)
58
+ [![PyPI version](https://badge.fury.io/py/exstruct.svg)](https://pypi.org/project/exstruct/) [![PyPI Downloads](https://static.pepy.tech/personalized-badge/exstruct?period=total&units=INTERNATIONAL_SYSTEM&left_color=BLACK&right_color=GREEN&left_text=downloads)](https://pepy.tech/projects/exstruct) ![Licence: BSD-3-Clause](https://img.shields.io/badge/license-BSD--3--Clause-blue?style=flat-square) [![pytest](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml/badge.svg)](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml) [![Codacy Badge](https://app.codacy.com/project/badge/Grade/e081cb4f634e4175b259eb7c34f54f60)](https://app.codacy.com/gh/harumiWeb/exstruct/dashboard?utm_source=gh&utm_medium=referral&utm_content=&utm_campaign=Badge_grade) [![codecov](https://codecov.io/gh/harumiWeb/exstruct/graph/badge.svg?token=2XI1O8TTA9)](https://codecov.io/gh/harumiWeb/exstruct)
59
59
 
60
60
  ![ExStruct Image](/docs/assets/icon.webp)
61
61
 
62
- ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, print areas/views, auto page-break areas, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
62
+ ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, smartart, print areas/views, auto page-break areas, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
63
63
 
64
64
  [日本版 README](README.ja.md)
65
65
 
66
66
  ## Features
67
67
 
68
- - **Excel → Structured JSON**: cells, shapes, charts, table candidates, print areas/views, and auto page-break areas per sheet.
69
- - **Output modes**: `light` (cells + table candidates + print areas; no COM, shapes/charts empty), `standard` (texted shapes + arrows, charts, print areas), `verbose` (all shapes with width/height, charts with size, print areas). Verbose also emits cell hyperlinks and `colors_map`. Size output is flag-controlled.
68
+ - **Excel → Structured JSON**: cells, shapes, charts, smartart, table candidates, print areas/views, and auto page-break areas per sheet.
69
+ - **Output modes**: `light` (cells + table candidates + print areas; no COM, shapes/charts empty), `standard` (texted shapes + arrows, charts, smartart, print areas), `verbose` (all shapes with width/height, charts with size, print areas). Verbose also emits cell hyperlinks and `colors_map`. Size output is flag-controlled.
70
70
  - **Auto page-break export (COM only)**: capture Excel-computed auto page breaks and write per-area JSON/YAML/TOON when requested (CLI option appears only when COM is available).
71
71
  - **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
72
72
  - **Table detection tuning**: adjust heuristics at runtime via API.
@@ -218,7 +218,6 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
218
218
  (Screenshot below is the actual sample Excel sheet)
219
219
  ![Sample Excel](/docs/assets/demo_sheet.png)
220
220
  Sample workbook: `sample/sample.xlsx`
221
- Sample workbook: `sample/sample.xlsx`
222
221
 
223
222
  ### 1. Input: Excel Sheet Overview
224
223
 
@@ -436,6 +435,27 @@ This project is suitable for teams that:
436
435
  - Use CLI `--auto-page-breaks-dir` (COM only), `DestinationOptions.auto_page_breaks_dir` (preferred), or `export_auto_page_breaks(...)` to write per-auto-page-break files; the API raises `ValueError` if no auto page breaks exist.
437
436
  - `PrintAreaView` includes rows and table candidates inside the area, plus shapes/charts that overlap the area (size-less shapes are treated as points). `normalize=True` rebases row/col indices to the area origin.
438
437
 
438
+ ## Architecture
439
+
440
+ ExStruct uses a pipeline-based architecture that separates
441
+ extraction strategy (Backend) from orchestration (Pipeline)
442
+ and semantic modeling.
443
+
444
+ → See: [docs/architecture/pipeline.md](docs/architecture/pipeline.md)
445
+
446
+ ## Contributing
447
+
448
+ If you plan to extend ExStruct internals,
449
+ please read the contributor architecture guide.
450
+
451
+ → [docs/contributors/architecture.md](docs/contributors/architecture.md)
452
+
453
+ ## Note on coverage
454
+
455
+ The cell-structure inference logic (cells.py) relies on heuristic rules
456
+ and Excel-specific behaviors. Full coverage is intentionally not pursued,
457
+ as exhaustive testing would not reflect real-world reliability.
458
+
439
459
  ## License
440
460
 
441
461
  BSD-3-Clause. See `LICENSE` for details.
@@ -1,17 +1,17 @@
1
1
  # ExStruct — Excel Structured Extraction Engine
2
2
 
3
- [![PyPI version](https://badge.fury.io/py/exstruct.svg)](https://pypi.org/project/exstruct/) [![PyPI Downloads](https://static.pepy.tech/personalized-badge/exstruct?period=total&units=INTERNATIONAL_SYSTEM&left_color=BLACK&right_color=GREEN&left_text=downloads)](https://pepy.tech/projects/exstruct) ![Licence: BSD-3-Clause](https://img.shields.io/badge/license-BSD--3--Clause-blue?style=flat-square) [![pytest](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml/badge.svg)](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml) [![Codacy Badge](https://app.codacy.com/project/badge/Grade/e081cb4f634e4175b259eb7c34f54f60)](https://app.codacy.com/gh/harumiWeb/exstruct/dashboard?utm_source=gh&utm_medium=referral&utm_content=&utm_campaign=Badge_grade)
3
+ [![PyPI version](https://badge.fury.io/py/exstruct.svg)](https://pypi.org/project/exstruct/) [![PyPI Downloads](https://static.pepy.tech/personalized-badge/exstruct?period=total&units=INTERNATIONAL_SYSTEM&left_color=BLACK&right_color=GREEN&left_text=downloads)](https://pepy.tech/projects/exstruct) ![Licence: BSD-3-Clause](https://img.shields.io/badge/license-BSD--3--Clause-blue?style=flat-square) [![pytest](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml/badge.svg)](https://github.com/harumiWeb/exstruct/actions/workflows/pytest.yml) [![Codacy Badge](https://app.codacy.com/project/badge/Grade/e081cb4f634e4175b259eb7c34f54f60)](https://app.codacy.com/gh/harumiWeb/exstruct/dashboard?utm_source=gh&utm_medium=referral&utm_content=&utm_campaign=Badge_grade) [![codecov](https://codecov.io/gh/harumiWeb/exstruct/graph/badge.svg?token=2XI1O8TTA9)](https://codecov.io/gh/harumiWeb/exstruct)
4
4
 
5
5
  ![ExStruct Image](/docs/assets/icon.webp)
6
6
 
7
- ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, print areas/views, auto page-break areas, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
7
+ ExStruct reads Excel workbooks and outputs structured data (cells, table candidates, shapes, charts, smartart, print areas/views, auto page-break areas, hyperlinks) as JSON by default, with optional YAML/TOON formats. It targets both COM/Excel environments (rich extraction) and non-COM environments (cells + table candidates + print areas), with tunable detection heuristics and multiple output modes to fit LLM/RAG pipelines.
8
8
 
9
9
  [日本版 README](README.ja.md)
10
10
 
11
11
  ## Features
12
12
 
13
- - **Excel → Structured JSON**: cells, shapes, charts, table candidates, print areas/views, and auto page-break areas per sheet.
14
- - **Output modes**: `light` (cells + table candidates + print areas; no COM, shapes/charts empty), `standard` (texted shapes + arrows, charts, print areas), `verbose` (all shapes with width/height, charts with size, print areas). Verbose also emits cell hyperlinks and `colors_map`. Size output is flag-controlled.
13
+ - **Excel → Structured JSON**: cells, shapes, charts, smartart, table candidates, print areas/views, and auto page-break areas per sheet.
14
+ - **Output modes**: `light` (cells + table candidates + print areas; no COM, shapes/charts empty), `standard` (texted shapes + arrows, charts, smartart, print areas), `verbose` (all shapes with width/height, charts with size, print areas). Verbose also emits cell hyperlinks and `colors_map`. Size output is flag-controlled.
15
15
  - **Auto page-break export (COM only)**: capture Excel-computed auto page breaks and write per-area JSON/YAML/TOON when requested (CLI option appears only when COM is available).
16
16
  - **Formats**: JSON (compact by default, `--pretty` available), YAML, TOON (optional dependencies).
17
17
  - **Table detection tuning**: adjust heuristics at runtime via API.
@@ -163,7 +163,6 @@ To show how well exstruct can structure Excel, we parse a workbook that combines
163
163
  (Screenshot below is the actual sample Excel sheet)
164
164
  ![Sample Excel](/docs/assets/demo_sheet.png)
165
165
  Sample workbook: `sample/sample.xlsx`
166
- Sample workbook: `sample/sample.xlsx`
167
166
 
168
167
  ### 1. Input: Excel Sheet Overview
169
168
 
@@ -381,6 +380,27 @@ This project is suitable for teams that:
381
380
  - Use CLI `--auto-page-breaks-dir` (COM only), `DestinationOptions.auto_page_breaks_dir` (preferred), or `export_auto_page_breaks(...)` to write per-auto-page-break files; the API raises `ValueError` if no auto page breaks exist.
382
381
  - `PrintAreaView` includes rows and table candidates inside the area, plus shapes/charts that overlap the area (size-less shapes are treated as points). `normalize=True` rebases row/col indices to the area origin.
383
382
 
383
+ ## Architecture
384
+
385
+ ExStruct uses a pipeline-based architecture that separates
386
+ extraction strategy (Backend) from orchestration (Pipeline)
387
+ and semantic modeling.
388
+
389
+ → See: [docs/architecture/pipeline.md](docs/architecture/pipeline.md)
390
+
391
+ ## Contributing
392
+
393
+ If you plan to extend ExStruct internals,
394
+ please read the contributor architecture guide.
395
+
396
+ → [docs/contributors/architecture.md](docs/contributors/architecture.md)
397
+
398
+ ## Note on coverage
399
+
400
+ The cell-structure inference logic (cells.py) relies on heuristic rules
401
+ and Excel-specific behaviors. Full coverage is intentionally not pursued,
402
+ as exhaustive testing would not reflect real-world reliability.
403
+
384
404
  ## License
385
405
 
386
406
  BSD-3-Clause. See `LICENSE` for details.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "exstruct"
3
- version = "0.2.90"
3
+ version = "0.3.1"
4
4
  description = "Excel to structured JSON (tables, shapes, charts) for LLM/RAG pipelines"
5
5
  readme = "README.md"
6
6
  license = { file = "LICENSE" }
@@ -24,6 +24,7 @@ build-backend = "uv_build"
24
24
 
25
25
  [dependency-groups]
26
26
  dev = [
27
+ "codecov-cli>=11.2.6",
27
28
  "mkdocs-material>=9.7.0",
28
29
  "mkdocstrings-python>=2.0.1",
29
30
  "mypy>=1.19.0",
@@ -32,6 +33,7 @@ dev = [
32
33
  "pytest-cov>=7.0.0",
33
34
  "pytest-mock>=3.15.1",
34
35
  "ruff>=0.14.8",
36
+ "taskipy>=1.14.1",
35
37
  ]
36
38
 
37
39
  [project.optional-dependencies]
@@ -117,3 +119,15 @@ markers = [
117
119
  "com: requires Excel COM (Windows + Excel)",
118
120
  "render: requires Excel COM and pypdfium2; set RUN_RENDER_SMOKE=1 to enable",
119
121
  ]
122
+
123
+ [tool.taskipy.tasks]
124
+ ruff = "ruff check ."
125
+ ruff-fix = "ruff check . --fix"
126
+ mypy = "mypy src/exstruct --strict"
127
+ test = "pytest -vv --cov=exstruct --cov-report=term-missing --cov-report=xml" # uv sync --extra render --extra toon
128
+ test-unit = "pytest -vv -m \"not com and not render\" --cov=exstruct --cov-report=term-missing --cov-report=xml"
129
+ test-com = "pytest -vv -m \"com\" --cov=exstruct --cov-report=term-missing --cov-report=xml"
130
+ codecov-unit = "codecov-cli upload-process -f coverage.xml -F unit -C %CODECOV_SHA% -t %CODECOV_TOKEN%"
131
+ codecov-com = "codecov-cli upload-process -f coverage.xml -F com -C %CODECOV_SHA% -t %CODECOV_TOKEN%"
132
+ docs = "mkdocs serve"
133
+ build-docs = "mkdocs build && python scripts/gen_json_schema.py && python scripts/gen_model_docs.py"
@@ -11,6 +11,7 @@ from .engine import (
11
11
  DestinationOptions,
12
12
  ExStructEngine,
13
13
  FilterOptions,
14
+ FormatOptions,
14
15
  OutputOptions,
15
16
  StructOptions,
16
17
  )
@@ -76,6 +77,7 @@ __all__ = [
76
77
  "StructOptions",
77
78
  "OutputOptions",
78
79
  "FilterOptions",
80
+ "FormatOptions",
79
81
  "DestinationOptions",
80
82
  "ColorsOptions",
81
83
  "serialize_workbook",
@@ -95,7 +97,7 @@ def extract(file_path: str | Path, mode: ExtractionMode = "standard") -> Workboo
95
97
  mode: "light" / "standard" / "verbose"
96
98
  - light: cells + table detection only (no COM, shapes/charts empty). Print areas via openpyxl.
97
99
  - standard: texted shapes + arrows + charts (COM if available), print areas included. Shape/chart size is kept but hidden by default in output.
98
- - verbose: all shapes (including textless) with size, charts with size.
100
+ - verbose: all shapes (including textless) with size, charts with size, and colors_map.
99
101
 
100
102
  Returns:
101
103
  WorkbookData containing sheets, rows, shapes, charts, and print areas.
@@ -365,16 +367,18 @@ def process_excel(
365
367
  engine = ExStructEngine(
366
368
  options=StructOptions(mode=mode),
367
369
  output=OutputOptions(
368
- fmt=out_fmt,
369
- pretty=pretty,
370
- indent=indent,
371
- sheets_dir=sheets_dir,
372
- print_areas_dir=print_areas_dir,
373
- auto_page_breaks_dir=auto_page_breaks_dir,
374
- include_print_areas=None if mode == "light" else True,
375
- include_shape_size=True if mode == "verbose" else False,
376
- include_chart_size=True if mode == "verbose" else False,
377
- stream=stream,
370
+ format=FormatOptions(fmt=out_fmt, pretty=pretty, indent=indent),
371
+ filters=FilterOptions(
372
+ include_print_areas=None if mode == "light" else True,
373
+ include_shape_size=True if mode == "verbose" else False,
374
+ include_chart_size=True if mode == "verbose" else False,
375
+ ),
376
+ destinations=DestinationOptions(
377
+ sheets_dir=sheets_dir,
378
+ print_areas_dir=print_areas_dir,
379
+ auto_page_breaks_dir=auto_page_breaks_dir,
380
+ stream=stream,
381
+ ),
378
382
  ),
379
383
  )
380
384
  engine.process(
@@ -0,0 +1,7 @@
1
+ from __future__ import annotations
2
+
3
+ from .base import Backend
4
+ from .com_backend import ComBackend
5
+ from .openpyxl_backend import OpenpyxlBackend
6
+
7
+ __all__ = ["Backend", "ComBackend", "OpenpyxlBackend"]
@@ -0,0 +1,38 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from typing import Protocol
5
+
6
+ from ...models import CellRow, PrintArea
7
+ from ..cells import WorkbookColorsMap
8
+
9
+ CellData = dict[str, list[CellRow]]
10
+ PrintAreaData = dict[str, list[PrintArea]]
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class BackendConfig:
15
+ """Configuration options shared across backends.
16
+
17
+ Attributes:
18
+ include_default_background: Whether to include default background colors.
19
+ ignore_colors: Optional set of color keys to ignore.
20
+ """
21
+
22
+ include_default_background: bool
23
+ ignore_colors: set[str] | None
24
+
25
+
26
+ class Backend(Protocol):
27
+ """Protocol for backend implementations."""
28
+
29
+ def extract_cells(self, *, include_links: bool) -> CellData:
30
+ """Extract cell rows from the workbook."""
31
+
32
+ def extract_print_areas(self) -> PrintAreaData:
33
+ """Extract print areas from the workbook."""
34
+
35
+ def extract_colors_map(
36
+ self, *, include_default_background: bool, ignore_colors: set[str] | None
37
+ ) -> WorkbookColorsMap | None:
38
+ """Extract colors map from the workbook."""
@@ -0,0 +1,226 @@
1
+ """COM backend for Excel workbook extraction via xlwings."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import logging
7
+ from typing import Any, cast
8
+
9
+ import xlwings as xw
10
+
11
+ from ...models import PrintArea
12
+ from ..cells import WorkbookColorsMap, extract_sheet_colors_map_com
13
+ from ..ranges import parse_range_zero_based
14
+ from .base import PrintAreaData
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class ComBackend:
21
+ """COM-based backend for extraction tasks.
22
+
23
+ Attributes:
24
+ workbook: xlwings workbook instance.
25
+ """
26
+
27
+ workbook: xw.Book
28
+
29
+ def extract_print_areas(self) -> PrintAreaData:
30
+ """Extract print areas per sheet via xlwings/COM.
31
+
32
+ Returns:
33
+ Mapping of sheet name to print area list.
34
+ """
35
+ areas: PrintAreaData = {}
36
+ for sheet in self.workbook.sheets:
37
+ raw = ""
38
+ try:
39
+ raw = sheet.api.PageSetup.PrintArea or ""
40
+ except Exception as exc:
41
+ logger.warning(
42
+ "Failed to read print area via COM for sheet '%s'. (%r)",
43
+ sheet.name,
44
+ exc,
45
+ )
46
+ if not raw:
47
+ continue
48
+ for part in str(raw).split(","):
49
+ parsed = _parse_print_area_range(part)
50
+ if not parsed:
51
+ continue
52
+ r1, c1, r2, c2 = parsed
53
+ areas.setdefault(sheet.name, []).append(
54
+ PrintArea(r1=r1 + 1, c1=c1, r2=r2 + 1, c2=c2)
55
+ )
56
+ return areas
57
+
58
+ def extract_colors_map(
59
+ self, *, include_default_background: bool, ignore_colors: set[str] | None
60
+ ) -> WorkbookColorsMap | None:
61
+ """Extract colors_map via COM; logs and skips on failure.
62
+
63
+ Args:
64
+ include_default_background: Whether to include default backgrounds.
65
+ ignore_colors: Optional set of color keys to ignore.
66
+
67
+ Returns:
68
+ WorkbookColorsMap or None when extraction fails.
69
+ """
70
+ try:
71
+ return extract_sheet_colors_map_com(
72
+ self.workbook,
73
+ include_default_background=include_default_background,
74
+ ignore_colors=ignore_colors,
75
+ )
76
+ except Exception as exc:
77
+ logger.warning(
78
+ "COM color map extraction failed; falling back to openpyxl. (%r)",
79
+ exc,
80
+ )
81
+ return None
82
+
83
+ def extract_auto_page_breaks(self) -> PrintAreaData:
84
+ """Compute auto page-break rectangles per sheet using Excel COM.
85
+
86
+ Returns:
87
+ Mapping of sheet name to auto page-break areas.
88
+ """
89
+ results: PrintAreaData = {}
90
+ for sheet in self.workbook.sheets:
91
+ ws_api: Any | None = None
92
+ original_display: bool | None = None
93
+ failed = False
94
+ try:
95
+ ws_api = cast(Any, sheet.api)
96
+ original_display = ws_api.DisplayPageBreaks
97
+ ws_api.DisplayPageBreaks = True
98
+ print_area = ws_api.PageSetup.PrintArea or ws_api.UsedRange.Address
99
+ parts_raw = _split_csv_respecting_quotes(str(print_area))
100
+ area_parts: list[str] = []
101
+ for part in parts_raw:
102
+ rng = _normalize_area_for_sheet(part, sheet.name)
103
+ if rng:
104
+ area_parts.append(rng)
105
+ hpb = cast(Any, ws_api.HPageBreaks)
106
+ vpb = cast(Any, ws_api.VPageBreaks)
107
+ h_break_rows = [
108
+ hpb.Item(i).Location.Row for i in range(1, int(hpb.Count) + 1)
109
+ ]
110
+ v_break_cols = [
111
+ vpb.Item(i).Location.Column for i in range(1, int(vpb.Count) + 1)
112
+ ]
113
+ for addr in area_parts:
114
+ range_obj = cast(Any, ws_api.Range(addr))
115
+ min_row = int(range_obj.Row)
116
+ max_row = min_row + int(range_obj.Rows.Count) - 1
117
+ min_col = int(range_obj.Column)
118
+ max_col = min_col + int(range_obj.Columns.Count) - 1
119
+ rows = (
120
+ [min_row]
121
+ + [r for r in h_break_rows if min_row < r <= max_row]
122
+ + [max_row + 1]
123
+ )
124
+ cols = (
125
+ [min_col]
126
+ + [c for c in v_break_cols if min_col < c <= max_col]
127
+ + [max_col + 1]
128
+ )
129
+ for i in range(len(rows) - 1):
130
+ r1, r2 = rows[i], rows[i + 1] - 1
131
+ for j in range(len(cols) - 1):
132
+ c1, c2 = cols[j], cols[j + 1] - 1
133
+ results.setdefault(sheet.name, []).append(
134
+ PrintArea(r1=r1, c1=c1 - 1, r2=r2, c2=c2 - 1)
135
+ )
136
+ except Exception as exc:
137
+ logger.warning(
138
+ "Failed to extract auto page breaks via COM for sheet '%s'. (%r)",
139
+ sheet.name,
140
+ exc,
141
+ )
142
+ failed = True
143
+ finally:
144
+ if ws_api is not None and original_display is not None:
145
+ try:
146
+ ws_api.DisplayPageBreaks = original_display
147
+ except Exception as exc:
148
+ logger.debug(
149
+ "Failed to restore DisplayPageBreaks for sheet '%s'. (%r)",
150
+ sheet.name,
151
+ exc,
152
+ )
153
+ if failed:
154
+ continue
155
+ return results
156
+
157
+
158
+ def _parse_print_area_range(range_str: str) -> tuple[int, int, int, int] | None:
159
+ """Parse an Excel range string into zero-based coordinates.
160
+
161
+ Args:
162
+ range_str: Excel range string.
163
+
164
+ Returns:
165
+ Zero-based (r1, c1, r2, c2) tuple or None on failure.
166
+ """
167
+ bounds = parse_range_zero_based(range_str)
168
+ if bounds is None:
169
+ return None
170
+ return (bounds.r1, bounds.c1, bounds.r2, bounds.c2)
171
+
172
+
173
+ def _normalize_area_for_sheet(part: str, ws_name: str) -> str | None:
174
+ """Strip sheet name from a range part when it matches the target sheet.
175
+
176
+ Args:
177
+ part: Raw range string part.
178
+ ws_name: Target worksheet name.
179
+
180
+ Returns:
181
+ Range without sheet prefix, or None if not matching.
182
+ """
183
+ s = part.strip()
184
+ if "!" not in s:
185
+ return s
186
+ sheet, rng = s.rsplit("!", 1)
187
+ sheet = sheet.strip()
188
+ if sheet.startswith("'") and sheet.endswith("'"):
189
+ sheet = sheet[1:-1].replace("''", "'")
190
+ return rng if sheet == ws_name else None
191
+
192
+
193
+ def _split_csv_respecting_quotes(raw: str) -> list[str]:
194
+ """Split a CSV-like string while keeping commas inside single quotes intact.
195
+
196
+ Args:
197
+ raw: Raw CSV-like string.
198
+
199
+ Returns:
200
+ List of split parts.
201
+ """
202
+ parts: list[str] = []
203
+ buf: list[str] = []
204
+ in_quote = False
205
+ i = 0
206
+ while i < len(raw):
207
+ ch = raw[i]
208
+ if ch == "'":
209
+ if in_quote and i + 1 < len(raw) and raw[i + 1] == "'":
210
+ buf.append("''")
211
+ i += 2
212
+ continue
213
+ in_quote = not in_quote
214
+ buf.append(ch)
215
+ i += 1
216
+ continue
217
+ if ch == "," and not in_quote:
218
+ parts.append("".join(buf).strip())
219
+ buf = []
220
+ i += 1
221
+ continue
222
+ buf.append(ch)
223
+ i += 1
224
+ if buf:
225
+ parts.append("".join(buf).strip())
226
+ return [p for p in parts if p]
@@ -0,0 +1,179 @@
1
+ """Openpyxl backend for Excel workbook extraction."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import logging
7
+ from pathlib import Path
8
+
9
+ from ...models import PrintArea
10
+ from ..cells import (
11
+ WorkbookColorsMap,
12
+ detect_tables_openpyxl,
13
+ extract_sheet_cells,
14
+ extract_sheet_cells_with_links,
15
+ extract_sheet_colors_map,
16
+ )
17
+ from ..ranges import parse_range_zero_based
18
+ from ..workbook import openpyxl_workbook
19
+ from .base import CellData, PrintAreaData
20
+
21
+ logger = logging.getLogger(__name__)
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class OpenpyxlBackend:
26
+ """Openpyxl-based backend for extraction tasks.
27
+
28
+ Attributes:
29
+ file_path: Path to the workbook file.
30
+ """
31
+
32
+ file_path: Path
33
+
34
+ def extract_cells(self, *, include_links: bool) -> CellData:
35
+ """Extract cell rows from the workbook.
36
+
37
+ Args:
38
+ include_links: Whether to include hyperlinks.
39
+
40
+ Returns:
41
+ Mapping of sheet name to cell rows.
42
+ """
43
+ return (
44
+ extract_sheet_cells_with_links(self.file_path)
45
+ if include_links
46
+ else extract_sheet_cells(self.file_path)
47
+ )
48
+
49
+ def extract_print_areas(self) -> PrintAreaData:
50
+ """Extract print areas per sheet using openpyxl defined names.
51
+
52
+ Returns:
53
+ Mapping of sheet name to print area list.
54
+ """
55
+ try:
56
+ with openpyxl_workbook(
57
+ self.file_path, data_only=True, read_only=False
58
+ ) as wb:
59
+ areas = _extract_print_areas_from_defined_names(wb)
60
+ if not areas:
61
+ areas = _extract_print_areas_from_sheet_props(wb)
62
+ return areas
63
+ except Exception:
64
+ return {}
65
+
66
+ def extract_colors_map(
67
+ self, *, include_default_background: bool, ignore_colors: set[str] | None
68
+ ) -> WorkbookColorsMap | None:
69
+ """Extract colors_map using openpyxl.
70
+
71
+ Args:
72
+ include_default_background: Whether to include default background colors.
73
+ ignore_colors: Optional set of color keys to ignore.
74
+
75
+ Returns:
76
+ WorkbookColorsMap or None when extraction fails.
77
+ """
78
+ try:
79
+ return extract_sheet_colors_map(
80
+ self.file_path,
81
+ include_default_background=include_default_background,
82
+ ignore_colors=ignore_colors,
83
+ )
84
+ except Exception as exc:
85
+ logger.warning(
86
+ "Color map extraction failed; skipping colors_map. (%r)", exc
87
+ )
88
+ return None
89
+
90
+ def detect_tables(self, sheet_name: str) -> list[str]:
91
+ """Detect table candidates for a single sheet.
92
+
93
+ Args:
94
+ sheet_name: Target worksheet name.
95
+
96
+ Returns:
97
+ List of table candidate ranges.
98
+ """
99
+ try:
100
+ return detect_tables_openpyxl(self.file_path, sheet_name)
101
+ except Exception:
102
+ return []
103
+
104
+
105
+ def _extract_print_areas_from_defined_names(workbook: object) -> PrintAreaData:
106
+ """Extract print areas from defined names in an openpyxl workbook.
107
+
108
+ Args:
109
+ workbook: openpyxl workbook instance.
110
+
111
+ Returns:
112
+ Mapping of sheet name to print area list.
113
+ """
114
+ defined = getattr(workbook, "defined_names", None)
115
+ if defined is None:
116
+ return {}
117
+ defined_area = defined.get("_xlnm.Print_Area")
118
+ if not defined_area:
119
+ return {}
120
+
121
+ areas: PrintAreaData = {}
122
+ sheetnames = set(getattr(workbook, "sheetnames", []))
123
+ for sheet_name, range_str in defined_area.destinations:
124
+ if sheet_name not in sheetnames:
125
+ continue
126
+ _append_print_areas(areas, sheet_name, str(range_str))
127
+ return areas
128
+
129
+
130
+ def _extract_print_areas_from_sheet_props(workbook: object) -> PrintAreaData:
131
+ """Extract print areas from sheet-level print area properties.
132
+
133
+ Args:
134
+ workbook: openpyxl workbook instance.
135
+
136
+ Returns:
137
+ Mapping of sheet name to print area list.
138
+ """
139
+ areas: PrintAreaData = {}
140
+ worksheets = getattr(workbook, "worksheets", [])
141
+ for ws in worksheets:
142
+ pa = getattr(ws, "_print_area", None)
143
+ if not pa:
144
+ continue
145
+ _append_print_areas(areas, str(getattr(ws, "title", "")), str(pa))
146
+ return areas
147
+
148
+
149
+ def _append_print_areas(areas: PrintAreaData, sheet_name: str, range_str: str) -> None:
150
+ """Append parsed print areas to the mapping.
151
+
152
+ Args:
153
+ areas: Mapping to update.
154
+ sheet_name: Target sheet name.
155
+ range_str: Raw range string, possibly comma-separated.
156
+ """
157
+ for part in str(range_str).split(","):
158
+ parsed = _parse_print_area_range(part)
159
+ if not parsed:
160
+ continue
161
+ r1, c1, r2, c2 = parsed
162
+ areas.setdefault(sheet_name, []).append(
163
+ PrintArea(r1=r1 + 1, c1=c1, r2=r2 + 1, c2=c2)
164
+ )
165
+
166
+
167
+ def _parse_print_area_range(range_str: str) -> tuple[int, int, int, int] | None:
168
+ """Parse an Excel range string into zero-based coordinates.
169
+
170
+ Args:
171
+ range_str: Excel range string.
172
+
173
+ Returns:
174
+ Zero-based (r1, c1, r2, c2) tuple or None on failure.
175
+ """
176
+ bounds = parse_range_zero_based(range_str)
177
+ if bounds is None:
178
+ return None
179
+ return (bounds.r1, bounds.c1, bounds.r2, bounds.c2)