pyhandlexl 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyhandlexl/__init__.py +46 -0
- pyhandlexl/_safety.py +116 -0
- pyhandlexl/core.py +250 -0
- pyhandlexl/errors.py +28 -0
- pyhandlexl/py.typed +0 -0
- pyhandlexl/table.py +373 -0
- pyhandlexl/validate.py +75 -0
- pyhandlexl-0.2.0.dist-info/METADATA +302 -0
- pyhandlexl-0.2.0.dist-info/RECORD +12 -0
- pyhandlexl-0.2.0.dist-info/WHEEL +5 -0
- pyhandlexl-0.2.0.dist-info/licenses/LICENSE +21 -0
- pyhandlexl-0.2.0.dist-info/top_level.txt +1 -0
pyhandlexl/__init__.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""pyhandlexl — read and write raw cell values in Excel .xlsx files."""
|
|
2
|
+
|
|
3
|
+
from pyhandlexl.core import (
|
|
4
|
+
append_rows,
|
|
5
|
+
create_sheet,
|
|
6
|
+
delete_sheet,
|
|
7
|
+
list_sheets,
|
|
8
|
+
read_sheet,
|
|
9
|
+
rename_sheet,
|
|
10
|
+
sheet_exists,
|
|
11
|
+
write_sheet,
|
|
12
|
+
)
|
|
13
|
+
from pyhandlexl.errors import (
|
|
14
|
+
DimensionError,
|
|
15
|
+
FileLockedError,
|
|
16
|
+
InvalidFileError,
|
|
17
|
+
PyhandlexlError,
|
|
18
|
+
SheetNameError,
|
|
19
|
+
SheetNotFoundError,
|
|
20
|
+
)
|
|
21
|
+
from pyhandlexl.table import Table, TableData
|
|
22
|
+
from pyhandlexl.validate import check_dimensions, check_sheet_name, is_valid_xlsx
|
|
23
|
+
|
|
24
|
+
__version__ = "0.2.0"
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"DimensionError",
|
|
28
|
+
"FileLockedError",
|
|
29
|
+
"InvalidFileError",
|
|
30
|
+
"PyhandlexlError",
|
|
31
|
+
"SheetNameError",
|
|
32
|
+
"SheetNotFoundError",
|
|
33
|
+
"Table",
|
|
34
|
+
"TableData",
|
|
35
|
+
"append_rows",
|
|
36
|
+
"check_dimensions",
|
|
37
|
+
"check_sheet_name",
|
|
38
|
+
"create_sheet",
|
|
39
|
+
"delete_sheet",
|
|
40
|
+
"is_valid_xlsx",
|
|
41
|
+
"list_sheets",
|
|
42
|
+
"read_sheet",
|
|
43
|
+
"rename_sheet",
|
|
44
|
+
"sheet_exists",
|
|
45
|
+
"write_sheet",
|
|
46
|
+
]
|
pyhandlexl/_safety.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Resilient workbook I/O: retry-on-lock loading and atomic saving.
|
|
2
|
+
|
|
3
|
+
Internal module — not part of the public API.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from secrets import token_hex
|
|
12
|
+
from time import sleep
|
|
13
|
+
from typing import TypeVar
|
|
14
|
+
from zipfile import BadZipFile
|
|
15
|
+
|
|
16
|
+
from openpyxl import Workbook, load_workbook
|
|
17
|
+
|
|
18
|
+
from pyhandlexl.errors import FileLockedError, InvalidFileError
|
|
19
|
+
from pyhandlexl.validate import is_valid_xlsx
|
|
20
|
+
|
|
21
|
+
_T = TypeVar("_T")
|
|
22
|
+
|
|
23
|
+
DEFAULT_RETRIES = 5
|
|
24
|
+
DEFAULT_DELAY = 0.5
|
|
25
|
+
|
|
26
|
+
# Raised when a file is briefly unavailable: open in Excel, or a competing
|
|
27
|
+
# writer has not finished flushing it yet. Worth retrying.
|
|
28
|
+
_TRANSIENT_ERRORS = (PermissionError, BadZipFile, EOFError)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _retry(
|
|
32
|
+
operation: Callable[[], _T],
|
|
33
|
+
*,
|
|
34
|
+
path: Path,
|
|
35
|
+
retries: int = DEFAULT_RETRIES,
|
|
36
|
+
delay: float = DEFAULT_DELAY,
|
|
37
|
+
) -> _T:
|
|
38
|
+
"""Run *operation*, retrying on transient file errors.
|
|
39
|
+
|
|
40
|
+
Raises FileLockedError if it never succeeds.
|
|
41
|
+
"""
|
|
42
|
+
last_error: Exception | None = None
|
|
43
|
+
for attempt in range(1, retries + 1):
|
|
44
|
+
try:
|
|
45
|
+
return operation()
|
|
46
|
+
except _TRANSIENT_ERRORS as error:
|
|
47
|
+
last_error = error
|
|
48
|
+
if attempt < retries:
|
|
49
|
+
sleep(delay)
|
|
50
|
+
raise FileLockedError(
|
|
51
|
+
f"could not access {path} after {retries} attempt(s): {last_error}"
|
|
52
|
+
) from last_error
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def safe_load(
|
|
56
|
+
path: str | Path,
|
|
57
|
+
*,
|
|
58
|
+
retries: int = DEFAULT_RETRIES,
|
|
59
|
+
delay: float = DEFAULT_DELAY,
|
|
60
|
+
) -> Workbook:
|
|
61
|
+
"""Load an existing workbook, retrying while the file is locked.
|
|
62
|
+
|
|
63
|
+
Raises:
|
|
64
|
+
FileNotFoundError: no file at *path*.
|
|
65
|
+
InvalidFileError: the file exists but is not a readable .xlsx.
|
|
66
|
+
FileLockedError: the file stayed locked through every retry.
|
|
67
|
+
"""
|
|
68
|
+
path = Path(path)
|
|
69
|
+
if not path.exists():
|
|
70
|
+
raise FileNotFoundError(f"no file at {path}")
|
|
71
|
+
if not is_valid_xlsx(path):
|
|
72
|
+
raise InvalidFileError(f"{path} is not a readable .xlsx workbook")
|
|
73
|
+
return _retry(lambda: load_workbook(path), path=path, retries=retries, delay=delay)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def load_or_create(
|
|
77
|
+
path: str | Path,
|
|
78
|
+
*,
|
|
79
|
+
retries: int = DEFAULT_RETRIES,
|
|
80
|
+
delay: float = DEFAULT_DELAY,
|
|
81
|
+
) -> Workbook:
|
|
82
|
+
"""Load the workbook at *path*, or return a new empty one if it does not exist."""
|
|
83
|
+
path = Path(path)
|
|
84
|
+
if not path.exists():
|
|
85
|
+
return Workbook()
|
|
86
|
+
return safe_load(path, retries=retries, delay=delay)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def atomic_save(
|
|
90
|
+
workbook: Workbook,
|
|
91
|
+
path: str | Path,
|
|
92
|
+
*,
|
|
93
|
+
retries: int = DEFAULT_RETRIES,
|
|
94
|
+
delay: float = DEFAULT_DELAY,
|
|
95
|
+
) -> None:
|
|
96
|
+
"""Save *workbook* to *path* without risking the file already there.
|
|
97
|
+
|
|
98
|
+
Writes to a temporary file in the same directory, verifies it is a
|
|
99
|
+
readable .xlsx, then atomically replaces the target. On any failure the
|
|
100
|
+
temporary file is removed and the original is left untouched.
|
|
101
|
+
|
|
102
|
+
Raises:
|
|
103
|
+
InvalidFileError: the freshly written file failed validation.
|
|
104
|
+
FileLockedError: the target stayed locked through every retry.
|
|
105
|
+
"""
|
|
106
|
+
path = Path(path)
|
|
107
|
+
tmp = path.parent / f".{path.stem}.{token_hex(6)}.tmp.xlsx"
|
|
108
|
+
try:
|
|
109
|
+
_retry(lambda: workbook.save(tmp), path=path, retries=retries, delay=delay)
|
|
110
|
+
if not is_valid_xlsx(tmp):
|
|
111
|
+
raise InvalidFileError(
|
|
112
|
+
f"wrote a temporary file for {path} but it failed validation; {path} left unchanged"
|
|
113
|
+
)
|
|
114
|
+
_retry(lambda: os.replace(tmp, path), path=path, retries=retries, delay=delay)
|
|
115
|
+
finally:
|
|
116
|
+
tmp.unlink(missing_ok=True)
|
pyhandlexl/core.py
ADDED
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
"""Public functions for reading and writing whole sheets of raw values."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from itertools import zip_longest
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Literal
|
|
9
|
+
|
|
10
|
+
from pyhandlexl._safety import atomic_save, load_or_create, safe_load
|
|
11
|
+
from pyhandlexl.errors import SheetNotFoundError
|
|
12
|
+
from pyhandlexl.validate import check_dimensions, check_sheet_name
|
|
13
|
+
|
|
14
|
+
Orientation = Literal["rows", "columns"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _cell_to_str(value: object) -> str:
|
|
18
|
+
"""Coerce a cell value to a string; ``None`` becomes ``""``."""
|
|
19
|
+
return "" if value is None else str(value)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def read_sheet(
|
|
23
|
+
path: str | Path,
|
|
24
|
+
sheet: str | None = None,
|
|
25
|
+
*,
|
|
26
|
+
pad: bool = False,
|
|
27
|
+
) -> list[list[str]]:
|
|
28
|
+
"""Read a worksheet as a list of rows of strings.
|
|
29
|
+
|
|
30
|
+
Every value is converted to ``str``; empty cells become ``""``. Trailing
|
|
31
|
+
empty cells are trimmed from each row, so a fully empty row becomes ``[]``.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
path: the .xlsx file.
|
|
35
|
+
sheet: worksheet name, or ``None`` for the active sheet.
|
|
36
|
+
pad: if true, right-pad every row with ``""`` to the length of the
|
|
37
|
+
longest row, making the result rectangular.
|
|
38
|
+
|
|
39
|
+
Raises:
|
|
40
|
+
FileNotFoundError: no file at *path*.
|
|
41
|
+
InvalidFileError: the file is not a readable .xlsx.
|
|
42
|
+
SheetNotFoundError: *sheet* names a worksheet that does not exist.
|
|
43
|
+
"""
|
|
44
|
+
workbook = safe_load(path)
|
|
45
|
+
try:
|
|
46
|
+
if sheet is None:
|
|
47
|
+
worksheet = workbook.active
|
|
48
|
+
elif sheet in workbook.sheetnames:
|
|
49
|
+
worksheet = workbook[sheet]
|
|
50
|
+
else:
|
|
51
|
+
raise SheetNotFoundError(sheet)
|
|
52
|
+
|
|
53
|
+
rows: list[list[str]] = []
|
|
54
|
+
for raw_row in worksheet.iter_rows(values_only=True):
|
|
55
|
+
row = [_cell_to_str(value) for value in raw_row]
|
|
56
|
+
while row and row[-1] == "":
|
|
57
|
+
row.pop()
|
|
58
|
+
rows.append(row)
|
|
59
|
+
finally:
|
|
60
|
+
workbook.close()
|
|
61
|
+
|
|
62
|
+
if pad and rows:
|
|
63
|
+
width = max(len(row) for row in rows)
|
|
64
|
+
rows = [row + [""] * (width - len(row)) for row in rows]
|
|
65
|
+
return rows
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def write_sheet(
|
|
69
|
+
path: str | Path,
|
|
70
|
+
rows: Iterable[Iterable[object]],
|
|
71
|
+
sheet: str | None = None,
|
|
72
|
+
*,
|
|
73
|
+
orientation: Orientation = "rows",
|
|
74
|
+
) -> None:
|
|
75
|
+
"""Replace a worksheet's contents with *rows*.
|
|
76
|
+
|
|
77
|
+
Other worksheets in the file are left untouched. The file is created if it
|
|
78
|
+
does not exist, and *sheet* is added if it does not exist.
|
|
79
|
+
|
|
80
|
+
Values are written as-is (``str``, ``int``, ``float``, ``bool``); ``None``
|
|
81
|
+
leaves the cell empty. No string-to-number conversion is performed.
|
|
82
|
+
|
|
83
|
+
Args:
|
|
84
|
+
path: the .xlsx file.
|
|
85
|
+
rows: an iterable of iterables of cell values.
|
|
86
|
+
sheet: worksheet name, or ``None`` for the active sheet.
|
|
87
|
+
orientation: ``"rows"`` writes each inner iterable as a row;
|
|
88
|
+
``"columns"`` writes each inner iterable down a column.
|
|
89
|
+
|
|
90
|
+
Raises:
|
|
91
|
+
SheetNameError: *sheet* is not a valid worksheet name.
|
|
92
|
+
DimensionError: the data exceeds the .xlsx row or column limits.
|
|
93
|
+
ValueError: *orientation* is not ``"rows"`` or ``"columns"``.
|
|
94
|
+
"""
|
|
95
|
+
if orientation not in ("rows", "columns"):
|
|
96
|
+
raise ValueError(f"orientation must be 'rows' or 'columns', got {orientation!r}")
|
|
97
|
+
|
|
98
|
+
grid = [list(row) for row in rows]
|
|
99
|
+
if orientation == "columns":
|
|
100
|
+
grid = [list(column) for column in zip_longest(*grid, fillvalue="")]
|
|
101
|
+
|
|
102
|
+
check_dimensions(len(grid), max((len(row) for row in grid), default=0))
|
|
103
|
+
|
|
104
|
+
if sheet is not None:
|
|
105
|
+
check_sheet_name(sheet)
|
|
106
|
+
|
|
107
|
+
file_existed = Path(path).is_file()
|
|
108
|
+
workbook = load_or_create(path)
|
|
109
|
+
try:
|
|
110
|
+
if not file_existed:
|
|
111
|
+
for name in list(workbook.sheetnames):
|
|
112
|
+
workbook.remove(workbook[name])
|
|
113
|
+
worksheet = workbook.create_sheet(title=sheet or "Sheet")
|
|
114
|
+
else:
|
|
115
|
+
name = sheet if sheet is not None else workbook.active.title
|
|
116
|
+
if name in workbook.sheetnames:
|
|
117
|
+
index = workbook.sheetnames.index(name)
|
|
118
|
+
workbook.remove(workbook[name])
|
|
119
|
+
worksheet = workbook.create_sheet(title=name, index=index)
|
|
120
|
+
else:
|
|
121
|
+
worksheet = workbook.create_sheet(title=name)
|
|
122
|
+
|
|
123
|
+
for row in grid:
|
|
124
|
+
worksheet.append(row)
|
|
125
|
+
atomic_save(workbook, path)
|
|
126
|
+
finally:
|
|
127
|
+
workbook.close()
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def append_rows(
|
|
131
|
+
path: str | Path,
|
|
132
|
+
rows: Iterable[Iterable[object]],
|
|
133
|
+
sheet: str | None = None,
|
|
134
|
+
) -> None:
|
|
135
|
+
"""Append *rows* to the end of a worksheet.
|
|
136
|
+
|
|
137
|
+
The file and *sheet* are created if they do not exist. An empty *rows* is
|
|
138
|
+
a no-op. Values follow the same rules as :func:`write_sheet`.
|
|
139
|
+
|
|
140
|
+
Raises:
|
|
141
|
+
SheetNameError: *sheet* is not a valid worksheet name.
|
|
142
|
+
DimensionError: appending would exceed the .xlsx row or column limits.
|
|
143
|
+
"""
|
|
144
|
+
grid = [list(row) for row in rows]
|
|
145
|
+
if not grid:
|
|
146
|
+
return
|
|
147
|
+
|
|
148
|
+
if sheet is not None:
|
|
149
|
+
check_sheet_name(sheet)
|
|
150
|
+
|
|
151
|
+
workbook = load_or_create(path)
|
|
152
|
+
try:
|
|
153
|
+
if sheet is None:
|
|
154
|
+
worksheet = workbook.active
|
|
155
|
+
elif sheet in workbook.sheetnames:
|
|
156
|
+
worksheet = workbook[sheet]
|
|
157
|
+
else:
|
|
158
|
+
worksheet = workbook.create_sheet(title=sheet)
|
|
159
|
+
|
|
160
|
+
widest = max(len(row) for row in grid)
|
|
161
|
+
check_dimensions(worksheet.max_row + len(grid), max(widest, worksheet.max_column))
|
|
162
|
+
|
|
163
|
+
for row in grid:
|
|
164
|
+
worksheet.append(row)
|
|
165
|
+
atomic_save(workbook, path)
|
|
166
|
+
finally:
|
|
167
|
+
workbook.close()
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def list_sheets(path: str | Path) -> list[str]:
|
|
171
|
+
"""Return the worksheet names in *path*, in order.
|
|
172
|
+
|
|
173
|
+
Raises:
|
|
174
|
+
FileNotFoundError: no file at *path*.
|
|
175
|
+
InvalidFileError: the file is not a readable .xlsx.
|
|
176
|
+
"""
|
|
177
|
+
workbook = safe_load(path)
|
|
178
|
+
try:
|
|
179
|
+
return list(workbook.sheetnames)
|
|
180
|
+
finally:
|
|
181
|
+
workbook.close()
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def sheet_exists(path: str | Path, name: str) -> bool:
|
|
185
|
+
"""Return whether *path* contains a worksheet called *name*."""
|
|
186
|
+
return name in list_sheets(path)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def create_sheet(path: str | Path, name: str) -> None:
|
|
190
|
+
"""Add an empty worksheet called *name*, creating the file if needed.
|
|
191
|
+
|
|
192
|
+
Raises:
|
|
193
|
+
SheetNameError: *name* is not a valid worksheet name.
|
|
194
|
+
ValueError: a worksheet called *name* already exists.
|
|
195
|
+
"""
|
|
196
|
+
check_sheet_name(name)
|
|
197
|
+
file_existed = Path(path).is_file()
|
|
198
|
+
workbook = load_or_create(path)
|
|
199
|
+
try:
|
|
200
|
+
if name in workbook.sheetnames:
|
|
201
|
+
raise ValueError(f"sheet {name!r} already exists")
|
|
202
|
+
if file_existed:
|
|
203
|
+
workbook.create_sheet(title=name)
|
|
204
|
+
else:
|
|
205
|
+
workbook.active.title = name
|
|
206
|
+
atomic_save(workbook, path)
|
|
207
|
+
finally:
|
|
208
|
+
workbook.close()
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def delete_sheet(path: str | Path, name: str) -> None:
|
|
212
|
+
"""Remove the worksheet called *name*.
|
|
213
|
+
|
|
214
|
+
Raises:
|
|
215
|
+
FileNotFoundError: no file at *path*.
|
|
216
|
+
SheetNotFoundError: no worksheet called *name*.
|
|
217
|
+
ValueError: *name* is the only worksheet (a workbook needs at least one).
|
|
218
|
+
"""
|
|
219
|
+
workbook = safe_load(path)
|
|
220
|
+
try:
|
|
221
|
+
if name not in workbook.sheetnames:
|
|
222
|
+
raise SheetNotFoundError(name)
|
|
223
|
+
if len(workbook.sheetnames) == 1:
|
|
224
|
+
raise ValueError("cannot delete the only sheet in the workbook")
|
|
225
|
+
del workbook[name]
|
|
226
|
+
atomic_save(workbook, path)
|
|
227
|
+
finally:
|
|
228
|
+
workbook.close()
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def rename_sheet(path: str | Path, old: str, new: str) -> None:
|
|
232
|
+
"""Rename worksheet *old* to *new*.
|
|
233
|
+
|
|
234
|
+
Raises:
|
|
235
|
+
SheetNameError: *new* is not a valid worksheet name.
|
|
236
|
+
FileNotFoundError: no file at *path*.
|
|
237
|
+
SheetNotFoundError: no worksheet called *old*.
|
|
238
|
+
ValueError: a different worksheet called *new* already exists.
|
|
239
|
+
"""
|
|
240
|
+
check_sheet_name(new)
|
|
241
|
+
workbook = safe_load(path)
|
|
242
|
+
try:
|
|
243
|
+
if old not in workbook.sheetnames:
|
|
244
|
+
raise SheetNotFoundError(old)
|
|
245
|
+
if new != old and new in workbook.sheetnames:
|
|
246
|
+
raise ValueError(f"sheet {new!r} already exists")
|
|
247
|
+
workbook[old].title = new
|
|
248
|
+
atomic_save(workbook, path)
|
|
249
|
+
finally:
|
|
250
|
+
workbook.close()
|
pyhandlexl/errors.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Exception types raised by pyhandlexl."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class PyhandlexlError(Exception):
|
|
5
|
+
"""Base class for every error raised by pyhandlexl.
|
|
6
|
+
|
|
7
|
+
Catch this to handle any failure from the library.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class SheetNameError(PyhandlexlError, ValueError):
|
|
12
|
+
"""A worksheet name is empty, too long, or contains illegal characters."""
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class DimensionError(PyhandlexlError, ValueError):
|
|
16
|
+
"""The data has more rows or columns than the .xlsx format allows."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class InvalidFileError(PyhandlexlError):
|
|
20
|
+
"""The file is missing, not a zip, or not a readable .xlsx workbook."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class FileLockedError(PyhandlexlError, OSError):
|
|
24
|
+
"""The file stayed locked (e.g. open in Excel) after every retry."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SheetNotFoundError(PyhandlexlError, KeyError):
|
|
28
|
+
"""No worksheet with the requested name exists in the workbook."""
|
pyhandlexl/py.typed
ADDED
|
File without changes
|
pyhandlexl/table.py
ADDED
|
@@ -0,0 +1,373 @@
|
|
|
1
|
+
"""The Table class: a worksheet read as column headers, row labels, and a data grid.
|
|
2
|
+
|
|
3
|
+
Layout convention: row 1 holds the column headers, column A holds the row
|
|
4
|
+
labels, cell A1 is the "corner", and the data region is everything from B2
|
|
5
|
+
onward. Row labels and column headers are always strings. Positional access
|
|
6
|
+
uses Excel coordinates (row 1 is the header row, column 1 is the label column).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from collections.abc import Iterable
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from openpyxl.utils import coordinate_to_tuple
|
|
16
|
+
|
|
17
|
+
from pyhandlexl.core import read_sheet, write_sheet
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class TableData:
|
|
22
|
+
"""A read-only snapshot of a Table's content."""
|
|
23
|
+
|
|
24
|
+
rows: list[list[object]]
|
|
25
|
+
columns: list[list[object]]
|
|
26
|
+
row_labels: list[str]
|
|
27
|
+
column_headers: list[str]
|
|
28
|
+
corner: object
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class Table:
|
|
32
|
+
"""A labeled table backed by a worksheet.
|
|
33
|
+
|
|
34
|
+
Construct directly from parts, or with :meth:`read` from a file. Mutation
|
|
35
|
+
methods edit the table in place and return ``None``. Row labels and column
|
|
36
|
+
headers are always ``str``.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
data: Iterable[Iterable[object]],
|
|
42
|
+
column_headers: Iterable[str] = (),
|
|
43
|
+
row_labels: Iterable[str] = (),
|
|
44
|
+
corner: object = "",
|
|
45
|
+
) -> None:
|
|
46
|
+
self._data: list[list[object]] = [list(row) for row in data]
|
|
47
|
+
self._column_headers: list[str] = list(column_headers)
|
|
48
|
+
self._row_labels: list[str] = list(row_labels)
|
|
49
|
+
self._corner: object = corner
|
|
50
|
+
self._validate()
|
|
51
|
+
|
|
52
|
+
def _validate(self) -> None:
|
|
53
|
+
for label in self._row_labels:
|
|
54
|
+
if not isinstance(label, str):
|
|
55
|
+
raise TypeError(f"row labels must be str, got {type(label).__name__}: {label!r}")
|
|
56
|
+
for header in self._column_headers:
|
|
57
|
+
if not isinstance(header, str):
|
|
58
|
+
raise TypeError(
|
|
59
|
+
f"column headers must be str, got {type(header).__name__}: {header!r}"
|
|
60
|
+
)
|
|
61
|
+
if self._row_labels and len(self._row_labels) != len(self._data):
|
|
62
|
+
raise ValueError(f"{len(self._data)} data rows but {len(self._row_labels)} row labels")
|
|
63
|
+
if self._column_headers:
|
|
64
|
+
width = len(self._column_headers)
|
|
65
|
+
for i, row in enumerate(self._data):
|
|
66
|
+
if len(row) != width:
|
|
67
|
+
raise ValueError(
|
|
68
|
+
f"data row {i} has {len(row)} values but there are {width} column headers"
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
# ------------------------------------------------------------------ read
|
|
72
|
+
|
|
73
|
+
@classmethod
|
|
74
|
+
def read(
|
|
75
|
+
cls,
|
|
76
|
+
path: str | Path,
|
|
77
|
+
sheet: str | None = None,
|
|
78
|
+
*,
|
|
79
|
+
column_headers: bool = True,
|
|
80
|
+
row_labels: bool = True,
|
|
81
|
+
) -> Table:
|
|
82
|
+
"""Read a worksheet into a Table.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
path: the .xlsx file.
|
|
86
|
+
sheet: worksheet name, or ``None`` for the active sheet.
|
|
87
|
+
column_headers: treat row 1 as column headers.
|
|
88
|
+
row_labels: treat column A as row labels.
|
|
89
|
+
"""
|
|
90
|
+
grid = read_sheet(path, sheet, pad=True)
|
|
91
|
+
if not grid:
|
|
92
|
+
return cls([], [], [], "")
|
|
93
|
+
|
|
94
|
+
row_start = 1 if column_headers else 0
|
|
95
|
+
col_start = 1 if row_labels else 0
|
|
96
|
+
|
|
97
|
+
corner = grid[0][0] if (column_headers and row_labels) else ""
|
|
98
|
+
headers = grid[0][col_start:] if column_headers else []
|
|
99
|
+
labels = [grid[r][0] for r in range(row_start, len(grid))] if row_labels else []
|
|
100
|
+
data = [grid[r][col_start:] for r in range(row_start, len(grid))]
|
|
101
|
+
|
|
102
|
+
return cls(data, headers, labels, corner)
|
|
103
|
+
|
|
104
|
+
# ------------------------------------------------------------ properties
|
|
105
|
+
|
|
106
|
+
@property
|
|
107
|
+
def corner(self) -> object:
|
|
108
|
+
"""The value of cell A1. Settable — this is how you change it."""
|
|
109
|
+
return self._corner
|
|
110
|
+
|
|
111
|
+
@corner.setter
|
|
112
|
+
def corner(self, value: object) -> None:
|
|
113
|
+
self._corner = value
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def data(self) -> TableData:
|
|
117
|
+
"""A snapshot of the table's rows, columns, row labels, column headers, and corner."""
|
|
118
|
+
return TableData(
|
|
119
|
+
rows=[list(row) for row in self._data],
|
|
120
|
+
columns=[list(column) for column in zip(*self._data, strict=True)],
|
|
121
|
+
row_labels=list(self._row_labels),
|
|
122
|
+
column_headers=list(self._column_headers),
|
|
123
|
+
corner=self._corner,
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
# ---------------------------------------------------------- label access
|
|
127
|
+
|
|
128
|
+
def _row_index(self, label: str) -> int:
|
|
129
|
+
try:
|
|
130
|
+
return self._row_labels.index(label)
|
|
131
|
+
except ValueError:
|
|
132
|
+
raise KeyError(f"no row labeled {label!r}") from None
|
|
133
|
+
|
|
134
|
+
def _column_index(self, header: str) -> int:
|
|
135
|
+
try:
|
|
136
|
+
return self._column_headers.index(header)
|
|
137
|
+
except ValueError:
|
|
138
|
+
raise KeyError(f"no column headed {header!r}") from None
|
|
139
|
+
|
|
140
|
+
def read_row(self, row_label: str) -> list[object]:
|
|
141
|
+
"""The data row for *row_label* (labels not included)."""
|
|
142
|
+
return list(self._data[self._row_index(row_label)])
|
|
143
|
+
|
|
144
|
+
def read_column(self, column_header: str) -> list[object]:
|
|
145
|
+
"""The data column for *column_header* (headers not included)."""
|
|
146
|
+
j = self._column_index(column_header)
|
|
147
|
+
return [row[j] for row in self._data]
|
|
148
|
+
|
|
149
|
+
# ------------------------------------------------- position/label access
|
|
150
|
+
|
|
151
|
+
def _width(self) -> int:
|
|
152
|
+
if self._column_headers:
|
|
153
|
+
return len(self._column_headers)
|
|
154
|
+
return len(self._data[0]) if self._data else 0
|
|
155
|
+
|
|
156
|
+
def _dimensions(self) -> tuple[int, int, int, int]:
|
|
157
|
+
row_start = 1 if self._column_headers else 0
|
|
158
|
+
col_start = 1 if self._row_labels else 0
|
|
159
|
+
n_rows = row_start + len(self._data)
|
|
160
|
+
n_cols = col_start + self._width()
|
|
161
|
+
return row_start, col_start, n_rows, n_cols
|
|
162
|
+
|
|
163
|
+
def _classify(self, row: int, col_num: int) -> tuple[str, int, int]:
|
|
164
|
+
"""Return ``(kind, i, j)`` where *kind* is corner/header/label/data.
|
|
165
|
+
|
|
166
|
+
*i* and *j* are indices into the relevant list (``_data`` is indexed
|
|
167
|
+
by both).
|
|
168
|
+
"""
|
|
169
|
+
row_start, col_start, n_rows, n_cols = self._dimensions()
|
|
170
|
+
if not (1 <= row <= n_rows and 1 <= col_num <= n_cols):
|
|
171
|
+
raise IndexError(f"cell ({row}, {col_num}) is outside the {n_rows}x{n_cols} table")
|
|
172
|
+
|
|
173
|
+
in_header_row = bool(self._column_headers) and row == 1
|
|
174
|
+
in_label_column = bool(self._row_labels) and col_num == 1
|
|
175
|
+
if in_header_row and in_label_column:
|
|
176
|
+
return "corner", 0, 0
|
|
177
|
+
if in_header_row:
|
|
178
|
+
return "header", 0, col_num - 1 - col_start
|
|
179
|
+
if in_label_column:
|
|
180
|
+
return "label", row - 1 - row_start, 0
|
|
181
|
+
return "data", row - 1 - row_start, col_num - 1 - col_start
|
|
182
|
+
|
|
183
|
+
def _dispatch(
|
|
184
|
+
self, ref: str | None, row: int | str | None, column: int | str | None
|
|
185
|
+
) -> tuple[bool, int | str, int | str]:
|
|
186
|
+
"""Resolve ``ref``/``row``/``column`` into ``(is_position, row_val, col_val)``.
|
|
187
|
+
|
|
188
|
+
``is_position=True`` — ``row_val``/``col_val`` are 1-based Excel ints.
|
|
189
|
+
``is_position=False`` — they are a row label / column header string.
|
|
190
|
+
"""
|
|
191
|
+
if ref is not None:
|
|
192
|
+
if row is not None or column is not None:
|
|
193
|
+
raise TypeError("give either ref or row=/column=, not both")
|
|
194
|
+
r, c = coordinate_to_tuple(ref)
|
|
195
|
+
return True, r, c
|
|
196
|
+
|
|
197
|
+
if row is None or column is None:
|
|
198
|
+
raise TypeError("give a ref like 'B2', or both row= and column=")
|
|
199
|
+
|
|
200
|
+
if isinstance(row, int) and isinstance(column, int):
|
|
201
|
+
return True, row, column
|
|
202
|
+
if isinstance(row, str) and isinstance(column, str):
|
|
203
|
+
return False, row, column
|
|
204
|
+
raise TypeError("row and column must both be int (position) or both be str (label)")
|
|
205
|
+
|
|
206
|
+
def read_cell(
|
|
207
|
+
self,
|
|
208
|
+
ref: str | None = None,
|
|
209
|
+
*,
|
|
210
|
+
row: int | str | None = None,
|
|
211
|
+
column: int | str | None = None,
|
|
212
|
+
) -> object:
|
|
213
|
+
"""A single value, addressed either by position or by label.
|
|
214
|
+
|
|
215
|
+
Give either a ref like ``"B2"``, or ``row=``/``column=`` as a matching
|
|
216
|
+
pair: both ints for a 1-based Excel position (row 1 is the header row,
|
|
217
|
+
column 1 is the label column, so ``read_cell(row=2, column=2)`` is the
|
|
218
|
+
first data cell), or both strings for a row label / column header pair
|
|
219
|
+
— e.g. ``read_cell(row="Alice", column="q1")``.
|
|
220
|
+
|
|
221
|
+
By position, ``read_cell`` can reach any cell — header, label, corner,
|
|
222
|
+
or data. By label it always reads data (the label-addressed equivalent
|
|
223
|
+
of ``read_cell(row=2, column=2)``, wherever that intersection lives).
|
|
224
|
+
"""
|
|
225
|
+
is_position, r, c = self._dispatch(ref, row, column)
|
|
226
|
+
if is_position:
|
|
227
|
+
kind, i, j = self._classify(r, c)
|
|
228
|
+
if kind == "corner":
|
|
229
|
+
return self._corner
|
|
230
|
+
if kind == "header":
|
|
231
|
+
return self._column_headers[j]
|
|
232
|
+
if kind == "label":
|
|
233
|
+
return self._row_labels[i]
|
|
234
|
+
return self._data[i][j]
|
|
235
|
+
return self._data[self._row_index(r)][self._column_index(c)]
|
|
236
|
+
|
|
237
|
+
# -------------------------------------------------------------- mutation
|
|
238
|
+
|
|
239
|
+
def set_cell(
|
|
240
|
+
self,
|
|
241
|
+
ref: str | None = None,
|
|
242
|
+
*,
|
|
243
|
+
row: int | str | None = None,
|
|
244
|
+
column: int | str | None = None,
|
|
245
|
+
value: object,
|
|
246
|
+
) -> None:
|
|
247
|
+
"""Set a single **data** value, addressed by position or by label.
|
|
248
|
+
|
|
249
|
+
Same addressing as :meth:`read_cell`. Only ever sets data — by
|
|
250
|
+
position, addressing a header, row label, or the corner raises
|
|
251
|
+
``ValueError``; use :meth:`rename_column`, :meth:`rename_row`, or the
|
|
252
|
+
``corner`` property for those. By label there's no other kind of cell
|
|
253
|
+
to reach, so it always sets data.
|
|
254
|
+
"""
|
|
255
|
+
is_position, r, c = self._dispatch(ref, row, column)
|
|
256
|
+
if is_position:
|
|
257
|
+
self._set_by_position(r, c, value)
|
|
258
|
+
else:
|
|
259
|
+
self._data[self._row_index(r)][self._column_index(c)] = value
|
|
260
|
+
|
|
261
|
+
def _set_by_position(self, row: int, col_num: int, value: object) -> None:
|
|
262
|
+
kind, i, j = self._classify(row, col_num)
|
|
263
|
+
if kind == "corner":
|
|
264
|
+
raise ValueError("that cell is the corner — set it with table.corner = value")
|
|
265
|
+
if kind == "header":
|
|
266
|
+
raise ValueError("that cell is a column header — rename it with rename_column()")
|
|
267
|
+
if kind == "label":
|
|
268
|
+
raise ValueError("that cell is a row label — rename it with rename_row()")
|
|
269
|
+
self._data[i][j] = value
|
|
270
|
+
|
|
271
|
+
def set_row(self, row_label: str, values: Iterable[object]) -> None:
|
|
272
|
+
"""Replace the data row for *row_label*; ``len(values)`` must match the column count."""
|
|
273
|
+
new_row = list(values)
|
|
274
|
+
i = self._row_index(row_label)
|
|
275
|
+
expected = self._width()
|
|
276
|
+
if len(new_row) != expected:
|
|
277
|
+
raise ValueError(f"expected {expected} values, got {len(new_row)}")
|
|
278
|
+
self._data[i] = new_row
|
|
279
|
+
|
|
280
|
+
def set_column(self, column_header: str, values: Iterable[object]) -> None:
|
|
281
|
+
"""Replace the data column for *column_header*; ``len(values)`` must match the row count."""
|
|
282
|
+
new_col = list(values)
|
|
283
|
+
j = self._column_index(column_header)
|
|
284
|
+
if len(new_col) != len(self._data):
|
|
285
|
+
raise ValueError(f"expected {len(self._data)} values, got {len(new_col)}")
|
|
286
|
+
for data_row, value in zip(self._data, new_col, strict=True):
|
|
287
|
+
data_row[j] = value
|
|
288
|
+
|
|
289
|
+
def add_row(self, label: str, values: Iterable[object]) -> None:
|
|
290
|
+
"""Append a labeled data row; ``len(values)`` must match the column count."""
|
|
291
|
+
if not isinstance(label, str):
|
|
292
|
+
raise TypeError(f"row label must be str, got {type(label).__name__}: {label!r}")
|
|
293
|
+
new_row = list(values)
|
|
294
|
+
if self._data and not self._row_labels:
|
|
295
|
+
raise ValueError("this table has no row labels; use the raw layer to add rows")
|
|
296
|
+
if (self._data or self._column_headers) and len(new_row) != self._width():
|
|
297
|
+
raise ValueError(f"expected {self._width()} values, got {len(new_row)}")
|
|
298
|
+
self._data.append(new_row)
|
|
299
|
+
self._row_labels.append(label)
|
|
300
|
+
|
|
301
|
+
def add_column(self, header: str, values: Iterable[object]) -> None:
|
|
302
|
+
"""Append a labeled data column; ``len(values)`` must match the row count."""
|
|
303
|
+
if not isinstance(header, str):
|
|
304
|
+
raise TypeError(f"column header must be str, got {type(header).__name__}: {header!r}")
|
|
305
|
+
new_col = list(values)
|
|
306
|
+
if any(self._data) and not self._column_headers:
|
|
307
|
+
raise ValueError("this table has no column headers; use the raw layer to add columns")
|
|
308
|
+
if len(new_col) != len(self._data):
|
|
309
|
+
raise ValueError(f"expected {len(self._data)} values, got {len(new_col)}")
|
|
310
|
+
for data_row, value in zip(self._data, new_col, strict=True):
|
|
311
|
+
data_row.append(value)
|
|
312
|
+
self._column_headers.append(header)
|
|
313
|
+
|
|
314
|
+
def drop_row(self, label: str) -> None:
|
|
315
|
+
"""Remove the row labeled *label*."""
|
|
316
|
+
i = self._row_index(label)
|
|
317
|
+
del self._data[i]
|
|
318
|
+
del self._row_labels[i]
|
|
319
|
+
|
|
320
|
+
def drop_column(self, header: str) -> None:
|
|
321
|
+
"""Remove the column headed *header*."""
|
|
322
|
+
j = self._column_index(header)
|
|
323
|
+
del self._column_headers[j]
|
|
324
|
+
for data_row in self._data:
|
|
325
|
+
del data_row[j]
|
|
326
|
+
|
|
327
|
+
def rename_row(self, old: str, new: str) -> None:
|
|
328
|
+
"""Change a row label."""
|
|
329
|
+
if not isinstance(new, str):
|
|
330
|
+
raise TypeError(f"row label must be str, got {type(new).__name__}: {new!r}")
|
|
331
|
+
self._row_labels[self._row_index(old)] = new
|
|
332
|
+
|
|
333
|
+
def rename_column(self, old: str, new: str) -> None:
|
|
334
|
+
"""Change a column header."""
|
|
335
|
+
if not isinstance(new, str):
|
|
336
|
+
raise TypeError(f"column header must be str, got {type(new).__name__}: {new!r}")
|
|
337
|
+
self._column_headers[self._column_index(old)] = new
|
|
338
|
+
|
|
339
|
+
# ----------------------------------------------------------------- write
|
|
340
|
+
|
|
341
|
+
def _assemble(self) -> list[list[object]]:
|
|
342
|
+
grid: list[list[object]] = []
|
|
343
|
+
if self._column_headers:
|
|
344
|
+
header_row: list[object] = [self._corner] if self._row_labels else []
|
|
345
|
+
header_row.extend(self._column_headers)
|
|
346
|
+
grid.append(header_row)
|
|
347
|
+
for i, data_row in enumerate(self._data):
|
|
348
|
+
out_row: list[object] = [self._row_labels[i]] if self._row_labels else []
|
|
349
|
+
out_row.extend(data_row)
|
|
350
|
+
grid.append(out_row)
|
|
351
|
+
return grid
|
|
352
|
+
|
|
353
|
+
def write(self, path: str | Path, sheet: str | None = None) -> None:
|
|
354
|
+
"""Write the table to *path*, reassembling headers into row 1 and labels into column A."""
|
|
355
|
+
write_sheet(path, self._assemble(), sheet)
|
|
356
|
+
|
|
357
|
+
# --------------------------------------------------------------- dunders
|
|
358
|
+
|
|
359
|
+
def __eq__(self, other: object) -> bool:
|
|
360
|
+
if not isinstance(other, Table):
|
|
361
|
+
return NotImplemented
|
|
362
|
+
return (
|
|
363
|
+
self._data == other._data
|
|
364
|
+
and self._column_headers == other._column_headers
|
|
365
|
+
and self._row_labels == other._row_labels
|
|
366
|
+
and self._corner == other._corner
|
|
367
|
+
)
|
|
368
|
+
|
|
369
|
+
def __repr__(self) -> str:
|
|
370
|
+
return (
|
|
371
|
+
f"Table(rows={len(self._data)}, columns={len(self._column_headers)}, "
|
|
372
|
+
f"corner={self._corner!r})"
|
|
373
|
+
)
|
pyhandlexl/validate.py
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""Validation helpers: sheet names, data dimensions, and file integrity."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import zipfile
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from openpyxl import load_workbook
|
|
9
|
+
|
|
10
|
+
from pyhandlexl.errors import DimensionError, SheetNameError
|
|
11
|
+
|
|
12
|
+
# Excel's hard limits for the .xlsx format.
|
|
13
|
+
MAX_ROWS = 1_048_576
|
|
14
|
+
MAX_COLUMNS = 16_384
|
|
15
|
+
|
|
16
|
+
# Characters Excel forbids in a worksheet name.
|
|
17
|
+
ILLEGAL_SHEET_CHARS = frozenset(r"\/?*[]:")
|
|
18
|
+
|
|
19
|
+
# Files every valid .xlsx zip must contain.
|
|
20
|
+
_REQUIRED_PARTS = frozenset({"xl/workbook.xml", "[Content_Types].xml"})
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def check_dimensions(n_rows: int, n_cols: int) -> None:
|
|
24
|
+
"""Raise DimensionError if a grid of this size won't fit in an .xlsx sheet."""
|
|
25
|
+
if n_rows > MAX_ROWS:
|
|
26
|
+
raise DimensionError(f"{n_rows} rows exceeds the .xlsx limit of {MAX_ROWS}")
|
|
27
|
+
if n_cols > MAX_COLUMNS:
|
|
28
|
+
raise DimensionError(f"{n_cols} columns exceeds the .xlsx limit of {MAX_COLUMNS}")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def check_sheet_name(sheet_name: str) -> None:
|
|
32
|
+
"""Raise SheetNameError if *sheet_name* is not a valid Excel worksheet name."""
|
|
33
|
+
if not isinstance(sheet_name, str):
|
|
34
|
+
raise SheetNameError(f"sheet name must be a string, got {type(sheet_name).__name__}")
|
|
35
|
+
|
|
36
|
+
if sheet_name == "":
|
|
37
|
+
raise SheetNameError("sheet name must not be empty")
|
|
38
|
+
|
|
39
|
+
if len(sheet_name) > 31:
|
|
40
|
+
raise SheetNameError(f"sheet name is longer than 31 characters: {sheet_name!r}")
|
|
41
|
+
|
|
42
|
+
illegal = sorted(set(sheet_name) & ILLEGAL_SHEET_CHARS)
|
|
43
|
+
if illegal:
|
|
44
|
+
raise SheetNameError(f"sheet name {sheet_name!r} contains illegal character(s): {illegal}")
|
|
45
|
+
|
|
46
|
+
if sheet_name.lower() == "history":
|
|
47
|
+
raise SheetNameError("'History' is reserved by Excel and cannot be used as a sheet name")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def is_valid_xlsx(path: str | Path) -> bool:
|
|
51
|
+
"""Return True only if *path* is a readable .xlsx workbook.
|
|
52
|
+
|
|
53
|
+
Never raises: every failure mode returns False.
|
|
54
|
+
"""
|
|
55
|
+
path = Path(path)
|
|
56
|
+
if not path.is_file():
|
|
57
|
+
return False
|
|
58
|
+
|
|
59
|
+
try:
|
|
60
|
+
with zipfile.ZipFile(path) as zf:
|
|
61
|
+
if not _REQUIRED_PARTS.issubset(zf.namelist()):
|
|
62
|
+
return False
|
|
63
|
+
if zf.testzip() is not None:
|
|
64
|
+
return False
|
|
65
|
+
except (zipfile.BadZipFile, OSError):
|
|
66
|
+
return False
|
|
67
|
+
|
|
68
|
+
# openpyxl raises several unrelated exception types for a malformed
|
|
69
|
+
# workbook; this function's job is to answer yes/no, not to blow up.
|
|
70
|
+
try:
|
|
71
|
+
load_workbook(path, read_only=True).close()
|
|
72
|
+
except Exception:
|
|
73
|
+
return False
|
|
74
|
+
|
|
75
|
+
return True
|
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pyhandlexl
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Use an Excel file as the database for your next project.
|
|
5
|
+
Author-email: Lewis Wainaina <lewyamendi@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/LewyAmendi/pyhandlexl
|
|
8
|
+
Project-URL: Repository, https://github.com/LewyAmendi/pyhandlexl
|
|
9
|
+
Project-URL: Issues, https://github.com/LewyAmendi/pyhandlexl/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/LewyAmendi/pyhandlexl/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: excel,xlsx,openpyxl,spreadsheet,xl
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Office/Business :: Financial :: Spreadsheet
|
|
21
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: openpyxl>=3.1
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
29
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# pyhandlexl
|
|
33
|
+
|
|
34
|
+
[](https://github.com/LewyAmendi/pyhandlexl/actions/workflows/ci.yml)
|
|
35
|
+
|
|
36
|
+
**Make an Excel file the database for your next project.**
|
|
37
|
+
|
|
38
|
+
A spreadsheet is the most portable data store there is: every machine opens it,
|
|
39
|
+
anyone can read or edit it without knowing a query language, it versions as a
|
|
40
|
+
single file, and there is no server to run. `pyhandlexl` makes driving one from
|
|
41
|
+
Python dependable — read and write raw cell values in a known order, with writes
|
|
42
|
+
that leave the file intact even when things go wrong.
|
|
43
|
+
|
|
44
|
+
Built on [openpyxl](https://openpyxl.readthedocs.io/) and built to grow.
|
|
45
|
+
|
|
46
|
+
- **`Table`** — the main way in. Row 1 holds your column headers, column A holds
|
|
47
|
+
your row labels, and everything from `B2` on is data. Read it, edit it by name,
|
|
48
|
+
write it back.
|
|
49
|
+
- **`read_sheet` / `write_sheet`** — direct grid access for sheets that aren't a
|
|
50
|
+
labelled table.
|
|
51
|
+
|
|
52
|
+
> **Version 0.1.0.** Usable today and under active development — expect new
|
|
53
|
+
> capabilities with each release, and some API changes as it matures.
|
|
54
|
+
|
|
55
|
+
## Install
|
|
56
|
+
|
|
57
|
+
Not on PyPI yet. From source:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
git clone https://github.com/LewyAmendi/pyhandlexl
|
|
61
|
+
cd pyhandlexl
|
|
62
|
+
pip install -e .
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Requires Python 3.10+.
|
|
66
|
+
|
|
67
|
+
## Quickstart
|
|
68
|
+
|
|
69
|
+
Given `budget.xlsx`:
|
|
70
|
+
|
|
71
|
+
| | q1 | q2 |
|
|
72
|
+
|--------|----|----|
|
|
73
|
+
| **Alice** | 10 | 20 |
|
|
74
|
+
| **Bob** | 30 | 40 |
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
from pyhandlexl import Table
|
|
78
|
+
|
|
79
|
+
t = Table.read("budget.xlsx")
|
|
80
|
+
|
|
81
|
+
t.read_cell(row="Alice", column="q2") # '20'
|
|
82
|
+
t.read_row("Bob") # ['30', '40']
|
|
83
|
+
t.read_column("q1") # ['10', '30']
|
|
84
|
+
|
|
85
|
+
t.set_cell(row="Alice", column="q1", value=99) # edit in place
|
|
86
|
+
t.add_row("Carol", [1, 2])
|
|
87
|
+
t.write("budget.xlsx") # one safe, atomic write
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
> **Values are always strings.** `read_sheet` and `Table` coerce every cell to
|
|
91
|
+
> `str` (empty cells become `""`). Convert to numbers yourself where you need to.
|
|
92
|
+
|
|
93
|
+
## The `Table` class
|
|
94
|
+
|
|
95
|
+
### Reading
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
Table.read(path, sheet=None, *, column_headers=True, row_labels=True)
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
- `sheet=None` reads the active sheet; pass a name for a specific one.
|
|
102
|
+
- `column_headers=False` — row 1 is ordinary data, `column_headers` is empty.
|
|
103
|
+
- `row_labels=False` — column A is ordinary data, `row_labels` is empty.
|
|
104
|
+
|
|
105
|
+
Row labels and column headers are always `str` — required if you build a
|
|
106
|
+
`Table` by hand, too (`Table(..., column_headers=[1, 2])` raises `TypeError`).
|
|
107
|
+
|
|
108
|
+
### The whole table at once
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
t.corner # value of cell A1 (settable: t.corner = "name")
|
|
112
|
+
t.data # a TableData snapshot
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
d = t.data
|
|
117
|
+
d.rows # [['10', '20'], ['30', '40']] (B2 onward, by row)
|
|
118
|
+
d.columns # [['10', '30'], ['20', '40']] (same data, by column)
|
|
119
|
+
d.row_labels # ['Alice', 'Bob'] (column A, from A2)
|
|
120
|
+
d.column_headers # ['q1', 'q2'] (row 1, from B1)
|
|
121
|
+
d.corner # value of cell A1
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Every field is a fresh copy — mutating `t.data.rows` does not change the table.
|
|
125
|
+
|
|
126
|
+
### Access by label
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
t.read_row("Bob") # a data row (no label)
|
|
130
|
+
t.read_column("q1") # a data column (no header)
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
Unknown labels raise `KeyError`. If a label appears twice, the first match wins.
|
|
134
|
+
|
|
135
|
+
### `read_cell` / `set_cell` — a single value, by position or by label
|
|
136
|
+
|
|
137
|
+
Both take the same addressing: a ref like `"B2"`, or `row=`/`column=` as a
|
|
138
|
+
matching pair — **both ints** for a 1-based Excel position (row 1 is the
|
|
139
|
+
header row, column 1 is the label column), or **both strings** for a row
|
|
140
|
+
label / column header pair. Mixing types raises `TypeError`.
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
t.read_cell("B2") # '10' — first data cell, by position
|
|
144
|
+
t.read_cell(row=2, column=2) # '10' — same thing, spelled out
|
|
145
|
+
t.read_cell(row="Alice", column="q1") # '10' — same value, by label
|
|
146
|
+
|
|
147
|
+
t.read_cell(row=1, column=2) # 'q1' — a column header
|
|
148
|
+
t.read_cell(row=2, column=1) # 'Alice' — a row label
|
|
149
|
+
t.read_cell("A1") # the corner
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
By position, `read_cell` can reach *any* cell — header, label, corner, or
|
|
153
|
+
data. By label it always reads data, wherever that row/column intersection
|
|
154
|
+
actually lives.
|
|
155
|
+
|
|
156
|
+
A string given through `row=`/`column=` is always a label lookup — it does
|
|
157
|
+
**not** accept a column letter like `"B"` for a position. Use a plain number
|
|
158
|
+
(`column=2`) or `ref="B2"` for letter-based positions.
|
|
159
|
+
|
|
160
|
+
### Editing (in place, returns `None`)
|
|
161
|
+
|
|
162
|
+
```python
|
|
163
|
+
t.set_cell("B2", value=99) # by position — ref
|
|
164
|
+
t.set_cell(row=2, column=2, value=99) # by position — row=/column= as ints
|
|
165
|
+
t.set_cell(row="Alice", column="q1", value=99) # by label — row=/column= as strings
|
|
166
|
+
t.set_row("Bob", [50, 60]) # replace a row (length must match)
|
|
167
|
+
t.set_column("q1", [1, 2]) # replace a column (length must match)
|
|
168
|
+
|
|
169
|
+
t.add_row("Carol", [1, 2]) # append a labelled row
|
|
170
|
+
t.add_column("q3", [5, 6]) # append a labelled column
|
|
171
|
+
|
|
172
|
+
t.drop_row("Bob")
|
|
173
|
+
t.drop_column("q2")
|
|
174
|
+
|
|
175
|
+
t.rename_row("Alice", "ALICE")
|
|
176
|
+
t.rename_column("q1", "Q1")
|
|
177
|
+
t.corner = "name"
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
`set_cell` only ever touches **data** — addressing a header, row label, or the
|
|
181
|
+
corner by position raises `ValueError`; use `rename_row`, `rename_column`, or
|
|
182
|
+
`t.corner = value` for those. Wrong-length values raise `ValueError`; unknown
|
|
183
|
+
labels raise `KeyError`; a non-`str` row label or column header (in `add_row`,
|
|
184
|
+
`add_column`, `rename_row`, `rename_column`) raises `TypeError`.
|
|
185
|
+
|
|
186
|
+
You can also build a table from nothing:
|
|
187
|
+
|
|
188
|
+
```python
|
|
189
|
+
t = Table([], column_headers=["q1", "q2"])
|
|
190
|
+
t.add_row("Alice", [10, 20])
|
|
191
|
+
t.write("new.xlsx")
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
### Writing
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
t.write(path, sheet=None)
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
Reassembles headers into row 1 and labels into column A, then writes the whole
|
|
201
|
+
sheet. Other sheets in the file are left untouched.
|
|
202
|
+
|
|
203
|
+
### Equality
|
|
204
|
+
|
|
205
|
+
```python
|
|
206
|
+
t1 == t2 # compares data, headers, labels, corner
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
Row count and row-label membership go through `t.data` instead of `len()`/`in`,
|
|
210
|
+
so the call site says what's being checked: `len(t.data.rows)`,
|
|
211
|
+
`"Bob" in t.data.row_labels`.
|
|
212
|
+
|
|
213
|
+
## The raw layer
|
|
214
|
+
|
|
215
|
+
For sheets that are not a labelled table — plain grids, exports, odd layouts.
|
|
216
|
+
|
|
217
|
+
```python
|
|
218
|
+
from pyhandlexl import read_sheet, write_sheet, append_rows
|
|
219
|
+
|
|
220
|
+
read_sheet(path, sheet=None, *, pad=False)
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
Returns `list[list[str]]`. Trailing empty cells are trimmed from each row (a
|
|
224
|
+
fully empty row becomes `[]`); `pad=True` right-pads every row to the widest
|
|
225
|
+
row's length instead.
|
|
226
|
+
|
|
227
|
+
```python
|
|
228
|
+
write_sheet(path, rows, sheet=None, *, orientation="rows")
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
Replaces the target sheet with `rows` (other sheets untouched), creating the
|
|
232
|
+
file and sheet if needed. Values are written as-is — `str` stays `str`, `int`
|
|
233
|
+
stays `int`, `None` leaves the cell empty; there is no string-to-number
|
|
234
|
+
conversion. `orientation="columns"` writes each inner list *down a column*
|
|
235
|
+
instead of across a row.
|
|
236
|
+
|
|
237
|
+
```python
|
|
238
|
+
append_rows(path, rows, sheet=None)
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
Appends after the last row. Empty input is a no-op.
|
|
242
|
+
|
|
243
|
+
## Sheet management
|
|
244
|
+
|
|
245
|
+
```python
|
|
246
|
+
from pyhandlexl import (
|
|
247
|
+
list_sheets, sheet_exists, create_sheet, delete_sheet, rename_sheet,
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
list_sheets(path) # ['Sheet1', 'Data']
|
|
251
|
+
sheet_exists(path, "Data") # True
|
|
252
|
+
create_sheet(path, "Results") # ValueError if it already exists
|
|
253
|
+
delete_sheet(path, "Old") # refuses to delete the last sheet
|
|
254
|
+
rename_sheet(path, "Old", "New")
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
Sheet names are validated everywhere: max 31 characters, none of `\ / ? * [ ] :`,
|
|
258
|
+
and `"History"` is reserved by Excel.
|
|
259
|
+
|
|
260
|
+
## Safe writes
|
|
261
|
+
|
|
262
|
+
Every write goes through the same steps:
|
|
263
|
+
|
|
264
|
+
1. Save to a temporary file in the same directory.
|
|
265
|
+
2. Verify it is a readable `.xlsx`.
|
|
266
|
+
3. Atomically replace the original (`os.replace`).
|
|
267
|
+
|
|
268
|
+
If any step fails the temporary file is removed and the original is left exactly
|
|
269
|
+
as it was. If the target is locked (open in Excel), writes retry briefly before
|
|
270
|
+
raising `FileLockedError`.
|
|
271
|
+
|
|
272
|
+
## Errors
|
|
273
|
+
|
|
274
|
+
All raised exceptions derive from `PyhandlexlError`:
|
|
275
|
+
|
|
276
|
+
| Exception | Also a | Meaning |
|
|
277
|
+
|---|---|---|
|
|
278
|
+
| `SheetNameError` | `ValueError` | invalid worksheet name |
|
|
279
|
+
| `DimensionError` | `ValueError` | data exceeds Excel's 1,048,576 × 16,384 grid |
|
|
280
|
+
| `SheetNotFoundError` | `KeyError` | no worksheet with that name |
|
|
281
|
+
| `FileLockedError` | `OSError` | file stayed locked through every retry |
|
|
282
|
+
| `InvalidFileError` | — | file is missing or not a readable `.xlsx` |
|
|
283
|
+
|
|
284
|
+
## Not in scope
|
|
285
|
+
|
|
286
|
+
`pyhandlexl` deliberately does **not** handle: cell formatting, styles, fonts,
|
|
287
|
+
formulas, charts, images, merged cells, `.xls` (old format), or password
|
|
288
|
+
protection / encryption. For any of that, use openpyxl directly.
|
|
289
|
+
|
|
290
|
+
## Development
|
|
291
|
+
|
|
292
|
+
```bash
|
|
293
|
+
python -m venv .venv
|
|
294
|
+
source .venv/bin/activate # Windows: .venv\Scripts\Activate.ps1
|
|
295
|
+
pip install -e ".[dev]"
|
|
296
|
+
pytest
|
|
297
|
+
ruff check . && ruff format --check .
|
|
298
|
+
```
|
|
299
|
+
|
|
300
|
+
## License
|
|
301
|
+
|
|
302
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
pyhandlexl/__init__.py,sha256=GJdaBC0sGOE-t0ib6aqNnUh1K4RvcZf04QRqimogL2Y,968
|
|
2
|
+
pyhandlexl/_safety.py,sha256=hsT5ZWTQG5nqSnhbnn4MwHb4X7dNbbsVXaEbmOX7TG4,3569
|
|
3
|
+
pyhandlexl/core.py,sha256=9-mmAQbTHl-o92ro8jc_FJ3HK_bInL9uaHuxyzEMhAQ,8089
|
|
4
|
+
pyhandlexl/errors.py,sha256=92b69FELB8gXMSdzKaECusSwgnrSt2FQmeoTeesj29M,828
|
|
5
|
+
pyhandlexl/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
pyhandlexl/table.py,sha256=_Gp3MCSUhZ0OwW9aNG7_xfI_IzbTwsendZBM6Zrvj9k,15385
|
|
7
|
+
pyhandlexl/validate.py,sha256=JGALihqR72V69uB6dmmivY7PG-xTFq-j8YRpY6xsWrQ,2492
|
|
8
|
+
pyhandlexl-0.2.0.dist-info/licenses/LICENSE,sha256=wW_wwDfDJzKBJB_4uifSG5f4AyyF7m-WMurztbxDx2Q,1070
|
|
9
|
+
pyhandlexl-0.2.0.dist-info/METADATA,sha256=N_3nkRJuqwkSnnurSgFg6_MQB8zTvrxLqBkPkAJ85_c,10137
|
|
10
|
+
pyhandlexl-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
11
|
+
pyhandlexl-0.2.0.dist-info/top_level.txt,sha256=VudupJVAGQn04F8AY6kUH4FsDOiX_AhkNpp80Tb80z4,11
|
|
12
|
+
pyhandlexl-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Lewis Wainaina
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pyhandlexl
|