pyOfficeEditor 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyofficeeditor-0.1.0/.gitignore +47 -0
- pyofficeeditor-0.1.0/LICENSE.md +21 -0
- pyofficeeditor-0.1.0/PKG-INFO +317 -0
- pyofficeeditor-0.1.0/README.md +280 -0
- pyofficeeditor-0.1.0/docs/architecture.md +579 -0
- pyofficeeditor-0.1.0/docs/changelog.md +82 -0
- pyofficeeditor-0.1.0/pyproject.toml +137 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/__init__.py +46 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/_xml.py +707 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/_zip.py +470 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/__init__.py +78 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_dimensions.py +192 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_formats.py +670 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_formulas.py +604 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_names.py +174 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_reference.py +378 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_rowcol.py +742 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_schema.py +141 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_sharedstrings.py +160 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_styles.py +607 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_tables.py +416 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_tokens.py +258 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/_values.py +381 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/workbook.py +906 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/excel/worksheet.py +1289 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/exceptions.py +38 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/opc.py +551 -0
- pyofficeeditor-0.1.0/src/pyofficeeditor/py.typed +0 -0
- pyofficeeditor-0.1.0/tests/fixtures/excel/README.md +72 -0
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# Agent working files. Local only: they point at absolute paths on the
|
|
2
|
+
# author's machine, so they mean nothing to anyone who clones this.
|
|
3
|
+
AGENTS.md
|
|
4
|
+
CLAUDE.md
|
|
5
|
+
.claude/
|
|
6
|
+
|
|
7
|
+
# Python bytecode
|
|
8
|
+
__pycache__/
|
|
9
|
+
*.py[cod]
|
|
10
|
+
*$py.class
|
|
11
|
+
|
|
12
|
+
# Build output
|
|
13
|
+
build/
|
|
14
|
+
dist/
|
|
15
|
+
*.egg-info/
|
|
16
|
+
.eggs/
|
|
17
|
+
|
|
18
|
+
# Virtual environments
|
|
19
|
+
.venv/
|
|
20
|
+
venv/
|
|
21
|
+
env/
|
|
22
|
+
|
|
23
|
+
# Test and type-check caches
|
|
24
|
+
.pytest_cache/
|
|
25
|
+
.ruff_cache/
|
|
26
|
+
.mypy_cache/
|
|
27
|
+
.pyright/
|
|
28
|
+
.coverage
|
|
29
|
+
.coverage.*
|
|
30
|
+
coverage.xml
|
|
31
|
+
htmlcov/
|
|
32
|
+
|
|
33
|
+
# Test artifacts written during a run
|
|
34
|
+
tests/output/
|
|
35
|
+
|
|
36
|
+
# Local environment
|
|
37
|
+
.env
|
|
38
|
+
.env.local
|
|
39
|
+
.env.*.local
|
|
40
|
+
|
|
41
|
+
# OS noise
|
|
42
|
+
.DS_Store
|
|
43
|
+
Thumbs.db
|
|
44
|
+
desktop.ini
|
|
45
|
+
|
|
46
|
+
# Office lock files, left behind when a live host test is interrupted
|
|
47
|
+
~$*
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 William Smith
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,317 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pyOfficeEditor
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Edit the document surface of Excel, Word, PowerPoint and Access in pure Python, no dependencies.
|
|
5
|
+
Project-URL: Homepage, https://github.com/WilliamSmithEdward/pyOfficeEditor
|
|
6
|
+
Project-URL: Repository, https://github.com/WilliamSmithEdward/pyOfficeEditor
|
|
7
|
+
Project-URL: Issues, https://github.com/WilliamSmithEdward/pyOfficeEditor/issues
|
|
8
|
+
Project-URL: Documentation, https://github.com/WilliamSmithEdward/pyOfficeEditor#readme
|
|
9
|
+
Author-email: William Smith <williamsmithe@icloud.com>
|
|
10
|
+
License: MIT
|
|
11
|
+
License-File: LICENSE.md
|
|
12
|
+
Keywords: accdb,access,docx,excel,formula,office,ooxml,opc,power-query,powerpoint,pptx,spreadsheetml,word,xlsm,xlsx
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
25
|
+
Classifier: Topic :: Office/Business
|
|
26
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
27
|
+
Classifier: Typing :: Typed
|
|
28
|
+
Requires-Python: >=3.10
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: build>=1.2; extra == 'dev'
|
|
31
|
+
Requires-Dist: openpyxl>=3.1; extra == 'dev'
|
|
32
|
+
Requires-Dist: pyright>=1.1.350; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
34
|
+
Requires-Dist: ruff>=0.15; extra == 'dev'
|
|
35
|
+
Requires-Dist: twine>=5; extra == 'dev'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# pyOfficeEditor
|
|
39
|
+
|
|
40
|
+
Edit the document surface of Microsoft Office files in pure Python. No Office
|
|
41
|
+
installation, no COM, no dependencies.
|
|
42
|
+
|
|
43
|
+
Its sister project [pyOpenVBA](https://github.com/WilliamSmithEdward/pyOpenVBA)
|
|
44
|
+
edits the VBA project inside an Office file. This one edits the document: the
|
|
45
|
+
cell, the formula, the paragraph, the slide, the table, the query.
|
|
46
|
+
|
|
47
|
+
> **Status: early, and growing.** The Excel surface reads and writes cells,
|
|
48
|
+
> values, formulas, dates, sheets, formatting, merged ranges, tables, row and
|
|
49
|
+
> column dimensions, frozen panes and defined names, each verified against real
|
|
50
|
+
> Excel. Rows and columns can be inserted and deleted, with every reference
|
|
51
|
+
> in the workbook following or breaking exactly as Excel breaks it,
|
|
52
|
+
> `A:A` and `2:4` included. Conditional formatting and data validation are
|
|
53
|
+
> the next gap. Word, PowerPoint and Access follow, in that order.
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
import datetime as dt
|
|
57
|
+
from pyofficeeditor.excel import Border, Workbook
|
|
58
|
+
|
|
59
|
+
with Workbook.open("orders.xlsx") as book:
|
|
60
|
+
sheet = book["Data"]
|
|
61
|
+
|
|
62
|
+
sheet["A2"].value # 'North' stored as an index
|
|
63
|
+
sheet["F2"].value # datetime.date(2026, 1, 15) stored as 46037
|
|
64
|
+
sheet["D3"].formula # 'B3*C3' stored nowhere at all
|
|
65
|
+
sheet["B8"].value # CellError('#DIV/0!') not the text of one
|
|
66
|
+
|
|
67
|
+
sheet["B2"].value = 200
|
|
68
|
+
sheet["G1"].value = dt.date(2026, 7, 4)
|
|
69
|
+
sheet["G2"].formula = "=SUM(B2:B5)"
|
|
70
|
+
|
|
71
|
+
sheet.range("A1:F1").apply_font(bold=True) # each cell keeps its own rest
|
|
72
|
+
sheet["B2"].fill = "FFFF00"
|
|
73
|
+
sheet["B3"].border = Border.all_sides("thin", "FF0000")
|
|
74
|
+
|
|
75
|
+
sheet["A20"].value = "wide heading"
|
|
76
|
+
sheet.merge("A20:C20")
|
|
77
|
+
sheet["B20"].merged_range # RangeRef('A20:C20')
|
|
78
|
+
|
|
79
|
+
table = sheet.add_table("Sales", "A1:F20", totals_row=True)
|
|
80
|
+
table.column_names # from the header row
|
|
81
|
+
table.data_range # excludes header and totals
|
|
82
|
+
|
|
83
|
+
sheet.set_row_height(1, 24) # points, exact
|
|
84
|
+
sheet.set_column_hidden(4, True)
|
|
85
|
+
sheet.freeze_panes("B2") # pins row 1 and column A
|
|
86
|
+
|
|
87
|
+
sheet.insert_rows(3, 2) # every reference follows
|
|
88
|
+
sheet.delete_columns(5, 1) # SUM(E2:E9) -> #REF!
|
|
89
|
+
|
|
90
|
+
summary = book.add_sheet("Summary", index=0)
|
|
91
|
+
summary["A1"].formula = "=SUM(Data!D2:D5)"
|
|
92
|
+
book.add_defined_name("Totals", "Data!$D$2:$D$5")
|
|
93
|
+
book.rename_sheet("Data", "Q1 Data") # formulas and names both follow
|
|
94
|
+
book.save()
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Each of those four reads has a plausible wrong answer that a naive
|
|
98
|
+
implementation gives instead: `6` for the string, `46037` for the date, an
|
|
99
|
+
empty formula for `D3`, and a string that compares equal to `"#DIV/0!"` for
|
|
100
|
+
the error. Getting them right is most of what the Excel modules do.
|
|
101
|
+
|
|
102
|
+
## Why this exists
|
|
103
|
+
|
|
104
|
+
openpyxl, python-docx and python-pptx already set a cell, a paragraph and a
|
|
105
|
+
slide. This library is aimed at the ground they leave uncovered:
|
|
106
|
+
|
|
107
|
+
- **One API across four hosts**, including Access, which none of them touch.
|
|
108
|
+
- **The legacy binary formats**, `.xls`, `.doc` and `.ppt`, which pyOpenVBA
|
|
109
|
+
deliberately treats as opaque.
|
|
110
|
+
- **The analysis layer**: Excel formula parsing and linting, legacy data
|
|
111
|
+
connections, and Power Query.
|
|
112
|
+
- **Byte fidelity as a correctness property**, not a nicety. See below.
|
|
113
|
+
|
|
114
|
+
## Byte fidelity
|
|
115
|
+
|
|
116
|
+
Editing one cell of a worksheet must leave every other byte of the package
|
|
117
|
+
alone. Otherwise a one-cell change produces a diff nobody can review, and a
|
|
118
|
+
save that should be a no-op is not one.
|
|
119
|
+
|
|
120
|
+
That turns out to rule out the obvious building blocks, for reasons that are
|
|
121
|
+
measurable rather than theoretical.
|
|
122
|
+
|
|
123
|
+
**The XML.** Excel writes a worksheet as a declaration, a CRLF, then a single
|
|
124
|
+
line whose root element declares `mc:Ignorable="x14ac xr xr2 xr3"` *before* it
|
|
125
|
+
declares the `x14ac` and `xr` prefixes themselves. Round-tripping that through
|
|
126
|
+
`xml.etree.ElementTree` renames the prefixes to `ns0` and `ns1`, reorders the
|
|
127
|
+
declarations, and drops the CRLF. So parts are parsed into a tree where every
|
|
128
|
+
node keeps the source text it was cut from. Serializing a node nobody touched
|
|
129
|
+
copies those bytes; serializing a node that changed rebuilds it and recurses.
|
|
130
|
+
Editing one cell rewrites that cell's row and copies the rest.
|
|
131
|
+
|
|
132
|
+
**The container.** In a freshly authored workbook, `[Content_Types].xml`
|
|
133
|
+
carries a 520-byte extra field in its *local* ZIP header and none in the
|
|
134
|
+
central directory. `zipfile.ZipInfo.extra` exposes only the central copy, so
|
|
135
|
+
anything built on `zipfile`'s writer silently drops those bytes. The field is
|
|
136
|
+
the Microsoft Open Packaging Growth Hint (tag `0xa220`): padding Excel reserves
|
|
137
|
+
so it can grow a part in place. This library reads and writes the container
|
|
138
|
+
itself, field for field, and copies each untouched member's stored bytes
|
|
139
|
+
without inflating them.
|
|
140
|
+
|
|
141
|
+
The gate: reading a package and writing it back with nothing changed
|
|
142
|
+
reproduces the input exactly. It holds for Excel-authored `.xlsm`, `.xlsb` and
|
|
143
|
+
`.xlsx`, for openpyxl-authored `.xlsx`, and for archives whose members carry
|
|
144
|
+
data descriptors.
|
|
145
|
+
|
|
146
|
+
## Architecture
|
|
147
|
+
|
|
148
|
+
```
|
|
149
|
+
+--------------------------------------------------------+
|
|
150
|
+
| excel/ Workbook, Worksheet, Range, Cell |
|
|
151
|
+
| _reference A1 notation, bijective base-26 columns |
|
|
152
|
+
| _values the six cell encodings, and serial dates |
|
|
153
|
+
| _styles number formats, which is how a date is |
|
|
154
|
+
| told from a number |
|
|
155
|
+
| _formats fonts, fills, borders, alignment, as |
|
|
156
|
+
| immutable values |
|
|
157
|
+
| _tables ListObjects: their own parts and wiring |
|
|
158
|
+
| _dimensions widths, heights, hiding, frozen panes |
|
|
159
|
+
| _names defined names, and the rules tables share |
|
|
160
|
+
| _rowcol inserting and deleting rows and columns, |
|
|
161
|
+
| and moving everything that records a |
|
|
162
|
+
| cell address |
|
|
163
|
+
| _tokens a formula, broken into editable pieces |
|
|
164
|
+
| _formulas shifting and breaking references, for |
|
|
165
|
+
| shared formulas, a renamed sheet, and |
|
|
166
|
+
| the #REF! a deletion leaves behind |
|
|
167
|
+
| _sharedstrings the per-workbook string table |
|
|
168
|
+
| _schema where a child element has to go |
|
|
169
|
+
+--------------------------------------------------------+
|
|
170
|
+
| word / powerpoint / access to follow, in that order |
|
|
171
|
+
+--------------------------------------------------------+
|
|
172
|
+
| opc.py Open Packaging Conventions |
|
|
173
|
+
| - parts, cached and flushed only when modified |
|
|
174
|
+
| - [Content_Types].xml: defaults and overrides |
|
|
175
|
+
| - relationships, resolved the way Office resolves |
|
|
176
|
+
+--------------------------------------------------------+
|
|
177
|
+
| _xml.py XML that reproduces its own source |
|
|
178
|
+
| - hand-written parser, no stdlib XML module |
|
|
179
|
+
| - per-node source spans and dirty propagation |
|
|
180
|
+
| - prefixes, attribute order and empty-tag form kept |
|
|
181
|
+
+--------------------------------------------------------+
|
|
182
|
+
| _zip.py the ZIP container, field for field |
|
|
183
|
+
| - local and central headers kept apart |
|
|
184
|
+
| - untouched members copied as stored bytes |
|
|
185
|
+
+--------------------------------------------------------+
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Each layer knows the one below it and not the one above. `_zip.py` knows
|
|
189
|
+
nothing about OOXML; `_xml.py` knows nothing about packages.
|
|
190
|
+
|
|
191
|
+
`docs/architecture.md` is the contributor reference.
|
|
192
|
+
|
|
193
|
+
## Untrusted input
|
|
194
|
+
|
|
195
|
+
A document is untrusted input, so the XML parser is deliberately less capable
|
|
196
|
+
than a conforming XML processor:
|
|
197
|
+
|
|
198
|
+
- `<!DOCTYPE` is refused outright. No DTD processing means no external entity
|
|
199
|
+
resolution and no entity-expansion amplification.
|
|
200
|
+
- Only the five predefined entities and numeric character references resolve.
|
|
201
|
+
Any other `&name;` raises.
|
|
202
|
+
- Nesting deeper than 256 elements raises instead of exhausting the stack.
|
|
203
|
+
- zip64 archives and compression methods other than stored and deflate are
|
|
204
|
+
refused by name rather than guessed at.
|
|
205
|
+
|
|
206
|
+
None of the five restricts anything OOXML is allowed to contain.
|
|
207
|
+
|
|
208
|
+
## Install
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
pip install pyOfficeEditor
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
Python 3.10 or newer. No runtime dependencies.
|
|
215
|
+
|
|
216
|
+
## Three things Excel does that catch readers out
|
|
217
|
+
|
|
218
|
+
Each is measured, each is pinned by a test, and each gives a wrong answer
|
|
219
|
+
rather than an error if you miss it.
|
|
220
|
+
|
|
221
|
+
**A formula assigned to a range is stored once.** Excel writes the text on the
|
|
222
|
+
group's first cell and leaves the rest pointing at it by index:
|
|
223
|
+
|
|
224
|
+
```xml
|
|
225
|
+
<c r="D2"><f t="shared" ref="D2:D5" si="0">B2*C2</f><v>510</v></c>
|
|
226
|
+
<c r="D3"><f t="shared" si="0"/><v>1445</v></c>
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
D3's formula is not in the file. It is D2's, shifted down a row. Finding the
|
|
230
|
+
references to shift is the delicate part, because `LOG10(x)` contains `G10`,
|
|
231
|
+
`"A1"` is a string literal, and `'My Sheet A1'!B2` has a reference inside a
|
|
232
|
+
quoted sheet name.
|
|
233
|
+
|
|
234
|
+
**A date is a number, and only its number format says otherwise.**
|
|
235
|
+
`2026-01-15` is stored as `46037`. Confirming it is a date means following
|
|
236
|
+
`s="2"` to `cellXfs[2]`, its `numFmtId` to a format code, and the code to its
|
|
237
|
+
date tokens, skipping the quoted, escaped and bracketed parts that only look
|
|
238
|
+
like them: `#,##0 "days"` is not a date and `[h]:mm` is. Excel also numbers
|
|
239
|
+
dates as though 1900 were a leap year, so serial 60 is a 29 February that
|
|
240
|
+
never happened and is refused rather than reported as 1 March.
|
|
241
|
+
|
|
242
|
+
**A changed cell invalidates cached results.** A formula cell stores the value
|
|
243
|
+
it last evaluated to, so setting `B2` leaves `D2`'s cached `510` behind. A
|
|
244
|
+
workbook this library modified is saved with `fullCalcOnLoad` set and the
|
|
245
|
+
`calcChain` part dropped, so Excel recalculates on open. The live gate proves
|
|
246
|
+
it: after changing an input, Excel reports the recomputed number rather than
|
|
247
|
+
the stale one still written in the file.
|
|
248
|
+
|
|
249
|
+
**A worksheet's children are a sequence, not a set.** SpreadsheetML declares
|
|
250
|
+
them in order, and Excel refuses a file that breaks it rather than repairing
|
|
251
|
+
one. The natural thing to do with a missing element is append it, and that is
|
|
252
|
+
wrong whenever anything that must follow it is already there: a sheet Excel
|
|
253
|
+
authored starts with `sheetPr`, so a missing `dimension` does not go at the
|
|
254
|
+
front, and a workbook usually ends with `extLst`, so a missing `calcPr` does
|
|
255
|
+
not go at the end. `_schema.py` writes all three orders down.
|
|
256
|
+
|
|
257
|
+
**Formatting is shared, so it cannot be edited in place.** A cell carries an
|
|
258
|
+
index into `cellXfs`, and two hundred cells may carry the same one. Changing
|
|
259
|
+
that entry restyles all of them, which is never what "embolden this cell"
|
|
260
|
+
meant. So formats are immutable values: read the cell's, derive a new one,
|
|
261
|
+
and the workbook finds or appends the entry that matches. Two consequences
|
|
262
|
+
worth knowing:
|
|
263
|
+
|
|
264
|
+
- A cell with no `s` attribute is not unformatted. It uses `cellXfs[0]`,
|
|
265
|
+
which names the workbook's default font. Resolving it to an empty format
|
|
266
|
+
instead would make "add bold" silently change the typeface.
|
|
267
|
+
- A solid fill's colour goes in `fgColor`, not `bgColor`. The names suggest
|
|
268
|
+
otherwise, and putting it in `bgColor` produces a cell that looks unfilled.
|
|
269
|
+
|
|
270
|
+
## Lower-level access
|
|
271
|
+
|
|
272
|
+
The packaging layer is public, for anything the host surfaces do not cover:
|
|
273
|
+
|
|
274
|
+
```python
|
|
275
|
+
from pyofficeeditor import OpcPackage
|
|
276
|
+
|
|
277
|
+
with OpcPackage.open("book.xlsx") as package:
|
|
278
|
+
# Navigate the way Office does: by relationship, not by path.
|
|
279
|
+
workbook_part = package.main_document_part() # 'xl/workbook.xml'
|
|
280
|
+
document = package.xml(workbook_part)
|
|
281
|
+
document.root.require("sheets")
|
|
282
|
+
package.save() # every untouched part keeps its original bytes
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
## Development
|
|
286
|
+
|
|
287
|
+
```bash
|
|
288
|
+
python -m pip install -e ".[dev]"
|
|
289
|
+
python -m pytest -p no:randomly
|
|
290
|
+
pyright src tests
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
`-p no:randomly` keeps ordering reproducible. Strict Pyright must report zero
|
|
294
|
+
errors on `src` and `tests` before anything merges, and a behavior change lands
|
|
295
|
+
with its test in the same commit.
|
|
296
|
+
|
|
297
|
+
The suite needs no Office installation. It tests against two committed
|
|
298
|
+
Excel-authored packages and against openpyxl-authored ones generated during the
|
|
299
|
+
run, because a reader that only ever sees one producer's output encodes that
|
|
300
|
+
producer's habits as rules.
|
|
301
|
+
|
|
302
|
+
There is also a live gate, which is the only check that can prove Excel
|
|
303
|
+
accepts what this library writes. It drives real Excel through
|
|
304
|
+
[pyVBAharness](https://github.com/WilliamSmithEdward/pyVBAharness), opens an
|
|
305
|
+
edited workbook, and reads the cells back through Excel's own object model:
|
|
306
|
+
|
|
307
|
+
```bash
|
|
308
|
+
python -m pip install -e ".[dev]" --group live
|
|
309
|
+
python scripts/build_excel_fixtures.py
|
|
310
|
+
RUN_LIVE_EXCEL=1 python -m pytest -m live -p no:randomly
|
|
311
|
+
```
|
|
312
|
+
|
|
313
|
+
Windows and Excel only. Everything else in the suite runs anywhere.
|
|
314
|
+
|
|
315
|
+
## License
|
|
316
|
+
|
|
317
|
+
MIT. See [LICENSE.md](LICENSE.md).
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
# pyOfficeEditor
|
|
2
|
+
|
|
3
|
+
Edit the document surface of Microsoft Office files in pure Python. No Office
|
|
4
|
+
installation, no COM, no dependencies.
|
|
5
|
+
|
|
6
|
+
Its sister project [pyOpenVBA](https://github.com/WilliamSmithEdward/pyOpenVBA)
|
|
7
|
+
edits the VBA project inside an Office file. This one edits the document: the
|
|
8
|
+
cell, the formula, the paragraph, the slide, the table, the query.
|
|
9
|
+
|
|
10
|
+
> **Status: early, and growing.** The Excel surface reads and writes cells,
|
|
11
|
+
> values, formulas, dates, sheets, formatting, merged ranges, tables, row and
|
|
12
|
+
> column dimensions, frozen panes and defined names, each verified against real
|
|
13
|
+
> Excel. Rows and columns can be inserted and deleted, with every reference
|
|
14
|
+
> in the workbook following or breaking exactly as Excel breaks it,
|
|
15
|
+
> `A:A` and `2:4` included. Conditional formatting and data validation are
|
|
16
|
+
> the next gap. Word, PowerPoint and Access follow, in that order.
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
import datetime as dt
|
|
20
|
+
from pyofficeeditor.excel import Border, Workbook
|
|
21
|
+
|
|
22
|
+
with Workbook.open("orders.xlsx") as book:
|
|
23
|
+
sheet = book["Data"]
|
|
24
|
+
|
|
25
|
+
sheet["A2"].value # 'North' stored as an index
|
|
26
|
+
sheet["F2"].value # datetime.date(2026, 1, 15) stored as 46037
|
|
27
|
+
sheet["D3"].formula # 'B3*C3' stored nowhere at all
|
|
28
|
+
sheet["B8"].value # CellError('#DIV/0!') not the text of one
|
|
29
|
+
|
|
30
|
+
sheet["B2"].value = 200
|
|
31
|
+
sheet["G1"].value = dt.date(2026, 7, 4)
|
|
32
|
+
sheet["G2"].formula = "=SUM(B2:B5)"
|
|
33
|
+
|
|
34
|
+
sheet.range("A1:F1").apply_font(bold=True) # each cell keeps its own rest
|
|
35
|
+
sheet["B2"].fill = "FFFF00"
|
|
36
|
+
sheet["B3"].border = Border.all_sides("thin", "FF0000")
|
|
37
|
+
|
|
38
|
+
sheet["A20"].value = "wide heading"
|
|
39
|
+
sheet.merge("A20:C20")
|
|
40
|
+
sheet["B20"].merged_range # RangeRef('A20:C20')
|
|
41
|
+
|
|
42
|
+
table = sheet.add_table("Sales", "A1:F20", totals_row=True)
|
|
43
|
+
table.column_names # from the header row
|
|
44
|
+
table.data_range # excludes header and totals
|
|
45
|
+
|
|
46
|
+
sheet.set_row_height(1, 24) # points, exact
|
|
47
|
+
sheet.set_column_hidden(4, True)
|
|
48
|
+
sheet.freeze_panes("B2") # pins row 1 and column A
|
|
49
|
+
|
|
50
|
+
sheet.insert_rows(3, 2) # every reference follows
|
|
51
|
+
sheet.delete_columns(5, 1) # SUM(E2:E9) -> #REF!
|
|
52
|
+
|
|
53
|
+
summary = book.add_sheet("Summary", index=0)
|
|
54
|
+
summary["A1"].formula = "=SUM(Data!D2:D5)"
|
|
55
|
+
book.add_defined_name("Totals", "Data!$D$2:$D$5")
|
|
56
|
+
book.rename_sheet("Data", "Q1 Data") # formulas and names both follow
|
|
57
|
+
book.save()
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Each of those four reads has a plausible wrong answer that a naive
|
|
61
|
+
implementation gives instead: `6` for the string, `46037` for the date, an
|
|
62
|
+
empty formula for `D3`, and a string that compares equal to `"#DIV/0!"` for
|
|
63
|
+
the error. Getting them right is most of what the Excel modules do.
|
|
64
|
+
|
|
65
|
+
## Why this exists
|
|
66
|
+
|
|
67
|
+
openpyxl, python-docx and python-pptx already set a cell, a paragraph and a
|
|
68
|
+
slide. This library is aimed at the ground they leave uncovered:
|
|
69
|
+
|
|
70
|
+
- **One API across four hosts**, including Access, which none of them touch.
|
|
71
|
+
- **The legacy binary formats**, `.xls`, `.doc` and `.ppt`, which pyOpenVBA
|
|
72
|
+
deliberately treats as opaque.
|
|
73
|
+
- **The analysis layer**: Excel formula parsing and linting, legacy data
|
|
74
|
+
connections, and Power Query.
|
|
75
|
+
- **Byte fidelity as a correctness property**, not a nicety. See below.
|
|
76
|
+
|
|
77
|
+
## Byte fidelity
|
|
78
|
+
|
|
79
|
+
Editing one cell of a worksheet must leave every other byte of the package
|
|
80
|
+
alone. Otherwise a one-cell change produces a diff nobody can review, and a
|
|
81
|
+
save that should be a no-op is not one.
|
|
82
|
+
|
|
83
|
+
That turns out to rule out the obvious building blocks, for reasons that are
|
|
84
|
+
measurable rather than theoretical.
|
|
85
|
+
|
|
86
|
+
**The XML.** Excel writes a worksheet as a declaration, a CRLF, then a single
|
|
87
|
+
line whose root element declares `mc:Ignorable="x14ac xr xr2 xr3"` *before* it
|
|
88
|
+
declares the `x14ac` and `xr` prefixes themselves. Round-tripping that through
|
|
89
|
+
`xml.etree.ElementTree` renames the prefixes to `ns0` and `ns1`, reorders the
|
|
90
|
+
declarations, and drops the CRLF. So parts are parsed into a tree where every
|
|
91
|
+
node keeps the source text it was cut from. Serializing a node nobody touched
|
|
92
|
+
copies those bytes; serializing a node that changed rebuilds it and recurses.
|
|
93
|
+
Editing one cell rewrites that cell's row and copies the rest.
|
|
94
|
+
|
|
95
|
+
**The container.** In a freshly authored workbook, `[Content_Types].xml`
|
|
96
|
+
carries a 520-byte extra field in its *local* ZIP header and none in the
|
|
97
|
+
central directory. `zipfile.ZipInfo.extra` exposes only the central copy, so
|
|
98
|
+
anything built on `zipfile`'s writer silently drops those bytes. The field is
|
|
99
|
+
the Microsoft Open Packaging Growth Hint (tag `0xa220`): padding Excel reserves
|
|
100
|
+
so it can grow a part in place. This library reads and writes the container
|
|
101
|
+
itself, field for field, and copies each untouched member's stored bytes
|
|
102
|
+
without inflating them.
|
|
103
|
+
|
|
104
|
+
The gate: reading a package and writing it back with nothing changed
|
|
105
|
+
reproduces the input exactly. It holds for Excel-authored `.xlsm`, `.xlsb` and
|
|
106
|
+
`.xlsx`, for openpyxl-authored `.xlsx`, and for archives whose members carry
|
|
107
|
+
data descriptors.
|
|
108
|
+
|
|
109
|
+
## Architecture
|
|
110
|
+
|
|
111
|
+
```
|
|
112
|
+
+--------------------------------------------------------+
|
|
113
|
+
| excel/ Workbook, Worksheet, Range, Cell |
|
|
114
|
+
| _reference A1 notation, bijective base-26 columns |
|
|
115
|
+
| _values the six cell encodings, and serial dates |
|
|
116
|
+
| _styles number formats, which is how a date is |
|
|
117
|
+
| told from a number |
|
|
118
|
+
| _formats fonts, fills, borders, alignment, as |
|
|
119
|
+
| immutable values |
|
|
120
|
+
| _tables ListObjects: their own parts and wiring |
|
|
121
|
+
| _dimensions widths, heights, hiding, frozen panes |
|
|
122
|
+
| _names defined names, and the rules tables share |
|
|
123
|
+
| _rowcol inserting and deleting rows and columns, |
|
|
124
|
+
| and moving everything that records a |
|
|
125
|
+
| cell address |
|
|
126
|
+
| _tokens a formula, broken into editable pieces |
|
|
127
|
+
| _formulas shifting and breaking references, for |
|
|
128
|
+
| shared formulas, a renamed sheet, and |
|
|
129
|
+
| the #REF! a deletion leaves behind |
|
|
130
|
+
| _sharedstrings the per-workbook string table |
|
|
131
|
+
| _schema where a child element has to go |
|
|
132
|
+
+--------------------------------------------------------+
|
|
133
|
+
| word / powerpoint / access to follow, in that order |
|
|
134
|
+
+--------------------------------------------------------+
|
|
135
|
+
| opc.py Open Packaging Conventions |
|
|
136
|
+
| - parts, cached and flushed only when modified |
|
|
137
|
+
| - [Content_Types].xml: defaults and overrides |
|
|
138
|
+
| - relationships, resolved the way Office resolves |
|
|
139
|
+
+--------------------------------------------------------+
|
|
140
|
+
| _xml.py XML that reproduces its own source |
|
|
141
|
+
| - hand-written parser, no stdlib XML module |
|
|
142
|
+
| - per-node source spans and dirty propagation |
|
|
143
|
+
| - prefixes, attribute order and empty-tag form kept |
|
|
144
|
+
+--------------------------------------------------------+
|
|
145
|
+
| _zip.py the ZIP container, field for field |
|
|
146
|
+
| - local and central headers kept apart |
|
|
147
|
+
| - untouched members copied as stored bytes |
|
|
148
|
+
+--------------------------------------------------------+
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Each layer knows the one below it and not the one above. `_zip.py` knows
|
|
152
|
+
nothing about OOXML; `_xml.py` knows nothing about packages.
|
|
153
|
+
|
|
154
|
+
`docs/architecture.md` is the contributor reference.
|
|
155
|
+
|
|
156
|
+
## Untrusted input
|
|
157
|
+
|
|
158
|
+
A document is untrusted input, so the XML parser is deliberately less capable
|
|
159
|
+
than a conforming XML processor:
|
|
160
|
+
|
|
161
|
+
- `<!DOCTYPE` is refused outright. No DTD processing means no external entity
|
|
162
|
+
resolution and no entity-expansion amplification.
|
|
163
|
+
- Only the five predefined entities and numeric character references resolve.
|
|
164
|
+
Any other `&name;` raises.
|
|
165
|
+
- Nesting deeper than 256 elements raises instead of exhausting the stack.
|
|
166
|
+
- zip64 archives and compression methods other than stored and deflate are
|
|
167
|
+
refused by name rather than guessed at.
|
|
168
|
+
|
|
169
|
+
None of the five restricts anything OOXML is allowed to contain.
|
|
170
|
+
|
|
171
|
+
## Install
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
pip install pyOfficeEditor
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
Python 3.10 or newer. No runtime dependencies.
|
|
178
|
+
|
|
179
|
+
## Three things Excel does that catch readers out
|
|
180
|
+
|
|
181
|
+
Each is measured, each is pinned by a test, and each gives a wrong answer
|
|
182
|
+
rather than an error if you miss it.
|
|
183
|
+
|
|
184
|
+
**A formula assigned to a range is stored once.** Excel writes the text on the
|
|
185
|
+
group's first cell and leaves the rest pointing at it by index:
|
|
186
|
+
|
|
187
|
+
```xml
|
|
188
|
+
<c r="D2"><f t="shared" ref="D2:D5" si="0">B2*C2</f><v>510</v></c>
|
|
189
|
+
<c r="D3"><f t="shared" si="0"/><v>1445</v></c>
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
D3's formula is not in the file. It is D2's, shifted down a row. Finding the
|
|
193
|
+
references to shift is the delicate part, because `LOG10(x)` contains `G10`,
|
|
194
|
+
`"A1"` is a string literal, and `'My Sheet A1'!B2` has a reference inside a
|
|
195
|
+
quoted sheet name.
|
|
196
|
+
|
|
197
|
+
**A date is a number, and only its number format says otherwise.**
|
|
198
|
+
`2026-01-15` is stored as `46037`. Confirming it is a date means following
|
|
199
|
+
`s="2"` to `cellXfs[2]`, its `numFmtId` to a format code, and the code to its
|
|
200
|
+
date tokens, skipping the quoted, escaped and bracketed parts that only look
|
|
201
|
+
like them: `#,##0 "days"` is not a date and `[h]:mm` is. Excel also numbers
|
|
202
|
+
dates as though 1900 were a leap year, so serial 60 is a 29 February that
|
|
203
|
+
never happened and is refused rather than reported as 1 March.
|
|
204
|
+
|
|
205
|
+
**A changed cell invalidates cached results.** A formula cell stores the value
|
|
206
|
+
it last evaluated to, so setting `B2` leaves `D2`'s cached `510` behind. A
|
|
207
|
+
workbook this library modified is saved with `fullCalcOnLoad` set and the
|
|
208
|
+
`calcChain` part dropped, so Excel recalculates on open. The live gate proves
|
|
209
|
+
it: after changing an input, Excel reports the recomputed number rather than
|
|
210
|
+
the stale one still written in the file.
|
|
211
|
+
|
|
212
|
+
**A worksheet's children are a sequence, not a set.** SpreadsheetML declares
|
|
213
|
+
them in order, and Excel refuses a file that breaks it rather than repairing
|
|
214
|
+
one. The natural thing to do with a missing element is append it, and that is
|
|
215
|
+
wrong whenever anything that must follow it is already there: a sheet Excel
|
|
216
|
+
authored starts with `sheetPr`, so a missing `dimension` does not go at the
|
|
217
|
+
front, and a workbook usually ends with `extLst`, so a missing `calcPr` does
|
|
218
|
+
not go at the end. `_schema.py` writes all three orders down.
|
|
219
|
+
|
|
220
|
+
**Formatting is shared, so it cannot be edited in place.** A cell carries an
|
|
221
|
+
index into `cellXfs`, and two hundred cells may carry the same one. Changing
|
|
222
|
+
that entry restyles all of them, which is never what "embolden this cell"
|
|
223
|
+
meant. So formats are immutable values: read the cell's, derive a new one,
|
|
224
|
+
and the workbook finds or appends the entry that matches. Two consequences
|
|
225
|
+
worth knowing:
|
|
226
|
+
|
|
227
|
+
- A cell with no `s` attribute is not unformatted. It uses `cellXfs[0]`,
|
|
228
|
+
which names the workbook's default font. Resolving it to an empty format
|
|
229
|
+
instead would make "add bold" silently change the typeface.
|
|
230
|
+
- A solid fill's colour goes in `fgColor`, not `bgColor`. The names suggest
|
|
231
|
+
otherwise, and putting it in `bgColor` produces a cell that looks unfilled.
|
|
232
|
+
|
|
233
|
+
## Lower-level access
|
|
234
|
+
|
|
235
|
+
The packaging layer is public, for anything the host surfaces do not cover:
|
|
236
|
+
|
|
237
|
+
```python
|
|
238
|
+
from pyofficeeditor import OpcPackage
|
|
239
|
+
|
|
240
|
+
with OpcPackage.open("book.xlsx") as package:
|
|
241
|
+
# Navigate the way Office does: by relationship, not by path.
|
|
242
|
+
workbook_part = package.main_document_part() # 'xl/workbook.xml'
|
|
243
|
+
document = package.xml(workbook_part)
|
|
244
|
+
document.root.require("sheets")
|
|
245
|
+
package.save() # every untouched part keeps its original bytes
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
## Development
|
|
249
|
+
|
|
250
|
+
```bash
|
|
251
|
+
python -m pip install -e ".[dev]"
|
|
252
|
+
python -m pytest -p no:randomly
|
|
253
|
+
pyright src tests
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
`-p no:randomly` keeps ordering reproducible. Strict Pyright must report zero
|
|
257
|
+
errors on `src` and `tests` before anything merges, and a behavior change lands
|
|
258
|
+
with its test in the same commit.
|
|
259
|
+
|
|
260
|
+
The suite needs no Office installation. It tests against two committed
|
|
261
|
+
Excel-authored packages and against openpyxl-authored ones generated during the
|
|
262
|
+
run, because a reader that only ever sees one producer's output encodes that
|
|
263
|
+
producer's habits as rules.
|
|
264
|
+
|
|
265
|
+
There is also a live gate, which is the only check that can prove Excel
|
|
266
|
+
accepts what this library writes. It drives real Excel through
|
|
267
|
+
[pyVBAharness](https://github.com/WilliamSmithEdward/pyVBAharness), opens an
|
|
268
|
+
edited workbook, and reads the cells back through Excel's own object model:
|
|
269
|
+
|
|
270
|
+
```bash
|
|
271
|
+
python -m pip install -e ".[dev]" --group live
|
|
272
|
+
python scripts/build_excel_fixtures.py
|
|
273
|
+
RUN_LIVE_EXCEL=1 python -m pytest -m live -p no:randomly
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
Windows and Excel only. Everything else in the suite runs anywhere.
|
|
277
|
+
|
|
278
|
+
## License
|
|
279
|
+
|
|
280
|
+
MIT. See [LICENSE.md](LICENSE.md).
|