pyOpenVBA 5.1.3__tar.gz → 5.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/PKG-INFO +8 -2
  2. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/README.md +6 -1
  3. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/power_query.md +41 -0
  4. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/pyproject.toml +5 -1
  5. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/__init__.py +1 -1
  6. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_vba.py +46 -16
  7. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/database.py +181 -45
  8. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_opc.py +25 -0
  9. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_sheets.py +75 -7
  10. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/workbook.py +15 -0
  11. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/.gitignore +0 -0
  12. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/LICENSE.md +0 -0
  13. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/architecture.md +0 -0
  14. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/ms-ovba-implementation-guide_v2.md +0 -0
  15. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/research/access_complex/README.md +0 -0
  16. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/research/access_designs/README.md +0 -0
  17. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/research/access_macros/README.md +0 -0
  18. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/research/access_write/README.md +0 -0
  19. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/research/pcode/README.md +0 -0
  20. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/docs/roadmap.md +0 -0
  21. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/__main__.py +0 -0
  22. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_deflate.py +0 -0
  23. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_host.py +0 -0
  24. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_oforms_pages.py +0 -0
  25. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_oforms_records.py +0 -0
  26. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_ppt_container.py +0 -0
  27. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/__init__.py +0 -0
  28. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_database.accdb +0 -0
  29. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_database_module.accdb +0 -0
  30. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_document.docm +0 -0
  31. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_excel_addin.xlam +0 -0
  32. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_presentation.pptm +0 -0
  33. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_workbook.xlsb +0 -0
  34. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_workbook.xlsm +0 -0
  35. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/blank_workbook.xlsx +0 -0
  36. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/engine_skeleton.accdb +0 -0
  37. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/blank_files/engine_skeleton.mdb +0 -0
  38. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/form.blob +0 -0
  39. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/form.lvprop +0 -0
  40. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/form.propdata +0 -0
  41. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/form.prototypes +0 -0
  42. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/form.typeinfo +0 -0
  43. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/report.blob +0 -0
  44. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/report.lvprop +0 -0
  45. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/report.propdata +0 -0
  46. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/report.prototypes +0 -0
  47. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/_templates/designs/report.typeinfo +0 -0
  48. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/__init__.py +0 -0
  49. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_alloc.py +0 -0
  50. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_btree.py +0 -0
  51. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_collation.py +0 -0
  52. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_collation_general_legacy.py +0 -0
  53. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_compact.py +0 -0
  54. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_complex.py +0 -0
  55. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_datapage.py +0 -0
  56. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_ddl.py +0 -0
  57. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_designs.py +0 -0
  58. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_facade.py +0 -0
  59. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_format.py +0 -0
  60. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_index.py +0 -0
  61. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_lval.py +0 -0
  62. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_macros.py +0 -0
  63. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_pages.py +0 -0
  64. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_props.py +0 -0
  65. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_queries.py +0 -0
  66. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_rows.py +0 -0
  67. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_schema.py +0 -0
  68. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_sql.py +0 -0
  69. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_storage.py +0 -0
  70. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_tdef.py +0 -0
  71. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access/_validate.py +0 -0
  72. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/access_read.py +0 -0
  73. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/cfb.py +0 -0
  74. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/excel.py +0 -0
  75. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/exceptions.py +0 -0
  76. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/forms.py +0 -0
  77. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerpoint.py +0 -0
  78. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/__init__.py +0 -0
  79. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_binary.py +0 -0
  80. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_files.py +0 -0
  81. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_mashup.py +0 -0
  82. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_metadata.py +0 -0
  83. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_package.py +0 -0
  84. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_refresh.py +0 -0
  85. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/powerquery/_section.py +0 -0
  86. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/vba.py +0 -0
  87. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/vba_pcode.py +0 -0
  88. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/src/pyopenvba/word.py +0 -0
  89. {pyopenvba-5.1.3 → pyopenvba-5.2.1}/tests/fuzz_corpus/README.md +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pyOpenVBA
3
- Version: 5.1.3
3
+ Version: 5.2.1
4
4
  Summary: Read and write VBA macros and Excel Power Query in pure Python, no dependencies.
5
5
  Project-URL: Homepage, https://github.com/WilliamSmithEdward/pyOpenVBA
6
6
  Project-URL: Repository, https://github.com/WilliamSmithEdward/pyOpenVBA
@@ -29,6 +29,7 @@ Classifier: Typing :: Typed
29
29
  Requires-Python: >=3.10
30
30
  Provides-Extra: dev
31
31
  Requires-Dist: build>=1.2; extra == 'dev'
32
+ Requires-Dist: openpyxl>=3.1; extra == 'dev'
32
33
  Requires-Dist: pyright>=1.1.350; extra == 'dev'
33
34
  Requires-Dist: pytest>=8; extra == 'dev'
34
35
  Requires-Dist: twine>=5; extra == 'dev'
@@ -55,7 +56,7 @@ VBA, four hosts and one API:
55
56
  * PowerPoint (`.pptm`, `.potm`, `.ppt`)
56
57
  * Access (`.accdb`, `.mdb`)
57
58
 
58
- Power Query, in any Excel package (`.xlsx` included):
59
+ Power Query, in any Excel package:
59
60
 
60
61
  ```python
61
62
  from pyopenvba import PowerQueryWorkbook
@@ -385,10 +386,15 @@ with AccessDatabase("app.accdb") as db:
385
386
 
386
387
  report = db.add_report("Monthly")
387
388
  report.add_control("Label", "Banner", section="PageHeaderSection", caption="Header band")
389
+ db.rename_form("Draft", "Invoice")
388
390
  db.delete_form("Old")
389
391
  db.save()
390
392
  ```
391
393
 
394
+ Renaming and deleting reach the code behind a design as well as the
395
+ design itself. A form's module is bound to it by name, so `Form_Draft`
396
+ becomes `Form_Invoice` and a deleted form takes its module with it.
397
+
392
398
  [examples/access_form_demo.py](https://github.com/WilliamSmithEdward/pyOpenVBA/blob/main/examples/access_form_demo.py)
393
399
  builds a working order calculator this way: a form whose buttons call a
394
400
  standard module and keep their running total in a class module, laid out
@@ -19,7 +19,7 @@ VBA, four hosts and one API:
19
19
  * PowerPoint (`.pptm`, `.potm`, `.ppt`)
20
20
  * Access (`.accdb`, `.mdb`)
21
21
 
22
- Power Query, in any Excel package (`.xlsx` included):
22
+ Power Query, in any Excel package:
23
23
 
24
24
  ```python
25
25
  from pyopenvba import PowerQueryWorkbook
@@ -349,10 +349,15 @@ with AccessDatabase("app.accdb") as db:
349
349
 
350
350
  report = db.add_report("Monthly")
351
351
  report.add_control("Label", "Banner", section="PageHeaderSection", caption="Header band")
352
+ db.rename_form("Draft", "Invoice")
352
353
  db.delete_form("Old")
353
354
  db.save()
354
355
  ```
355
356
 
357
+ Renaming and deleting reach the code behind a design as well as the
358
+ design itself. A form's module is bound to it by name, so `Form_Draft`
359
+ becomes `Form_Invoice` and a deleted form takes its module with it.
360
+
356
361
  [examples/access_form_demo.py](https://github.com/WilliamSmithEdward/pyOpenVBA/blob/main/examples/access_form_demo.py)
357
362
  builds a working order calculator this way: a form whose buttons call a
358
363
  standard module and keep their running total in a class module, laid out
@@ -277,6 +277,47 @@ this file exists because nothing here is written on a guess.
277
277
 
278
278
  ---
279
279
 
280
+ ## Workbooks another tool wrote
281
+
282
+ Excel is not the only writer of `.xlsx`, and the parts it produces are
283
+ one legal spelling among several. Three assumptions here came from
284
+ reading only Excel's output, and each one broke on openpyxl's:
285
+
286
+ * **Attribute order carries no meaning.** Excel opens a relationship with
287
+ `Id`, openpyxl closes with it. Matching in a fixed order found nothing,
288
+ so a sheet looked as though it had no part behind it and loading to it
289
+ failed outright.
290
+ * **An empty element may be written closed.** `<definedNames />` is the
291
+ same element as `<definedNames></definedNames>`. Appending a second
292
+ block beside it left two in the workbook, which Excel refuses.
293
+ * **A namespace prefix is declared where it is used.** Excel puts
294
+ `xmlns:r` on every worksheet; openpyxl puts it on a worksheet that
295
+ needs one, and a sheet with no table does not. Adding a `tablePart`
296
+ that used the prefix made the part not well formed.
297
+
298
+ Loading a query onto a sheet of a workbook openpyxl wrote works, and
299
+ Excel opens and refreshes the result.
300
+
301
+ **The other direction does not, and cannot be fixed here.** openpyxl
302
+ rebuilds the package from the parts it models and drops the rest, custom
303
+ XML included, so saving a workbook through it removes the Power Query
304
+ package and the queries with it. Nothing signals this: the file opens and
305
+ simply has no queries. Put pyOpenVBA last in the pipeline, or carry the
306
+ queries across with `pull_queries()` and `push_queries()`.
307
+
308
+ ## `[trash]` parts
309
+
310
+ Excel's own file recovery leaves the parts it threw out under
311
+ `[trash]/NNNN.dat`, beside the real ones. An OPC part name cannot open a
312
+ segment with a bracket, and Excel holds itself to that when reading: the
313
+ same workbook opens before such an entry is added and fails to open at
314
+ all after, measured both ways.
315
+
316
+ The container preserves every entry as it arrived, which would hand back
317
+ a file that stays broken, so `save()` drops these and warns. Nothing in
318
+ the document depends on them: they carry no content type and no
319
+ relationship points at them.
320
+
280
321
  ## What Excel refuses
281
322
 
282
323
  * A query name containing a dot. `Queries.Add` rejects it with
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pyOpenVBA"
7
- version = "5.1.3"
7
+ version = "5.2.1"
8
8
  description = "Read and write VBA macros and Excel Power Query in pure Python, no dependencies."
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -61,6 +61,10 @@ dev = [
61
61
  "pyright>=1.1.350",
62
62
  "build>=1.2",
63
63
  "twine>=5",
64
+ # Test-time only, and the runtime keeps its promise of no dependencies.
65
+ # openpyxl writes the same parts a different legal way, which is what
66
+ # the interop tests are for.
67
+ "openpyxl>=3.1",
64
68
  ]
65
69
 
66
70
  [project.urls]
@@ -291,4 +291,4 @@ def push_power_query(
291
291
  """
292
292
  return _push_queries(src_dir, workbook, out=out, encoding=encoding, remove_missing=remove_missing)
293
293
 
294
- __version__ = "5.1.3"
294
+ __version__ = "5.2.1"
@@ -368,6 +368,12 @@ def rename_attribute(text: str, old: str, new: str) -> str:
368
368
 
369
369
 
370
370
  # --- PROJECTwm and PROJECT --------------------------------------------------
371
+ #: How PROJECT names the module behind a form or report: the keyword, the
372
+ #: module's name, a slash, and a flag word Access owns. A design's module
373
+ #: is listed this way and never as ``Module=`` or ``Class=``.
374
+ DOC_CLASS = "DocClass"
375
+ #: The three keywords that open the module block.
376
+ MODULE_KEYWORDS = ("Module=", "Class=", DOC_CLASS + "=")
371
377
 
372
378
 
373
379
  def project_wm_entry(name: str, encoding: str) -> bytes:
@@ -395,34 +401,56 @@ def rename_project_wm(payload: bytes, old: str, new: str, encoding: str) -> byte
395
401
  def add_to_project(text: str, name: str, kind: str) -> str:
396
402
  """Access lists a standard module as ``Module=`` and a class as
397
403
  ``Class=``, both in the same block, and gives each a window rectangle
398
- under ``[Workspace]``."""
404
+ under ``[Workspace]``.
405
+
406
+ The block can be empty -- delete a project's last module and there is
407
+ no line to sit after -- and Access opens it right below the ``ID=``
408
+ line, which is where the first one goes.
409
+ """
399
410
  lines = text.split(CRLF)
400
- last = max(i for i, line in enumerate(lines) if line.startswith(("Module=", "Class=")))
401
- lines.insert(last + 1, ("Class=" if kind == "class" else "Module=") + name)
411
+ entry = ("Class=" if kind == "class" else "Module=") + name
412
+ listed = [i for i, line in enumerate(lines) if line.startswith(MODULE_KEYWORDS)]
413
+ if listed:
414
+ lines.insert(max(listed) + 1, entry)
415
+ else:
416
+ opener = next((i for i, line in enumerate(lines) if line.startswith("ID=")), -1)
417
+ lines.insert(opener + 1, entry)
402
418
  if any(line.strip() == "[Workspace]" for line in lines):
403
419
  lines.insert(len(lines) - 1, f"{name}=38, 38, 1786, 1030, ")
404
420
  return CRLF.join(lines)
405
421
 
406
422
 
407
423
  def remove_from_project(text: str, name: str) -> str:
424
+ """Every line naming a module: its ``Module=``, ``Class=`` or
425
+ ``DocClass=`` line, and its ``[Workspace]`` rectangle."""
408
426
  lines = [
409
427
  line
410
428
  for line in text.split(CRLF)
411
429
  if line not in (f"Module={name}", f"Class={name}")
430
+ and not line.startswith(f"{DOC_CLASS}={name}/")
412
431
  and not re.match(re.escape(name) + "=", line)
413
432
  ]
414
433
  return CRLF.join(lines)
415
434
 
416
435
 
417
436
  def rename_project(text: str, old: str, new: str) -> str:
418
- """The ``Module=``/``Class=`` line and the ``[Workspace]`` line. The
419
- stream's lines end CR LF, so the end anchor has to allow the CR."""
437
+ """The ``Module=``, ``Class=`` or ``DocClass=`` line and the
438
+ ``[Workspace]`` line. The stream's lines end CR LF, so the end anchor
439
+ has to allow the CR.
440
+
441
+ A design's module is listed only as ``DocClass=<name>/<flags>``, never
442
+ as ``Module=`` or ``Class=``. Renaming without reaching it leaves a
443
+ DocClass naming a module the project no longer has, which Access
444
+ reports as a corrupt project on the first VBE reference (GitHub issue
445
+ #21). The flag word after the slash is Access's, so the match stops
446
+ at it and it is carried through.
447
+ """
448
+ quoted = re.escape(old)
420
449
  tail = "(?=" + chr(92) + "r?$)"
421
450
  for keyword in ("Module", "Class"):
422
- text = re.sub(
423
- "(?m)^" + keyword + "=" + re.escape(old) + tail, keyword + "=" + new, text
424
- )
425
- return re.sub("(?m)^" + re.escape(old) + "=", new + "=", text)
451
+ text = re.sub("(?m)^" + keyword + "=" + quoted + tail, keyword + "=" + new, text)
452
+ text = re.sub("(?m)^" + DOC_CLASS + "=" + quoted + "(?=/)", DOC_CLASS + "=" + new, text)
453
+ return re.sub("(?m)^" + quoted + "=", new + "=", text)
426
454
 
427
455
 
428
456
  # --- the folder list ---------------------------------------------------------
@@ -476,22 +504,24 @@ def document_attributes(name: str, clsid: str) -> list[str]:
476
504
  def add_to_project_documents(text: str, name: str) -> str:
477
505
  """A `DocClass=` line, and a window rectangle under `[Workspace]`."""
478
506
  lines = text.split(CRLF)
479
- last = max(
480
- i
481
- for i, line in enumerate(lines)
482
- if line.startswith(("Module=", "Class=", "DocClass="))
483
- )
484
- lines.insert(last + 1, f"DocClass={name}{DOC_CLASS_SUFFIX}")
507
+ last = max(i for i, line in enumerate(lines) if line.startswith(MODULE_KEYWORDS))
508
+ lines.insert(last + 1, f"{DOC_CLASS}={name}{DOC_CLASS_SUFFIX}")
485
509
  if any(line.strip() == "[Workspace]" for line in lines):
486
510
  lines.insert(len(lines) - 1, f"{name}={DOC_WORKSPACE}")
487
511
  return CRLF.join(lines)
488
512
 
489
513
 
490
514
  def remove_from_project_documents(text: str, name: str) -> str:
515
+ """A document module's ``DocClass=`` line and its workspace rectangle.
516
+
517
+ Matched on the prefix, because the flag word after the slash and the
518
+ rectangle are Access's and a database it has edited need not carry the
519
+ ones written here.
520
+ """
491
521
  return CRLF.join(
492
522
  line
493
523
  for line in text.split(CRLF)
494
- if line != f"DocClass={name}{DOC_CLASS_SUFFIX}" and line != f"{name}={DOC_WORKSPACE}"
524
+ if not line.startswith(f"{DOC_CLASS}={name}/") and not line.startswith(f"{name}=")
495
525
  )
496
526
 
497
527
 
@@ -2641,6 +2641,43 @@ class AccessDatabase:
2641
2641
  footer sections."""
2642
2642
  return AccessForm(self, self._create_design("report", name, updated=updated))
2643
2643
 
2644
+ def _list_in_container(
2645
+ self, container: int, name: str, folder: str, when: dt.datetime | float
2646
+ ) -> None:
2647
+ """Name an object in its container's listing and claim its folder
2648
+ in the container's folder list, creating either stream when the
2649
+ container has none yet.
2650
+
2651
+ A container that has never held an object carries neither, so a
2652
+ module added to a project with none went unlisted, and Access
2653
+ showed nothing under `AllModules` for a module its VBE still
2654
+ listed (GitHub issue #21).
2655
+ """
2656
+ storage = self.table(STORAGE_TABLE)
2657
+ adders: tuple[tuple[str, Callable[[bytes], bytes]], ...] = (
2658
+ (DIR_DATA, lambda payload: add_to_dir_data(payload, name, folder)),
2659
+ ("PropData", lambda payload: add_to_folder_list(payload, folder)),
2660
+ )
2661
+ for stream, add in adders:
2662
+ found = next(
2663
+ (
2664
+ (rid, row.get("Lv"))
2665
+ for rid, row in storage.rows_with_ids()
2666
+ if _as_int(row["ParentId"]) == container and str(row["Name"]) == stream
2667
+ ),
2668
+ None,
2669
+ )
2670
+ if found is None:
2671
+ storage.insert_row(
2672
+ {"ParentId": container, "Name": stream, "Type": TYPE_VALUE,
2673
+ "Lv": add(bytes(4)), "DateCreate": when, "DateUpdate": when}
2674
+ )
2675
+ else:
2676
+ rid, payload = found
2677
+ storage.update_row(
2678
+ rid, {"Lv": add(payload if isinstance(payload, bytes) else bytes(4))}
2679
+ )
2680
+
2644
2681
  def _create_design(self, kind: str, name: str, *, updated: object | None) -> AccessDesign:
2645
2682
  """The design itself comes from a captured template -- an empty one
2646
2683
  as Access writes it -- with a GUID of its own patched in, since the
@@ -2686,29 +2723,7 @@ class AccessDatabase:
2686
2723
  values["Lv"] = payload
2687
2724
  storage.insert_row(values)
2688
2725
 
2689
- adders: tuple[tuple[str, Callable[[bytes], bytes]], ...] = (
2690
- (DIR_DATA, lambda payload: add_to_dir_data(payload, name, folder)),
2691
- ("PropData", lambda payload: add_to_folder_list(payload, folder)),
2692
- )
2693
- for stream, add in adders:
2694
- found = next(
2695
- (
2696
- (rid, row.get("Lv"))
2697
- for rid, row in storage.rows_with_ids()
2698
- if _as_int(row["ParentId"]) == container and str(row["Name"]) == stream
2699
- ),
2700
- None,
2701
- )
2702
- if found is None:
2703
- storage.insert_row(
2704
- {"ParentId": container, "Name": stream, "Type": TYPE_VALUE,
2705
- "Lv": add(bytes(4)), "DateCreate": when, "DateUpdate": when}
2706
- )
2707
- else:
2708
- rid, payload = found
2709
- storage.update_row(
2710
- rid, {"Lv": add(payload if isinstance(payload, bytes) else bytes(4))}
2711
- )
2726
+ self._list_in_container(container, name, folder, when)
2712
2727
 
2713
2728
  objects = self.table("MSysObjects")
2714
2729
  parent = next(e.id for e in self.catalog() if e.name == CATALOG_CONTAINERS[kind] and e.type == 3)
@@ -2988,6 +3003,99 @@ class AccessDatabase:
2988
3003
  return self.module(name)
2989
3004
 
2990
3005
 
3006
+ def _delete_module_streams(self, module: VBAModule) -> None:
3007
+ """Take a module out of the VBA project alone: its stream row, its
3008
+ dir block, its PROJECTwm entry, and whichever line of PROJECT
3009
+ names it.
3010
+
3011
+ The Modules container is deliberately untouched, because the code
3012
+ behind a form or report is listed in neither container: the
3013
+ project carries `Form_Calculator` while the Forms listing carries
3014
+ `Calculator`.
3015
+ """
3016
+ storage = self.table(STORAGE_TABLE)
3017
+ encoding = encoding_of(self._vba_dir()[1])
3018
+ _modules, project_id, streams_id = self._vba_storage_ids()
3019
+ for rid, row in list(storage.rows_with_ids()):
3020
+ value = row.get("Lv")
3021
+ row_name, parent = str(row["Name"]), _as_int(row["ParentId"])
3022
+ if parent == streams_id and row_name == module.stream_name:
3023
+ storage.delete_row(rid, retire_empty=False)
3024
+ continue
3025
+ if not isinstance(value, bytes) or not value:
3026
+ continue
3027
+ if row_name == "dir":
3028
+ storage.update_row(
3029
+ rid, {"Lv": compress(remove_from_dir(decompress(value), module.name))}
3030
+ )
3031
+ elif row_name == "_VBA_PROJECT":
3032
+ storage.update_row(rid, {"Lv": invalidate_cache(value)})
3033
+ elif row_name == "PROJECTwm" and parent == project_id:
3034
+ storage.update_row(rid, {"Lv": remove_from_project_wm(value, module.name, encoding)})
3035
+ elif row_name == "PROJECT" and parent == project_id:
3036
+ text = value.decode(encoding, errors="replace")
3037
+ fixed = remove_from_project(text, module.name)
3038
+ if fixed != text:
3039
+ storage.update_row(rid, {"Lv": encode_mbcs(fixed, encoding)})
3040
+ self._drop_srp()
3041
+
3042
+ def _design_module(self, kind: str, name: str) -> VBAModule | None:
3043
+ """The module behind a design, if Access ever opened a code window
3044
+ for it. A design that has none simply has no such module."""
3045
+ wanted = self.DESIGN_MODULE_PREFIX[kind] + name
3046
+ return next(
3047
+ (module for module in self.modules() if module.name.lower() == wanted.lower()),
3048
+ None,
3049
+ )
3050
+
3051
+ def rename_form(self, name: str, new_name: str) -> None:
3052
+ """Rename a form everywhere its name lives."""
3053
+ self._rename_design("form", name, new_name)
3054
+
3055
+ def rename_report(self, name: str, new_name: str) -> None:
3056
+ """Rename a report everywhere its name lives."""
3057
+ self._rename_design("report", name, new_name)
3058
+
3059
+ def _rename_design(self, kind: str, name: str, new_name: str) -> None:
3060
+ """A design's name lives in four places: its container's listing,
3061
+ its catalog row, its navigation-pane row, and the module behind
3062
+ it, which Access binds by name as `Form_<name>` or
3063
+ `Report_<name>`.
3064
+
3065
+ A design Access has never opened a code window for has no module,
3066
+ and then there are three. The module is the part with the sharp
3067
+ edge: `PROJECT` names it only as a `DocClass=` line, and leaving
3068
+ that naming a module the project no longer has makes Access report
3069
+ the whole project as corrupt on the first VBE reference (GitHub
3070
+ issue #21).
3071
+ """
3072
+ if not new_name or len(new_name) > 64:
3073
+ raise AccessError(f"a {kind} name is 1 to 64 characters")
3074
+ found = self._design(kind, name)
3075
+ if new_name.lower() != found.name.lower() and any(
3076
+ other.name.lower() == new_name.lower() for other in self._designs(kind)
3077
+ ):
3078
+ raise AccessError(f"a {kind} named {new_name!r} already exists")
3079
+
3080
+ module = self._design_module(kind, found.name)
3081
+ if module is not None:
3082
+ self._rename_module_streams(module, self.DESIGN_MODULE_PREFIX[kind] + new_name)
3083
+
3084
+ container = self._design_container(kind)
3085
+ storage = self.table(STORAGE_TABLE)
3086
+ for rid, row in list(storage.rows_with_ids()):
3087
+ value = row.get("Lv")
3088
+ if (
3089
+ isinstance(value, bytes)
3090
+ and value
3091
+ and str(row["Name"]) == DIR_DATA
3092
+ and _as_int(row["ParentId"]) == container
3093
+ ):
3094
+ storage.update_row(rid, {"Lv": rename_dir_data(value, found.name, new_name)})
3095
+ self._rename_catalog_rows(found.name, new_name, OBJECT_TYPES[kind])
3096
+ self.forget_catalog()
3097
+ self._drop_srp()
3098
+
2991
3099
  def delete_form(self, name: str) -> None:
2992
3100
  """Remove a form and every structure it occupies."""
2993
3101
  self._delete_design("form", name)
@@ -2998,6 +3106,13 @@ class AccessDatabase:
2998
3106
 
2999
3107
  def _delete_design(self, kind: str, name: str) -> None:
3000
3108
  found = self._design(kind, name)
3109
+ # The module behind it goes too, or the project is left naming a
3110
+ # DocClass whose design is gone, which Access reads as corrupt
3111
+ # (GitHub issue #21). It has to go first, while the design is
3112
+ # still there to name it.
3113
+ module = self._design_module(kind, found.name)
3114
+ if module is not None:
3115
+ self._delete_module_streams(module)
3001
3116
  container = self._design_container(kind)
3002
3117
  storage = self.table(STORAGE_TABLE)
3003
3118
  listing_rid, listing_payload = next(
@@ -4031,12 +4146,6 @@ class AccessDatabase:
4031
4146
  row_name, parent = str(row["Name"]), _as_int(row["ParentId"])
4032
4147
  if row_name == "_VBA_PROJECT":
4033
4148
  storage.update_row(rid, {"Lv": invalidate_cache(payload)})
4034
- elif row_name == "\x03DirData" and parent == modules_id:
4035
- # The four bytes an entry ends with name the object's
4036
- # storage folder, not a terminator.
4037
- storage.update_row(rid, {"Lv": add_to_dir_data(payload, name, folder)})
4038
- elif row_name == "PropData" and parent == modules_id:
4039
- storage.update_row(rid, {"Lv": add_to_folder_list(payload, folder)})
4040
4149
  elif row_name == "PROJECTwm" and parent == project_id:
4041
4150
  storage.update_row(rid, {"Lv": add_to_project_wm(payload, name, encoding)})
4042
4151
  elif row_name == "PROJECT" and parent == project_id:
@@ -4051,6 +4160,9 @@ class AccessDatabase:
4051
4160
  )
4052
4161
  },
4053
4162
  )
4163
+ # The four bytes an entry ends with name the object's storage
4164
+ # folder, not a terminator.
4165
+ self._list_in_container(modules_id, name, folder, when)
4054
4166
  storage.update_row(
4055
4167
  dir_rid,
4056
4168
  {"Lv": compress(add_to_dir(dir_stream, dir_block(name, stream_name, cookie, kind, encoding)))},
@@ -4098,16 +4210,16 @@ class AccessDatabase:
4098
4210
  self._invalidate_vba_cache()
4099
4211
  self._drop_srp()
4100
4212
 
4101
- def rename_module(self, name: str, new_name: str) -> None:
4102
- """Rename a module in all eight places its name lives."""
4103
- if not new_name or len(new_name) > 64:
4104
- raise AccessError("a module name is 1 to 64 characters")
4105
- module = self.module(name)
4106
- if new_name.lower() != name.lower() and any(
4107
- other.name.lower() == new_name.lower() for other in self.modules()
4108
- ):
4109
- raise AccessError(f"a module named {new_name!r} already exists")
4213
+ def _rename_module_streams(self, module: VBAModule, new_name: str) -> None:
4214
+ """Rename a module wherever the VBA project itself names it: the
4215
+ two dir records, the module's own `Attribute VB_Name`, PROJECT and
4216
+ PROJECTwm.
4110
4217
 
4218
+ The Modules listing is deliberately left alone here, because the
4219
+ code behind a form or report appears in neither container listing:
4220
+ the project carries `Form_Calculator` while the Forms listing
4221
+ carries `Calculator`.
4222
+ """
4111
4223
  storage = self.table(STORAGE_TABLE)
4112
4224
  dir_rid, dir_stream = self._vba_dir()
4113
4225
  encoding = encoding_of(dir_stream)
@@ -4120,7 +4232,7 @@ class AccessDatabase:
4120
4232
  stream = set_module_offset(rename_in_dir(dir_stream, module.name, new_name), new_name, 0)
4121
4233
  storage.update_row(dir_rid, {"Lv": compress(stream)})
4122
4234
 
4123
- modules_id, project_id, _streams = self._vba_storage_ids()
4235
+ _modules, project_id, _streams = self._vba_storage_ids()
4124
4236
  for row_rid, row in list(storage.rows_with_ids()):
4125
4237
  value = row.get("Lv")
4126
4238
  if not isinstance(value, bytes) or not value:
@@ -4128,8 +4240,6 @@ class AccessDatabase:
4128
4240
  row_name, parent = str(row["Name"]), _as_int(row["ParentId"])
4129
4241
  if row_name == "_VBA_PROJECT":
4130
4242
  storage.update_row(row_rid, {"Lv": invalidate_cache(value)})
4131
- elif row_name == "\x03DirData" and parent == modules_id:
4132
- storage.update_row(row_rid, {"Lv": rename_dir_data(value, module.name, new_name)})
4133
4243
  elif row_name == "PROJECTwm" and parent == project_id:
4134
4244
  storage.update_row(
4135
4245
  row_rid, {"Lv": rename_project_wm(value, module.name, new_name, encoding)}
@@ -4140,16 +4250,42 @@ class AccessDatabase:
4140
4250
  if fixed != text:
4141
4251
  storage.update_row(row_rid, {"Lv": encode_mbcs(fixed, encoding)})
4142
4252
 
4253
+ def _rename_catalog_rows(self, old: str, new: str, object_type: int) -> None:
4254
+ """The object's catalog row and its navigation-pane row."""
4143
4255
  objects = self.table("MSysObjects")
4144
4256
  for row_rid, row in objects.rows_with_ids():
4145
- if row["Type"] == OBJECT_MODULE and str(row["Name"]) == module.name:
4146
- objects.update_row(row_rid, {"Name": new_name})
4257
+ if row["Type"] == object_type and str(row["Name"]) == old:
4258
+ objects.update_row(row_rid, {"Name": new})
4147
4259
  break
4148
4260
  nav = self.table("MSysNavPaneObjectIDs")
4149
4261
  for row_rid, row in nav.rows_with_ids():
4150
- if str(row["Name"]) == module.name:
4151
- nav.update_row(row_rid, {"Name": new_name})
4262
+ if str(row["Name"]) == old:
4263
+ nav.update_row(row_rid, {"Name": new})
4152
4264
  break
4265
+
4266
+ def rename_module(self, name: str, new_name: str) -> None:
4267
+ """Rename a module in all eight places its name lives."""
4268
+ if not new_name or len(new_name) > 64:
4269
+ raise AccessError("a module name is 1 to 64 characters")
4270
+ module = self.module(name)
4271
+ if new_name.lower() != name.lower() and any(
4272
+ other.name.lower() == new_name.lower() for other in self.modules()
4273
+ ):
4274
+ raise AccessError(f"a module named {new_name!r} already exists")
4275
+
4276
+ self._rename_module_streams(module, new_name)
4277
+ storage = self.table(STORAGE_TABLE)
4278
+ modules_id = self._vba_storage_ids()[0]
4279
+ for row_rid, row in list(storage.rows_with_ids()):
4280
+ value = row.get("Lv")
4281
+ if (
4282
+ isinstance(value, bytes)
4283
+ and value
4284
+ and str(row["Name"]) == DIR_DATA
4285
+ and _as_int(row["ParentId"]) == modules_id
4286
+ ):
4287
+ storage.update_row(row_rid, {"Lv": rename_dir_data(value, module.name, new_name)})
4288
+ self._rename_catalog_rows(module.name, new_name, OBJECT_MODULE)
4153
4289
  self._drop_srp()
4154
4290
  self._catalog = None
4155
4291
 
@@ -29,6 +29,9 @@ _STORED = 0
29
29
  _FLAGS = 0x0006
30
30
  _MADE_BY = 45
31
31
  _NEEDED = 20
32
+ #: Where Excel's file recovery parks the parts it threw out. Not a legal
33
+ #: OPC part name, and Excel will not open a package holding one.
34
+ _RESERVED = "[trash]/"
32
35
 
33
36
 
34
37
  @dataclass
@@ -174,6 +177,28 @@ class OpcFile:
174
177
  self.entries = kept
175
178
  self.source = None
176
179
 
180
+ def drop_reserved(self) -> list[str]:
181
+ """Take out entries whose names are not part names, and say which.
182
+
183
+ Excel's own file recovery leaves `[trash]/NNNN.dat` beside the
184
+ real parts. An OPC part name is built of segments that cannot
185
+ open with a bracket, and Excel holds itself to that when reading:
186
+ a workbook that opens cleanly stops opening at all once such an
187
+ entry is added, which is measured both ways in
188
+ `tests/test_powerquery_opc.py`.
189
+
190
+ Nothing in the document depends on them. They carry no content
191
+ type and no relationship points at them, so a package is repaired
192
+ by dropping them rather than damaged.
193
+ """
194
+ doomed = [entry.name for entry in self.entries if entry.name.startswith(_RESERVED)]
195
+ if doomed:
196
+ self.entries = [
197
+ entry for entry in self.entries if not entry.name.startswith(_RESERVED)
198
+ ]
199
+ self.source = None
200
+ return doomed
201
+
177
202
  def serialize(self) -> bytes:
178
203
  if self.source is not None:
179
204
  return self.source
@@ -102,6 +102,38 @@ def _relationships(package: OpcFile, part: str) -> str:
102
102
  )
103
103
 
104
104
 
105
+ def _attributes(element: str) -> dict[str, str]:
106
+ """The attributes of one element, in whatever order they were written.
107
+
108
+ XML gives attribute order no meaning, and writers differ: Excel opens
109
+ a relationship with ``Id``, openpyxl closes with it. Matching them in
110
+ a fixed order silently found nothing in the second case, which left a
111
+ sheet looking as though it had no part behind it.
112
+ """
113
+ return dict(re.findall(r'([\w.:-]+)\s*=\s*"([^"]*)"', element))
114
+
115
+
116
+ def _relationship_targets(rels: str) -> dict[str, str]:
117
+ """``{relationship id: target}`` for one ``.rels`` part."""
118
+ out: dict[str, str] = {}
119
+ for element in re.findall(r"<Relationship\b[^>]*>", rels):
120
+ attributes = _attributes(element)
121
+ identifier, target = attributes.get("Id"), attributes.get("Target")
122
+ if identifier and target is not None:
123
+ out[identifier] = target
124
+ return out
125
+
126
+
127
+ def _relationship_for(rels: str, target: str) -> str | None:
128
+ """The id pointing at ``target``, comparing the part it names rather
129
+ than the spelling: a target may be relative or absolute."""
130
+ wanted = target.lstrip("/").removeprefix("../")
131
+ for identifier, found in _relationship_targets(rels).items():
132
+ if found.lstrip("/").removeprefix("../") == wanted:
133
+ return identifier
134
+ return None
135
+
136
+
105
137
  def add_relationship(package: OpcFile, part: str, kind: str, target: str) -> str:
106
138
  raw = _relationships(package, part)
107
139
  identifier = _next_relationship(raw)
@@ -142,8 +174,17 @@ def sheet_part(package: OpcFile, sheet: str | int) -> tuple[str, str]:
142
174
  name of the sheet."""
143
175
  workbook = package.read(_WORKBOOK).decode("utf-8")
144
176
  rels = package.read(_WORKBOOK_RELS).decode("utf-8")
145
- targets = dict(re.findall(r'<Relationship Id="([^"]+)"[^>]*Target="([^"]+)"', rels))
146
- sheets = re.findall(r'<sheet\b[^>]*name="([^"]*)"[^>]*r:id="([^"]+)"', workbook)
177
+ targets = _relationship_targets(rels)
178
+ sheets: list[tuple[str, str]] = []
179
+ for element in re.findall(r"<sheet\b[^>]*>", workbook):
180
+ attributes = _attributes(element)
181
+ # The relationship attribute carries a namespace prefix, and which
182
+ # prefix is the writer's to choose.
183
+ identifier = next(
184
+ (value for key, value in attributes.items() if key.endswith(":id")), None
185
+ )
186
+ if "name" in attributes and identifier:
187
+ sheets.append((attributes["name"], identifier))
147
188
  if not sheets:
148
189
  raise PowerQueryError("this workbook lists no worksheets")
149
190
  if isinstance(sheet, int):
@@ -306,6 +347,7 @@ def _add_table_to_sheet(
306
347
  for index, column in enumerate(columns)
307
348
  )
308
349
  raw = _write_header(raw, header, start, len(columns))
350
+ raw = _declare_relationship_namespace(raw)
309
351
  if "<tableParts" in raw:
310
352
  raw = re.sub(
311
353
  r'<tableParts count="(\d+)">',
@@ -321,6 +363,28 @@ def _add_table_to_sheet(
321
363
  package.write(part, raw.encode("utf-8"))
322
364
 
323
365
 
366
+ #: The namespace a `<tablePart r:id=...>` reference lives in.
367
+ _RELATIONSHIPS_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
368
+
369
+
370
+ def _declare_relationship_namespace(sheet: str) -> str:
371
+ """Make sure the worksheet element declares the `r:` prefix.
372
+
373
+ Excel always writes it on the root element, so the reference added
374
+ below resolved. openpyxl writes the prefix only where it uses one,
375
+ and a worksheet with no table has no use for it, which left the
376
+ `r:id` added here pointing at a prefix nothing declared. That is not
377
+ well-formed XML, and Excel would not open the workbook at all.
378
+ """
379
+ opening = re.search(r"<worksheet\b[^>]*>", sheet)
380
+ if opening is None or 'xmlns:r=' in opening.group(0):
381
+ return sheet
382
+ fixed = opening.group(0).replace(
383
+ "<worksheet", f'<worksheet xmlns:r="{_RELATIONSHIPS_NS}"', 1
384
+ )
385
+ return sheet.replace(opening.group(0), fixed, 1)
386
+
387
+
324
388
  def _write_header(sheet: str, cells: str, start: CellRef, width: int) -> str:
325
389
  """Put the header cells on the sheet.
326
390
 
@@ -377,8 +441,14 @@ def _add_defined_name(package: OpcFile, sheet_name: str, number: int, reference:
377
441
  f'<definedName name="ExternalData_{number}" localSheetId="0" hidden="1">'
378
442
  f"{quoted}!{absolute}</definedName>"
379
443
  )
444
+ empty = re.search(r"<definedNames\s*/>", raw)
380
445
  if "<definedNames>" in raw:
381
446
  raw = raw.replace("<definedNames>", "<definedNames>" + defined)
447
+ elif empty is not None:
448
+ # An empty element written closed, which openpyxl does and Excel
449
+ # does not. Appending a second block beside it puts two in the
450
+ # workbook, and Excel refuses that outright.
451
+ raw = raw.replace(empty.group(0), f"<definedNames>{defined}</definedNames>", 1)
382
452
  elif "<calcPr" in raw:
383
453
  raw = raw.replace("<calcPr", f"<definedNames>{defined}</definedNames><calcPr", 1)
384
454
  else:
@@ -407,16 +477,14 @@ def unload_from_sheet(package: OpcFile, query: str) -> bool:
407
477
  if not package.has(rels_part):
408
478
  continue
409
479
  rels = package.read(rels_part).decode("utf-8")
410
- found = re.search(
411
- rf'<Relationship Id="([^"]+)"[^>]*Target="\.\./tables/table{number}\.xml"[^>]*/>', rels
412
- )
413
- if found is None:
480
+ identifier = _relationship_for(rels, f"../tables/table{number}.xml")
481
+ if identifier is None:
414
482
  continue
415
483
  drop_relationship(package, rels_part, f"../tables/table{number}.xml")
416
484
  if "<Relationship " not in package.read(rels_part).decode("utf-8"):
417
485
  package.remove(rels_part)
418
486
  sheet = package.read(part).decode("utf-8")
419
- sheet = re.sub(rf'<tablePart r:id="{found.group(1)}"\s*/>', "", sheet)
487
+ sheet = re.sub(rf'<tablePart\b[^>]*:id="{re.escape(identifier)}"[^>]*/>', "", sheet)
420
488
  sheet = re.sub(
421
489
  r'<tableParts count="(\d+)">',
422
490
  lambda match: f'<tableParts count="{max(int(match.group(1)) - 1, 0)}">',
@@ -25,6 +25,7 @@ from __future__ import annotations
25
25
  import base64
26
26
  import re
27
27
  import uuid
28
+ import warnings
28
29
  from pathlib import Path
29
30
  from typing import ClassVar
30
31
  from xml.etree import ElementTree
@@ -564,6 +565,20 @@ class PowerQueryWorkbook:
564
565
 
565
566
  def save(self, path: str | Path | None = None) -> Path:
566
567
  out = Path(path) if path is not None else self.path
568
+ # A workbook Excel has recovered carries the parts it threw out
569
+ # under `[trash]`, and Excel will not open a package holding one.
570
+ # Preserving those faithfully would hand back a file that stays
571
+ # broken, so they go, and the caller is told.
572
+ dropped = self._opc.drop_reserved()
573
+ if dropped:
574
+ warnings.warn(
575
+ "dropped "
576
+ + ", ".join(repr(name) for name in dropped)
577
+ + ": Excel's file recovery leaves those behind and will not "
578
+ "open a workbook that carries them",
579
+ UserWarning,
580
+ stacklevel=2,
581
+ )
567
582
  raw = self.to_bytes()
568
583
  out.write_bytes(raw)
569
584
  if path is None:
File without changes
File without changes
File without changes