pymppwriter 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,12 @@
1
+ from .writer import (MppWriter, Project, Task, Relation, Resource, Assignment,
2
+ Calendar, CalendarException, ScheduleWarning, validate)
3
+ from .reader import read_project, MppReadError
4
+
5
+ try: # the installed distribution's version
6
+ from importlib.metadata import PackageNotFoundError, version as _version
7
+ __version__ = _version("pymppwriter")
8
+ except (ImportError, PackageNotFoundError): # running from a source tree
9
+ __version__ = "0.0.0.dev0"
10
+ __all__ = ["MppWriter", "Project", "Task", "Relation", "Resource", "Assignment",
11
+ "Calendar", "CalendarException", "ScheduleWarning", "validate",
12
+ "read_project", "MppReadError", "__version__"]
pymppwriter/blocks.py ADDED
@@ -0,0 +1,363 @@
1
+ """Decoders/encoders for the block structures inside an MPP14 file.
2
+
3
+ Layouts derived from the public behaviour of the LGPL MPXJ reader and from
4
+ diffing files saved by Microsoft Project.
5
+ """
6
+ from __future__ import annotations
7
+ import struct
8
+ from dataclasses import dataclass
9
+ from datetime import datetime, timedelta
10
+ from typing import Dict, List, Optional, Tuple
11
+
12
+ MAGIC = 0xFADFADBA
13
+ EPOCH = datetime(1983, 12, 31)
14
+
15
+ PROPS_TASK_FIELD_MAP = 131092
16
+ PROPS_RESOURCE_FIELD_MAP = 131093
17
+ PROPS_RELATION_FIELD_MAP = 131094
18
+ PROPS_ASSIGNMENT_FIELD_MAP = 131095
19
+ PROPS_PROJECT_START_DATE = 37748738
20
+ PROPS_PROJECT_FINISH_DATE = 37748739 # 0x2400003
21
+ PROPS_TITLE = 37748744
22
+ PROPS_DEFAULT_CALENDAR_NAME = 37748750 # UTF-16 name + 4 NUL bytes
23
+ PROPS_CURRENCY_SYMBOL = 37748752 # 0x2400010
24
+ PROPS_STATUS_DATE = 37748805 # 0x2400045; 0xFFFFFFFF = NA
25
+ PROPS_CURRENCY_CODE = 37753787 # 0x24013BB, e.g. "USD"
26
+ PROPS_LEGACY_NEXT_UIDS = 37748910 # 0x24000AE: 2010-era only; stale values make
27
+ # Project renumber task uids, M365 drops it on save
28
+ PROPS_EDITED_BASE_CALENDARS = 8388609 # 0x800001: base calendars with custom data
29
+ # 0x10001..: Var2Data byte length per storage — Project truncates its var-data
30
+ # read at the declared length, so a stale value hides var entries
31
+ PROPS_VAR2DATA_SIZE = {"TBkndTask": 65537, "TBkndRsc": 65538, "TBkndCal": 65539,
32
+ "TBkndAssn": 65540}
33
+ # record-count dwords: Project sizes its tables from these on load and drops
34
+ # records beyond the count (verified against four Project-written files)
35
+ PROPS_TASK_RECORD_COUNT = 16777217 # 0x1000001, includes stubs + uid-0 summary
36
+ PROPS_RESOURCE_RECORD_COUNT = 16777218 # 0x1000002
37
+ PROPS_ASSN_RECORD_COUNT = 16777220 # 0x1000004
38
+ PROPS_REL_RECORD_COUNT = 16777221 # 0x1000005
39
+
40
+
41
+ # ---------------------------------------------------------------- Props ----
42
+ PROPS_TYPES: Dict[int, int] = {} # key -> type code (third dword of each entry; 0/2/4/9 observed)
43
+
44
+
45
+ def parse_props(data: bytes) -> Tuple[bytes, Dict[int, bytes], List[int]]:
46
+ """Return (16-byte header, {key: value}, key order). Entry type codes are kept in PROPS_TYPES."""
47
+ header = data[:16]
48
+ count = struct.unpack_from("<H", header, 12)[0]
49
+ pos, out, order = 16, {}, []
50
+ for _ in range(count):
51
+ if len(data) - pos < 12:
52
+ break
53
+ size, key, ptype = struct.unpack_from("<III", data, pos)
54
+ pos += 12
55
+ val = data[pos:pos + size]
56
+ pos += size + (size & 1) # 2-byte alignment
57
+ out[key] = val
58
+ PROPS_TYPES[key] = ptype
59
+ order.append(key)
60
+ return header, out, order
61
+
62
+
63
+ def build_props(header: bytes, values: Dict[int, bytes], order: List[int]) -> bytes:
64
+ hdr = bytearray(header)
65
+ struct.pack_into("<H", hdr, 12, len(order))
66
+ body = bytearray()
67
+ for key in order:
68
+ val = values[key]
69
+ body += struct.pack("<III", len(val), key, PROPS_TYPES.get(key, 0)) + val
70
+ if len(val) & 1:
71
+ body += b"\0"
72
+ total = len(hdr) + len(body)
73
+ struct.pack_into("<II", hdr, 0, total - 4, total - 4) # header dwords 0,1 = stream size - 4
74
+ return bytes(hdr) + bytes(body)
75
+
76
+
77
+ # ------------------------------------------------------------- FieldMap ----
78
+ @dataclass
79
+ class FieldItem:
80
+ type_value: int # MS Project field ID (TaskField/ResourceField numeric)
81
+ block: int # 0 = FixedData, 1 = Fixed2Data
82
+ offset: int # byte offset within the fixed record (65535 = not fixed)
83
+ var_key: int # key in Var2Data when stored as variable data
84
+ category: int
85
+ mask: int
86
+ raw: bytes
87
+
88
+ @property
89
+ def in_fixed(self) -> bool:
90
+ return self.category not in (0x0B, 0x64) and self.offset != 65535
91
+
92
+ @property
93
+ def in_meta(self) -> bool:
94
+ return self.category in (0x0B, 0x64)
95
+
96
+
97
+ def parse_field_map(data: bytes) -> List[FieldItem]:
98
+ items, last, block = [], 0, 0
99
+ for i in range(0, len(data) - 27, 28):
100
+ mask = struct.unpack_from("<I", data, i)[0]
101
+ offset = struct.unpack_from("<H", data, i + 4)[0]
102
+ var_key = data[i + 6]
103
+ type_value = struct.unpack_from("<I", data, i + 12)[0]
104
+ category = struct.unpack_from("<H", data, i + 20)[0]
105
+ if category not in (0x0B, 0x64) and offset != 65535:
106
+ if offset < last:
107
+ block += 1
108
+ last = offset
109
+ items.append(FieldItem(type_value, block, offset, var_key, category, mask, data[i:i + 28]))
110
+ return items
111
+
112
+
113
+ # -------------------------------------------------------- Fixed blocks -----
114
+ def parse_fixed_meta(data: bytes, item_size: int) -> Tuple[bytes, int, List[bytes]]:
115
+ magic, unk, count, unk2 = struct.unpack_from("<IIII", data, 0)
116
+ assert magic == MAGIC, hex(magic)
117
+ n = (len(data) - 16) // item_size
118
+ items = [data[16 + i * item_size:16 + (i + 1) * item_size] for i in range(n)]
119
+ return data[:16], count, items
120
+
121
+
122
+ def parse_fixed_meta_auto(data: bytes, default_size: int) -> Tuple[bytes, int, List[bytes]]:
123
+ """parse_fixed_meta with the item size derived from the header count, so
124
+ files of any Project vintage parse (M365 uses 96/51/10-byte Fixed2Meta
125
+ items where 2010-era files use 92/50/9). Falls back to default_size when
126
+ the stream length is not an exact multiple (trailing slack)."""
127
+ count = struct.unpack_from("<I", data, 8)[0]
128
+ size = default_size
129
+ if count and (len(data) - 16) % count == 0:
130
+ size = (len(data) - 16) // count
131
+ return parse_fixed_meta(data, size)
132
+
133
+
134
+ def build_fixed_meta(header: bytes, items: List[bytes], data_len: Optional[int] = None) -> bytes:
135
+ hdr = bytearray(header)
136
+ struct.pack_into("<I", hdr, 8, len(items))
137
+ if data_len is not None:
138
+ struct.pack_into("<I", hdr, 12, data_len) # header dword 3 = FixedData byte length
139
+ return bytes(hdr) + b"".join(items)
140
+
141
+
142
+ def split_fixed_data(data: bytes, meta_items: List[bytes]) -> List[bytes]:
143
+ out = []
144
+ for i, m in enumerate(meta_items):
145
+ off = struct.unpack_from("<I", m, 4)[0]
146
+ if i + 1 < len(meta_items):
147
+ nxt = struct.unpack_from("<I", meta_items[i + 1], 4)[0]
148
+ else:
149
+ nxt = len(data)
150
+ out.append(data[off:nxt])
151
+ return out
152
+
153
+
154
+ # ------------------------------------------------------- meta bitmaps ------
155
+ # A FixedMeta / Fixed2Meta item is: uint32 flags, uint32 offset-in-FixedData,
156
+ # then a bitmap with one bit per TASK_FIELD_MAP entry (little-endian bit order).
157
+ # FixedMeta carries entries 0..(item_size-8)*8-1; Fixed2Meta continues from there.
158
+ # Boolean fields (category 0x0B/0x64) store their value in their entry's bit;
159
+ # for other fields the bit marks the field as populated.
160
+
161
+ def meta_bit(meta: bytes, meta2: bytes, entry_index: int) -> Optional[int]:
162
+ nbits0 = (len(meta) - 8) * 8
163
+ buf, i = (meta, entry_index) if entry_index < nbits0 else (meta2, entry_index - nbits0)
164
+ byte = 8 + i // 8
165
+ if byte >= len(buf):
166
+ return None
167
+ return (buf[byte] >> (i % 8)) & 1
168
+
169
+
170
+ def set_meta_bit(meta: bytearray, meta2: bytearray, entry_index: int, value: bool) -> None:
171
+ nbits0 = (len(meta) - 8) * 8
172
+ buf, i = (meta, entry_index) if entry_index < nbits0 else (meta2, entry_index - nbits0)
173
+ byte = 8 + i // 8
174
+ if byte >= len(buf):
175
+ return
176
+ if value:
177
+ buf[byte] |= 1 << (i % 8)
178
+ else:
179
+ buf[byte] &= ~(1 << (i % 8)) & 0xFF
180
+
181
+
182
+ # ---------------------------------------------------------- Var blocks -----
183
+ def parse_var_meta(data: bytes) -> Tuple[bytes, Dict[int, Dict[int, int]], List[Tuple[int, int, int, int]]]:
184
+ """VarMeta12: 24-byte header then 12-byte entries (uid, offset, type, unk)."""
185
+ magic, unk, count, unk2, unk3, data_size = struct.unpack_from("<IIIIII", data, 0)
186
+ table: Dict[int, Dict[int, int]] = {}
187
+ entries = []
188
+ pos = 24
189
+ for _ in range(count):
190
+ if len(data) - pos < 12:
191
+ break
192
+ uid, off, typ, unk4 = struct.unpack_from("<IIHH", data, pos)
193
+ pos += 12
194
+ table.setdefault(uid, {})[typ] = off
195
+ entries.append((uid, off, typ, unk4))
196
+ return data[:24], table, entries
197
+
198
+
199
+ def read_var(data: bytes, off: int) -> bytes:
200
+ size = struct.unpack_from("<I", data, off)[0]
201
+ return data[off + 4:off + 4 + size]
202
+
203
+
204
+ TASK_FIELD_HI, RESOURCE_FIELD_HI, ASSIGNMENT_FIELD_HI = 0x0B40, 0x0C40, 0x0F40
205
+
206
+
207
+ def build_var_blocks(header: bytes, values: List[Tuple[int, int, bytes]], field_hi: int = TASK_FIELD_HI) -> Tuple[bytes, bytes]:
208
+ """values: [(uid, type, payload)] -> (VarMeta, Var2Data).
209
+
210
+ Each entry is (uid, offset, fieldId low16, fieldId high16); the high word is the
211
+ native field-class prefix (0x0B40 tasks, 0x0C40 resources, 0x0F40 assignments).
212
+ Entries must be ordered by (uid, fieldId)."""
213
+ meta = bytearray(header)
214
+ var = bytearray()
215
+ entries = bytearray()
216
+ for uid, typ, payload in sorted(values, key=lambda v: (v[0], v[1])):
217
+ entries += struct.pack("<IIHH", uid, len(var), typ, field_hi)
218
+ var += struct.pack("<I", len(payload)) + payload
219
+ struct.pack_into("<I", meta, 8, len(values))
220
+ struct.pack_into("<I", meta, 20, len(var))
221
+ return bytes(meta) + bytes(entries), bytes(var)
222
+
223
+
224
+ # ------------------------------------- OLE property sets (MS-OLEPS) --------
225
+ VT_LPSTR = 30
226
+
227
+
228
+ def update_property_set_strings(data: bytes, updates: Dict[int, str], section: int = 0,
229
+ codepage: str = "cp1252") -> bytes:
230
+ """Replace or add VT_LPSTR properties in one section of an OLE property set
231
+ stream (SummaryInformation / DocumentSummaryInformation), keeping every
232
+ other property byte-for-byte."""
233
+ nsec = struct.unpack_from("<I", data, 24)[0]
234
+ header = data[:28]
235
+ secs = []
236
+ for s in range(nsec):
237
+ fmtid = data[28 + s * 20:44 + s * 20]
238
+ off = struct.unpack_from("<I", data, 44 + s * 20)[0]
239
+ size, cnt = struct.unpack_from("<II", data, off)
240
+ entries = [struct.unpack_from("<II", data, off + 8 + i * 8) for i in range(cnt)]
241
+ bounds = sorted(e[1] for e in entries) + [size]
242
+ raw, order = {}, []
243
+ for pid, poff in entries:
244
+ nxt = min(b for b in bounds if b > poff)
245
+ raw[pid] = data[off + poff:off + nxt]
246
+ order.append(pid)
247
+ secs.append([fmtid, order, raw])
248
+ fmtid, order, raw = secs[section]
249
+ for pid, val in updates.items():
250
+ b = val.encode(codepage, "replace") + b"\0"
251
+ v = struct.pack("<II", VT_LPSTR, len(b)) + b
252
+ raw[pid] = v + b"\0" * ((-len(v)) % 4)
253
+ if pid not in order:
254
+ order.append(pid)
255
+ out_secs = []
256
+ for fm, ord_, rw in secs:
257
+ base = 8 + len(ord_) * 8
258
+ body, entries = bytearray(), []
259
+ for pid in ord_:
260
+ v = rw[pid] + b"\0" * ((-len(rw[pid])) % 4)
261
+ entries.append((pid, base + len(body)))
262
+ body += v
263
+ sec = struct.pack("<II", base + len(body), len(ord_))
264
+ for pid, poff in entries:
265
+ sec += struct.pack("<II", pid, poff)
266
+ out_secs.append((fm, sec + bytes(body)))
267
+ dir_, pos = bytearray(), 28 + len(out_secs) * 20
268
+ for fm, sec in out_secs:
269
+ dir_ += fm + struct.pack("<I", pos)
270
+ pos += len(sec)
271
+ return bytes(header) + bytes(dir_) + b"".join(sec for _, sec in out_secs)
272
+
273
+
274
+ # ------------------------------------------------------ calendar data ------
275
+ CAL_DAY_NONWORKING, CAL_DAY_DEFAULT, CAL_DAY_WORKING = 0, 1, 2
276
+
277
+
278
+ def build_calendar_data(days, exceptions=()) -> bytes:
279
+ """Calendar definition blob (var-data key 8 on a TBkndCal record), in the
280
+ dialect Microsoft Project M365 writes (an earlier form using day type 2
281
+ for working days is readable by MPXJ but ignored by Project).
282
+
283
+ days: 7 (day_type, ranges) tuples, SUNDAY first; ranges are
284
+ (start_minute, end_minute) pairs from midnight, at most 5 per day and only
285
+ meaningful for CAL_DAY_WORKING.
286
+ exceptions: (from_date, to_date, name) tuples of non-working days, sorted.
287
+
288
+ Each 60-byte day block: uint16 day type on the wire (1 = default,
289
+ 0 = explicit; working vs non-working is the range count), uint16 range
290
+ count, uint32 total working tenths-of-a-minute at +4, range start times as
291
+ uint16 tenths at +8 (stride 2), range durations as uint32 tenths at +20
292
+ (stride 4) and duplicated at +40. Exceptions: uint32 count, then per
293
+ exception a 92-byte record (uint16 from-day, uint16 to-day, uint16 day
294
+ count, recurrence dwords 1, 0, 1, 0x4000 at +72, uint32 name byte length
295
+ at +88) followed by the UTF-16 name, zero-padded to a 4-byte boundary plus
296
+ 4 more zero bytes (both as observed in Project-written files).
297
+ """
298
+ if len(days) != 7:
299
+ raise ValueError("days must have exactly 7 entries, Sunday first")
300
+ out = bytearray()
301
+ for dtype, ranges in days:
302
+ b = bytearray(60)
303
+ struct.pack_into("<H", b, 0, 1 if dtype == CAL_DAY_DEFAULT else 0)
304
+ if dtype == CAL_DAY_WORKING:
305
+ ranges = list(ranges)[:5]
306
+ struct.pack_into("<H", b, 2, len(ranges))
307
+ struct.pack_into("<I", b, 4, sum(end - start for start, end in ranges) * 10)
308
+ cumulative = 0
309
+ for i, (start, end) in enumerate(ranges):
310
+ struct.pack_into("<H", b, 8 + 2 * i, start * 10)
311
+ struct.pack_into("<I", b, 20 + 4 * i, (end - start) * 10)
312
+ cumulative += (end - start) * 10
313
+ struct.pack_into("<I", b, 40 + 4 * i, cumulative)
314
+ out += b
315
+ if exceptions:
316
+ out += struct.pack("<I", len(exceptions))
317
+ for i, (from_date, to_date, name) in enumerate(exceptions):
318
+ rec = bytearray(92)
319
+ d1 = (from_date - EPOCH.date()).days
320
+ d2 = (to_date - EPOCH.date()).days
321
+ struct.pack_into("<HH", rec, 0, d1, d2)
322
+ struct.pack_into("<H", rec, 4, d2 - d1 + 1)
323
+ struct.pack_into("<I", rec, 72, 1)
324
+ struct.pack_into("<I", rec, 80, 1)
325
+ struct.pack_into("<I", rec, 84, 0x4000)
326
+ nb = (name + "\0").encode("utf-16-le")
327
+ struct.pack_into("<I", rec, 88, len(nb))
328
+ # next record 4-byte aligned; 4 extra zero bytes close the blob
329
+ out += rec + nb + b"\0" * ((-len(nb)) % 4)
330
+ if i == len(exceptions) - 1:
331
+ out += b"\0\0\0\0"
332
+ return bytes(out)
333
+
334
+
335
+ # ------------------------------------------------------- primitives --------
336
+ def decode_timestamp(b: bytes, off: int) -> Optional[datetime]:
337
+ time, days = struct.unpack_from("<HH", b, off)
338
+ if days <= 1 or days == 65535:
339
+ return None
340
+ if time == 65535:
341
+ time = 0
342
+ return EPOCH + timedelta(days=days, seconds=time * 6)
343
+
344
+
345
+ def encode_timestamp(dt: Optional[datetime]) -> bytes:
346
+ if dt is None:
347
+ return b"\xff\xff\xff\xff"
348
+ delta = dt - EPOCH
349
+ days = delta.days
350
+ tenths = delta.seconds // 6
351
+ return struct.pack("<HH", tenths, days)
352
+
353
+
354
+ def encode_unicode(s: str) -> bytes:
355
+ return s.encode("utf-16-le") + b"\0\0"
356
+
357
+
358
+ def decode_unicode(b: bytes) -> str:
359
+ return b.decode("utf-16-le", errors="replace").split("\0", 1)[0]
360
+
361
+
362
+ # duration units (MS Project codes) -> tenths-of-a-minute divisor
363
+ DURATION_UNITS = {3: ("m", 10), 5: ("h", 600), 7: ("d", 4800), 9: ("w", 24000), 11: ("mo", 96000)}
pymppwriter/cfb.py ADDED
@@ -0,0 +1,248 @@
1
+ """Minimal writer for Microsoft Compound File Binary (OLE2) containers.
2
+
3
+ Implements enough of [MS-CFB] v3 (512-byte sectors, 64-byte mini sectors,
4
+ 4096-byte mini-stream cutoff) to produce files that Apache POI / olefile /
5
+ MS Project can read. Written from the public [MS-CFB] specification.
6
+ """
7
+ from __future__ import annotations
8
+ import struct
9
+ from dataclasses import dataclass, field
10
+ from typing import Dict, List, Optional, Union
11
+
12
+ FREESECT = 0xFFFFFFFF
13
+ ENDOFCHAIN = 0xFFFFFFFE
14
+ FATSECT = 0xFFFFFFFD
15
+ DIFSECT = 0xFFFFFFFC
16
+ NOSTREAM = 0xFFFFFFFF
17
+
18
+ SECTOR = 512
19
+ MINI_SECTOR = 64
20
+ MINI_CUTOFF = 4096
21
+ FAT_PER_SECTOR = SECTOR // 4
22
+
23
+ TYPE_STORAGE, TYPE_STREAM, TYPE_ROOT = 1, 2, 5
24
+
25
+
26
+ @dataclass
27
+ class Storage:
28
+ children: Dict[str, Union["Storage", bytes]] = field(default_factory=dict)
29
+
30
+ def add_stream(self, name: str, data: bytes) -> None:
31
+ self.children[name] = bytes(data)
32
+
33
+ def add_storage(self, name: str) -> "Storage":
34
+ s = self.children.get(name)
35
+ if not isinstance(s, Storage):
36
+ s = Storage()
37
+ self.children[name] = s
38
+ return s
39
+
40
+ def set_path(self, path: str, data: bytes) -> None:
41
+ parts = path.split("/")
42
+ s = self
43
+ for p in parts[:-1]:
44
+ s = s.add_storage(p)
45
+ s.add_stream(parts[-1], data)
46
+
47
+ def storage_path(self, path: str) -> "Storage":
48
+ s = self
49
+ for p in path.split("/"):
50
+ s = s.add_storage(p)
51
+ return s
52
+
53
+
54
+ class _Entry:
55
+ __slots__ = ("name", "etype", "data", "child", "left", "right", "start", "size", "index", "clsid")
56
+
57
+ def __init__(self, name: str, etype: int, data: Optional[bytes] = None):
58
+ self.name = name
59
+ self.etype = etype
60
+ self.data = data
61
+ self.child = NOSTREAM
62
+ self.left = NOSTREAM
63
+ self.right = NOSTREAM
64
+ self.start = ENDOFCHAIN
65
+ self.size = 0
66
+ self.index = -1
67
+ self.clsid = b"\0" * 16
68
+
69
+
70
+ def _name_key(name: str):
71
+ # [MS-CFB] 2.6.4: compare by UTF-16 length first, then upper-cased code units.
72
+ u = name.encode("utf-16-le")
73
+ return (len(u), name.upper().encode("utf-16-le"))
74
+
75
+
76
+ def _build_tree(entries: List[_Entry]) -> int:
77
+ """Return index of root of a balanced binary tree over sorted entries."""
78
+ if not entries:
79
+ return NOSTREAM
80
+ mid = len(entries) // 2
81
+ node = entries[mid]
82
+ node.left = _build_tree(entries[:mid])
83
+ node.right = _build_tree(entries[mid + 1:])
84
+ return node.index
85
+
86
+
87
+ def _pad(data: bytes, unit: int) -> bytes:
88
+ rem = len(data) % unit
89
+ return data if rem == 0 else data + b"\0" * (unit - rem)
90
+
91
+
92
+ def write_cfb(root: Storage, root_clsid: bytes = b"\0" * 16) -> bytes:
93
+ # ---- flatten directory -------------------------------------------------
94
+ entries: List[_Entry] = []
95
+ root_entry = _Entry("Root Entry", TYPE_ROOT)
96
+ root_entry.clsid = root_clsid
97
+ entries.append(root_entry)
98
+
99
+ def flatten(storage: Storage, parent: _Entry) -> None:
100
+ kids: List[_Entry] = []
101
+ for name, val in storage.children.items():
102
+ if isinstance(val, Storage):
103
+ e = _Entry(name, TYPE_STORAGE)
104
+ else:
105
+ e = _Entry(name, TYPE_STREAM, val)
106
+ entries.append(e)
107
+ kids.append(e)
108
+ for e in kids:
109
+ e.index = entries.index(e)
110
+ kids.sort(key=lambda e: _name_key(e.name))
111
+ parent.child = _build_tree(kids)
112
+ for e in kids:
113
+ if e.etype == TYPE_STORAGE:
114
+ flatten(storage.children[e.name], e)
115
+
116
+ root_entry.index = 0
117
+ flatten(root, root_entry)
118
+ for i, e in enumerate(entries):
119
+ e.index = i
120
+
121
+ # ---- mini stream -------------------------------------------------------
122
+ mini_stream = bytearray()
123
+ minifat: List[int] = []
124
+ for e in entries:
125
+ if e.etype == TYPE_STREAM and 0 < len(e.data) < MINI_CUTOFF:
126
+ e.size = len(e.data)
127
+ e.start = len(minifat)
128
+ padded = _pad(e.data, MINI_SECTOR)
129
+ n = len(padded) // MINI_SECTOR
130
+ mini_stream += padded
131
+ minifat.extend(range(len(minifat) + 1, len(minifat) + n))
132
+ minifat.append(ENDOFCHAIN)
133
+ root_entry.size = len(mini_stream)
134
+
135
+ # ---- regular sectors ---------------------------------------------------
136
+ sectors: List[bytes] = [] # sector payloads (512 bytes each)
137
+ fat: List[int] = [] # FAT entry per sector
138
+
139
+ def add_chain(data: bytes) -> int:
140
+ padded = _pad(data, SECTOR)
141
+ n = len(padded) // SECTOR
142
+ if n == 0:
143
+ return ENDOFCHAIN
144
+ first = len(sectors)
145
+ for i in range(n):
146
+ sectors.append(padded[i * SECTOR:(i + 1) * SECTOR])
147
+ fat.append(first + i + 1 if i < n - 1 else ENDOFCHAIN)
148
+ return first
149
+
150
+ # large streams
151
+ for e in entries:
152
+ if e.etype == TYPE_STREAM and len(e.data) >= MINI_CUTOFF:
153
+ e.size = len(e.data)
154
+ e.start = add_chain(e.data)
155
+
156
+ # mini stream itself lives in regular sectors, pointed to by root entry
157
+ root_entry.start = add_chain(bytes(mini_stream)) if mini_stream else ENDOFCHAIN
158
+
159
+ # minifat
160
+ minifat_start, minifat_count = ENDOFCHAIN, 0
161
+ if minifat:
162
+ raw = b"".join(struct.pack("<I", v) for v in minifat)
163
+ raw = _pad(raw, SECTOR)
164
+ # pad unused minifat slots with FREESECT
165
+ raw = raw[: len(minifat) * 4] + b"\xff" * (len(raw) - len(minifat) * 4)
166
+ minifat_start = add_chain(raw)
167
+ minifat_count = len(raw) // SECTOR
168
+
169
+ # directory
170
+ dir_raw = bytearray()
171
+ for e in entries:
172
+ dir_raw += _dir_entry(e)
173
+ dir_raw = _pad(bytes(dir_raw), SECTOR)
174
+ # fill trailing unused entries with valid empty entries
175
+ n_slots = len(dir_raw) // 128
176
+ for i in range(len(entries), n_slots):
177
+ dir_raw = dir_raw[: i * 128] + _dir_entry(None) + dir_raw[(i + 1) * 128:]
178
+ dir_start = add_chain(bytes(dir_raw))
179
+
180
+ # FAT sectors (iterate: FAT sectors — and any DIFAT sectors listing them
181
+ # beyond the header's 109 slots — need FAT entries too)
182
+ DIFAT_PER_SECTOR = FAT_PER_SECTOR - 1 # last dword chains to the next DIFAT sector
183
+ n_data = len(sectors)
184
+ n_fat = n_difat = 0
185
+ while True:
186
+ needed_fat = -(-(n_data + n_fat + n_difat) // FAT_PER_SECTOR)
187
+ needed_difat = -(-max(0, needed_fat - 109) // DIFAT_PER_SECTOR)
188
+ if (needed_fat, needed_difat) == (n_fat, n_difat):
189
+ break
190
+ n_fat, n_difat = needed_fat, needed_difat
191
+
192
+ fat_full = fat + [FATSECT] * n_fat + [DIFSECT] * n_difat
193
+ fat_full += [FREESECT] * (n_fat * FAT_PER_SECTOR - len(fat_full))
194
+ fat_raw = b"".join(struct.pack("<I", v) for v in fat_full)
195
+ fat_sector_ids = list(range(n_data, n_data + n_fat))
196
+ difat_sector_ids = list(range(n_data + n_fat, n_data + n_fat + n_difat))
197
+ for i in range(n_fat):
198
+ sectors.append(fat_raw[i * SECTOR:(i + 1) * SECTOR])
199
+ for i in range(n_difat):
200
+ chunk = fat_sector_ids[109 + i * DIFAT_PER_SECTOR: 109 + (i + 1) * DIFAT_PER_SECTOR]
201
+ chunk += [FREESECT] * (DIFAT_PER_SECTOR - len(chunk))
202
+ nxt = difat_sector_ids[i + 1] if i + 1 < n_difat else ENDOFCHAIN
203
+ sectors.append(b"".join(struct.pack("<I", v) for v in chunk) + struct.pack("<I", nxt))
204
+
205
+ # ---- header ------------------------------------------------------------
206
+ hdr = bytearray(SECTOR)
207
+ struct.pack_into("<8s16sHHHHH6sIIIIIIIII", hdr, 0,
208
+ b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1", b"\0" * 16,
209
+ 0x003E, 0x0003, 0xFFFE, 9, 6, b"\0" * 6,
210
+ 0, # number of directory sectors (v3: 0)
211
+ n_fat,
212
+ dir_start,
213
+ 0, # transaction signature
214
+ MINI_CUTOFF,
215
+ minifat_start, minifat_count,
216
+ difat_sector_ids[0] if n_difat else ENDOFCHAIN, n_difat)
217
+ difat = fat_sector_ids[:109] + [FREESECT] * max(0, 109 - len(fat_sector_ids))
218
+ struct.pack_into("<109I", hdr, 76, *difat)
219
+
220
+ return bytes(hdr) + b"".join(sectors)
221
+
222
+
223
+ def _dir_entry(e: Optional[_Entry]) -> bytes:
224
+ if e is None:
225
+ return struct.pack("<64sHBBIII16sIQQIQ", b"\0" * 64, 0, 0, 0,
226
+ NOSTREAM, NOSTREAM, NOSTREAM, b"\0" * 16, 0, 0, 0, 0, 0)
227
+ name = e.name.encode("utf-16-le") + b"\0\0"
228
+ if len(name) > 64:
229
+ raise ValueError(f"name too long: {e.name!r}")
230
+ return struct.pack("<64sHBBIII16sIQQIQ",
231
+ name.ljust(64, b"\0"), len(name), e.etype, 1, # colour: black
232
+ e.left, e.right, e.child, e.clsid, 0, 0, 0,
233
+ e.start if e.start != ENDOFCHAIN else (0 if e.etype != TYPE_STREAM else ENDOFCHAIN),
234
+ e.size)
235
+
236
+
237
+ def load_cfb(path: str) -> Storage:
238
+ """Read an existing compound file into a Storage tree (via olefile)."""
239
+ import olefile
240
+ ole = olefile.OleFileIO(path)
241
+ root = Storage()
242
+ for parts in ole.listdir(streams=True, storages=True):
243
+ p = "/".join(parts)
244
+ if ole.get_type(p) == olefile.STGTY_STORAGE:
245
+ root.storage_path(p)
246
+ else:
247
+ root.set_path(p, ole.openstream(p).read())
248
+ return root