pymppwriter 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pymppwriter/__init__.py +12 -0
- pymppwriter/blocks.py +363 -0
- pymppwriter/cfb.py +248 -0
- pymppwriter/cli.py +107 -0
- pymppwriter/native_fields.json +1991 -0
- pymppwriter/reader.py +248 -0
- pymppwriter/writer.py +1258 -0
- pymppwriter-0.3.0.dist-info/METADATA +259 -0
- pymppwriter-0.3.0.dist-info/RECORD +13 -0
- pymppwriter-0.3.0.dist-info/WHEEL +5 -0
- pymppwriter-0.3.0.dist-info/entry_points.txt +2 -0
- pymppwriter-0.3.0.dist-info/licenses/LICENSE +21 -0
- pymppwriter-0.3.0.dist-info/top_level.txt +1 -0
pymppwriter/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from .writer import (MppWriter, Project, Task, Relation, Resource, Assignment,
|
|
2
|
+
Calendar, CalendarException, ScheduleWarning, validate)
|
|
3
|
+
from .reader import read_project, MppReadError
|
|
4
|
+
|
|
5
|
+
try: # the installed distribution's version
|
|
6
|
+
from importlib.metadata import PackageNotFoundError, version as _version
|
|
7
|
+
__version__ = _version("pymppwriter")
|
|
8
|
+
except (ImportError, PackageNotFoundError): # running from a source tree
|
|
9
|
+
__version__ = "0.0.0.dev0"
|
|
10
|
+
__all__ = ["MppWriter", "Project", "Task", "Relation", "Resource", "Assignment",
|
|
11
|
+
"Calendar", "CalendarException", "ScheduleWarning", "validate",
|
|
12
|
+
"read_project", "MppReadError", "__version__"]
|
pymppwriter/blocks.py
ADDED
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
"""Decoders/encoders for the block structures inside an MPP14 file.
|
|
2
|
+
|
|
3
|
+
Layouts derived from the public behaviour of the LGPL MPXJ reader and from
|
|
4
|
+
diffing files saved by Microsoft Project.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
import struct
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from datetime import datetime, timedelta
|
|
10
|
+
from typing import Dict, List, Optional, Tuple
|
|
11
|
+
|
|
12
|
+
MAGIC = 0xFADFADBA
|
|
13
|
+
EPOCH = datetime(1983, 12, 31)
|
|
14
|
+
|
|
15
|
+
PROPS_TASK_FIELD_MAP = 131092
|
|
16
|
+
PROPS_RESOURCE_FIELD_MAP = 131093
|
|
17
|
+
PROPS_RELATION_FIELD_MAP = 131094
|
|
18
|
+
PROPS_ASSIGNMENT_FIELD_MAP = 131095
|
|
19
|
+
PROPS_PROJECT_START_DATE = 37748738
|
|
20
|
+
PROPS_PROJECT_FINISH_DATE = 37748739 # 0x2400003
|
|
21
|
+
PROPS_TITLE = 37748744
|
|
22
|
+
PROPS_DEFAULT_CALENDAR_NAME = 37748750 # UTF-16 name + 4 NUL bytes
|
|
23
|
+
PROPS_CURRENCY_SYMBOL = 37748752 # 0x2400010
|
|
24
|
+
PROPS_STATUS_DATE = 37748805 # 0x2400045; 0xFFFFFFFF = NA
|
|
25
|
+
PROPS_CURRENCY_CODE = 37753787 # 0x24013BB, e.g. "USD"
|
|
26
|
+
PROPS_LEGACY_NEXT_UIDS = 37748910 # 0x24000AE: 2010-era only; stale values make
|
|
27
|
+
# Project renumber task uids, M365 drops it on save
|
|
28
|
+
PROPS_EDITED_BASE_CALENDARS = 8388609 # 0x800001: base calendars with custom data
|
|
29
|
+
# 0x10001..: Var2Data byte length per storage — Project truncates its var-data
|
|
30
|
+
# read at the declared length, so a stale value hides var entries
|
|
31
|
+
PROPS_VAR2DATA_SIZE = {"TBkndTask": 65537, "TBkndRsc": 65538, "TBkndCal": 65539,
|
|
32
|
+
"TBkndAssn": 65540}
|
|
33
|
+
# record-count dwords: Project sizes its tables from these on load and drops
|
|
34
|
+
# records beyond the count (verified against four Project-written files)
|
|
35
|
+
PROPS_TASK_RECORD_COUNT = 16777217 # 0x1000001, includes stubs + uid-0 summary
|
|
36
|
+
PROPS_RESOURCE_RECORD_COUNT = 16777218 # 0x1000002
|
|
37
|
+
PROPS_ASSN_RECORD_COUNT = 16777220 # 0x1000004
|
|
38
|
+
PROPS_REL_RECORD_COUNT = 16777221 # 0x1000005
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# ---------------------------------------------------------------- Props ----
|
|
42
|
+
PROPS_TYPES: Dict[int, int] = {} # key -> type code (third dword of each entry; 0/2/4/9 observed)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def parse_props(data: bytes) -> Tuple[bytes, Dict[int, bytes], List[int]]:
|
|
46
|
+
"""Return (16-byte header, {key: value}, key order). Entry type codes are kept in PROPS_TYPES."""
|
|
47
|
+
header = data[:16]
|
|
48
|
+
count = struct.unpack_from("<H", header, 12)[0]
|
|
49
|
+
pos, out, order = 16, {}, []
|
|
50
|
+
for _ in range(count):
|
|
51
|
+
if len(data) - pos < 12:
|
|
52
|
+
break
|
|
53
|
+
size, key, ptype = struct.unpack_from("<III", data, pos)
|
|
54
|
+
pos += 12
|
|
55
|
+
val = data[pos:pos + size]
|
|
56
|
+
pos += size + (size & 1) # 2-byte alignment
|
|
57
|
+
out[key] = val
|
|
58
|
+
PROPS_TYPES[key] = ptype
|
|
59
|
+
order.append(key)
|
|
60
|
+
return header, out, order
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def build_props(header: bytes, values: Dict[int, bytes], order: List[int]) -> bytes:
|
|
64
|
+
hdr = bytearray(header)
|
|
65
|
+
struct.pack_into("<H", hdr, 12, len(order))
|
|
66
|
+
body = bytearray()
|
|
67
|
+
for key in order:
|
|
68
|
+
val = values[key]
|
|
69
|
+
body += struct.pack("<III", len(val), key, PROPS_TYPES.get(key, 0)) + val
|
|
70
|
+
if len(val) & 1:
|
|
71
|
+
body += b"\0"
|
|
72
|
+
total = len(hdr) + len(body)
|
|
73
|
+
struct.pack_into("<II", hdr, 0, total - 4, total - 4) # header dwords 0,1 = stream size - 4
|
|
74
|
+
return bytes(hdr) + bytes(body)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ------------------------------------------------------------- FieldMap ----
|
|
78
|
+
@dataclass
|
|
79
|
+
class FieldItem:
|
|
80
|
+
type_value: int # MS Project field ID (TaskField/ResourceField numeric)
|
|
81
|
+
block: int # 0 = FixedData, 1 = Fixed2Data
|
|
82
|
+
offset: int # byte offset within the fixed record (65535 = not fixed)
|
|
83
|
+
var_key: int # key in Var2Data when stored as variable data
|
|
84
|
+
category: int
|
|
85
|
+
mask: int
|
|
86
|
+
raw: bytes
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def in_fixed(self) -> bool:
|
|
90
|
+
return self.category not in (0x0B, 0x64) and self.offset != 65535
|
|
91
|
+
|
|
92
|
+
@property
|
|
93
|
+
def in_meta(self) -> bool:
|
|
94
|
+
return self.category in (0x0B, 0x64)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def parse_field_map(data: bytes) -> List[FieldItem]:
|
|
98
|
+
items, last, block = [], 0, 0
|
|
99
|
+
for i in range(0, len(data) - 27, 28):
|
|
100
|
+
mask = struct.unpack_from("<I", data, i)[0]
|
|
101
|
+
offset = struct.unpack_from("<H", data, i + 4)[0]
|
|
102
|
+
var_key = data[i + 6]
|
|
103
|
+
type_value = struct.unpack_from("<I", data, i + 12)[0]
|
|
104
|
+
category = struct.unpack_from("<H", data, i + 20)[0]
|
|
105
|
+
if category not in (0x0B, 0x64) and offset != 65535:
|
|
106
|
+
if offset < last:
|
|
107
|
+
block += 1
|
|
108
|
+
last = offset
|
|
109
|
+
items.append(FieldItem(type_value, block, offset, var_key, category, mask, data[i:i + 28]))
|
|
110
|
+
return items
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# -------------------------------------------------------- Fixed blocks -----
|
|
114
|
+
def parse_fixed_meta(data: bytes, item_size: int) -> Tuple[bytes, int, List[bytes]]:
|
|
115
|
+
magic, unk, count, unk2 = struct.unpack_from("<IIII", data, 0)
|
|
116
|
+
assert magic == MAGIC, hex(magic)
|
|
117
|
+
n = (len(data) - 16) // item_size
|
|
118
|
+
items = [data[16 + i * item_size:16 + (i + 1) * item_size] for i in range(n)]
|
|
119
|
+
return data[:16], count, items
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def parse_fixed_meta_auto(data: bytes, default_size: int) -> Tuple[bytes, int, List[bytes]]:
|
|
123
|
+
"""parse_fixed_meta with the item size derived from the header count, so
|
|
124
|
+
files of any Project vintage parse (M365 uses 96/51/10-byte Fixed2Meta
|
|
125
|
+
items where 2010-era files use 92/50/9). Falls back to default_size when
|
|
126
|
+
the stream length is not an exact multiple (trailing slack)."""
|
|
127
|
+
count = struct.unpack_from("<I", data, 8)[0]
|
|
128
|
+
size = default_size
|
|
129
|
+
if count and (len(data) - 16) % count == 0:
|
|
130
|
+
size = (len(data) - 16) // count
|
|
131
|
+
return parse_fixed_meta(data, size)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def build_fixed_meta(header: bytes, items: List[bytes], data_len: Optional[int] = None) -> bytes:
|
|
135
|
+
hdr = bytearray(header)
|
|
136
|
+
struct.pack_into("<I", hdr, 8, len(items))
|
|
137
|
+
if data_len is not None:
|
|
138
|
+
struct.pack_into("<I", hdr, 12, data_len) # header dword 3 = FixedData byte length
|
|
139
|
+
return bytes(hdr) + b"".join(items)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def split_fixed_data(data: bytes, meta_items: List[bytes]) -> List[bytes]:
|
|
143
|
+
out = []
|
|
144
|
+
for i, m in enumerate(meta_items):
|
|
145
|
+
off = struct.unpack_from("<I", m, 4)[0]
|
|
146
|
+
if i + 1 < len(meta_items):
|
|
147
|
+
nxt = struct.unpack_from("<I", meta_items[i + 1], 4)[0]
|
|
148
|
+
else:
|
|
149
|
+
nxt = len(data)
|
|
150
|
+
out.append(data[off:nxt])
|
|
151
|
+
return out
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
# ------------------------------------------------------- meta bitmaps ------
|
|
155
|
+
# A FixedMeta / Fixed2Meta item is: uint32 flags, uint32 offset-in-FixedData,
|
|
156
|
+
# then a bitmap with one bit per TASK_FIELD_MAP entry (little-endian bit order).
|
|
157
|
+
# FixedMeta carries entries 0..(item_size-8)*8-1; Fixed2Meta continues from there.
|
|
158
|
+
# Boolean fields (category 0x0B/0x64) store their value in their entry's bit;
|
|
159
|
+
# for other fields the bit marks the field as populated.
|
|
160
|
+
|
|
161
|
+
def meta_bit(meta: bytes, meta2: bytes, entry_index: int) -> Optional[int]:
|
|
162
|
+
nbits0 = (len(meta) - 8) * 8
|
|
163
|
+
buf, i = (meta, entry_index) if entry_index < nbits0 else (meta2, entry_index - nbits0)
|
|
164
|
+
byte = 8 + i // 8
|
|
165
|
+
if byte >= len(buf):
|
|
166
|
+
return None
|
|
167
|
+
return (buf[byte] >> (i % 8)) & 1
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def set_meta_bit(meta: bytearray, meta2: bytearray, entry_index: int, value: bool) -> None:
|
|
171
|
+
nbits0 = (len(meta) - 8) * 8
|
|
172
|
+
buf, i = (meta, entry_index) if entry_index < nbits0 else (meta2, entry_index - nbits0)
|
|
173
|
+
byte = 8 + i // 8
|
|
174
|
+
if byte >= len(buf):
|
|
175
|
+
return
|
|
176
|
+
if value:
|
|
177
|
+
buf[byte] |= 1 << (i % 8)
|
|
178
|
+
else:
|
|
179
|
+
buf[byte] &= ~(1 << (i % 8)) & 0xFF
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
# ---------------------------------------------------------- Var blocks -----
|
|
183
|
+
def parse_var_meta(data: bytes) -> Tuple[bytes, Dict[int, Dict[int, int]], List[Tuple[int, int, int, int]]]:
|
|
184
|
+
"""VarMeta12: 24-byte header then 12-byte entries (uid, offset, type, unk)."""
|
|
185
|
+
magic, unk, count, unk2, unk3, data_size = struct.unpack_from("<IIIIII", data, 0)
|
|
186
|
+
table: Dict[int, Dict[int, int]] = {}
|
|
187
|
+
entries = []
|
|
188
|
+
pos = 24
|
|
189
|
+
for _ in range(count):
|
|
190
|
+
if len(data) - pos < 12:
|
|
191
|
+
break
|
|
192
|
+
uid, off, typ, unk4 = struct.unpack_from("<IIHH", data, pos)
|
|
193
|
+
pos += 12
|
|
194
|
+
table.setdefault(uid, {})[typ] = off
|
|
195
|
+
entries.append((uid, off, typ, unk4))
|
|
196
|
+
return data[:24], table, entries
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def read_var(data: bytes, off: int) -> bytes:
|
|
200
|
+
size = struct.unpack_from("<I", data, off)[0]
|
|
201
|
+
return data[off + 4:off + 4 + size]
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
TASK_FIELD_HI, RESOURCE_FIELD_HI, ASSIGNMENT_FIELD_HI = 0x0B40, 0x0C40, 0x0F40
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def build_var_blocks(header: bytes, values: List[Tuple[int, int, bytes]], field_hi: int = TASK_FIELD_HI) -> Tuple[bytes, bytes]:
|
|
208
|
+
"""values: [(uid, type, payload)] -> (VarMeta, Var2Data).
|
|
209
|
+
|
|
210
|
+
Each entry is (uid, offset, fieldId low16, fieldId high16); the high word is the
|
|
211
|
+
native field-class prefix (0x0B40 tasks, 0x0C40 resources, 0x0F40 assignments).
|
|
212
|
+
Entries must be ordered by (uid, fieldId)."""
|
|
213
|
+
meta = bytearray(header)
|
|
214
|
+
var = bytearray()
|
|
215
|
+
entries = bytearray()
|
|
216
|
+
for uid, typ, payload in sorted(values, key=lambda v: (v[0], v[1])):
|
|
217
|
+
entries += struct.pack("<IIHH", uid, len(var), typ, field_hi)
|
|
218
|
+
var += struct.pack("<I", len(payload)) + payload
|
|
219
|
+
struct.pack_into("<I", meta, 8, len(values))
|
|
220
|
+
struct.pack_into("<I", meta, 20, len(var))
|
|
221
|
+
return bytes(meta) + bytes(entries), bytes(var)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# ------------------------------------- OLE property sets (MS-OLEPS) --------
|
|
225
|
+
VT_LPSTR = 30
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def update_property_set_strings(data: bytes, updates: Dict[int, str], section: int = 0,
|
|
229
|
+
codepage: str = "cp1252") -> bytes:
|
|
230
|
+
"""Replace or add VT_LPSTR properties in one section of an OLE property set
|
|
231
|
+
stream (SummaryInformation / DocumentSummaryInformation), keeping every
|
|
232
|
+
other property byte-for-byte."""
|
|
233
|
+
nsec = struct.unpack_from("<I", data, 24)[0]
|
|
234
|
+
header = data[:28]
|
|
235
|
+
secs = []
|
|
236
|
+
for s in range(nsec):
|
|
237
|
+
fmtid = data[28 + s * 20:44 + s * 20]
|
|
238
|
+
off = struct.unpack_from("<I", data, 44 + s * 20)[0]
|
|
239
|
+
size, cnt = struct.unpack_from("<II", data, off)
|
|
240
|
+
entries = [struct.unpack_from("<II", data, off + 8 + i * 8) for i in range(cnt)]
|
|
241
|
+
bounds = sorted(e[1] for e in entries) + [size]
|
|
242
|
+
raw, order = {}, []
|
|
243
|
+
for pid, poff in entries:
|
|
244
|
+
nxt = min(b for b in bounds if b > poff)
|
|
245
|
+
raw[pid] = data[off + poff:off + nxt]
|
|
246
|
+
order.append(pid)
|
|
247
|
+
secs.append([fmtid, order, raw])
|
|
248
|
+
fmtid, order, raw = secs[section]
|
|
249
|
+
for pid, val in updates.items():
|
|
250
|
+
b = val.encode(codepage, "replace") + b"\0"
|
|
251
|
+
v = struct.pack("<II", VT_LPSTR, len(b)) + b
|
|
252
|
+
raw[pid] = v + b"\0" * ((-len(v)) % 4)
|
|
253
|
+
if pid not in order:
|
|
254
|
+
order.append(pid)
|
|
255
|
+
out_secs = []
|
|
256
|
+
for fm, ord_, rw in secs:
|
|
257
|
+
base = 8 + len(ord_) * 8
|
|
258
|
+
body, entries = bytearray(), []
|
|
259
|
+
for pid in ord_:
|
|
260
|
+
v = rw[pid] + b"\0" * ((-len(rw[pid])) % 4)
|
|
261
|
+
entries.append((pid, base + len(body)))
|
|
262
|
+
body += v
|
|
263
|
+
sec = struct.pack("<II", base + len(body), len(ord_))
|
|
264
|
+
for pid, poff in entries:
|
|
265
|
+
sec += struct.pack("<II", pid, poff)
|
|
266
|
+
out_secs.append((fm, sec + bytes(body)))
|
|
267
|
+
dir_, pos = bytearray(), 28 + len(out_secs) * 20
|
|
268
|
+
for fm, sec in out_secs:
|
|
269
|
+
dir_ += fm + struct.pack("<I", pos)
|
|
270
|
+
pos += len(sec)
|
|
271
|
+
return bytes(header) + bytes(dir_) + b"".join(sec for _, sec in out_secs)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
# ------------------------------------------------------ calendar data ------
|
|
275
|
+
CAL_DAY_NONWORKING, CAL_DAY_DEFAULT, CAL_DAY_WORKING = 0, 1, 2
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def build_calendar_data(days, exceptions=()) -> bytes:
|
|
279
|
+
"""Calendar definition blob (var-data key 8 on a TBkndCal record), in the
|
|
280
|
+
dialect Microsoft Project M365 writes (an earlier form using day type 2
|
|
281
|
+
for working days is readable by MPXJ but ignored by Project).
|
|
282
|
+
|
|
283
|
+
days: 7 (day_type, ranges) tuples, SUNDAY first; ranges are
|
|
284
|
+
(start_minute, end_minute) pairs from midnight, at most 5 per day and only
|
|
285
|
+
meaningful for CAL_DAY_WORKING.
|
|
286
|
+
exceptions: (from_date, to_date, name) tuples of non-working days, sorted.
|
|
287
|
+
|
|
288
|
+
Each 60-byte day block: uint16 day type on the wire (1 = default,
|
|
289
|
+
0 = explicit; working vs non-working is the range count), uint16 range
|
|
290
|
+
count, uint32 total working tenths-of-a-minute at +4, range start times as
|
|
291
|
+
uint16 tenths at +8 (stride 2), range durations as uint32 tenths at +20
|
|
292
|
+
(stride 4) and duplicated at +40. Exceptions: uint32 count, then per
|
|
293
|
+
exception a 92-byte record (uint16 from-day, uint16 to-day, uint16 day
|
|
294
|
+
count, recurrence dwords 1, 0, 1, 0x4000 at +72, uint32 name byte length
|
|
295
|
+
at +88) followed by the UTF-16 name, zero-padded to a 4-byte boundary plus
|
|
296
|
+
4 more zero bytes (both as observed in Project-written files).
|
|
297
|
+
"""
|
|
298
|
+
if len(days) != 7:
|
|
299
|
+
raise ValueError("days must have exactly 7 entries, Sunday first")
|
|
300
|
+
out = bytearray()
|
|
301
|
+
for dtype, ranges in days:
|
|
302
|
+
b = bytearray(60)
|
|
303
|
+
struct.pack_into("<H", b, 0, 1 if dtype == CAL_DAY_DEFAULT else 0)
|
|
304
|
+
if dtype == CAL_DAY_WORKING:
|
|
305
|
+
ranges = list(ranges)[:5]
|
|
306
|
+
struct.pack_into("<H", b, 2, len(ranges))
|
|
307
|
+
struct.pack_into("<I", b, 4, sum(end - start for start, end in ranges) * 10)
|
|
308
|
+
cumulative = 0
|
|
309
|
+
for i, (start, end) in enumerate(ranges):
|
|
310
|
+
struct.pack_into("<H", b, 8 + 2 * i, start * 10)
|
|
311
|
+
struct.pack_into("<I", b, 20 + 4 * i, (end - start) * 10)
|
|
312
|
+
cumulative += (end - start) * 10
|
|
313
|
+
struct.pack_into("<I", b, 40 + 4 * i, cumulative)
|
|
314
|
+
out += b
|
|
315
|
+
if exceptions:
|
|
316
|
+
out += struct.pack("<I", len(exceptions))
|
|
317
|
+
for i, (from_date, to_date, name) in enumerate(exceptions):
|
|
318
|
+
rec = bytearray(92)
|
|
319
|
+
d1 = (from_date - EPOCH.date()).days
|
|
320
|
+
d2 = (to_date - EPOCH.date()).days
|
|
321
|
+
struct.pack_into("<HH", rec, 0, d1, d2)
|
|
322
|
+
struct.pack_into("<H", rec, 4, d2 - d1 + 1)
|
|
323
|
+
struct.pack_into("<I", rec, 72, 1)
|
|
324
|
+
struct.pack_into("<I", rec, 80, 1)
|
|
325
|
+
struct.pack_into("<I", rec, 84, 0x4000)
|
|
326
|
+
nb = (name + "\0").encode("utf-16-le")
|
|
327
|
+
struct.pack_into("<I", rec, 88, len(nb))
|
|
328
|
+
# next record 4-byte aligned; 4 extra zero bytes close the blob
|
|
329
|
+
out += rec + nb + b"\0" * ((-len(nb)) % 4)
|
|
330
|
+
if i == len(exceptions) - 1:
|
|
331
|
+
out += b"\0\0\0\0"
|
|
332
|
+
return bytes(out)
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
# ------------------------------------------------------- primitives --------
|
|
336
|
+
def decode_timestamp(b: bytes, off: int) -> Optional[datetime]:
|
|
337
|
+
time, days = struct.unpack_from("<HH", b, off)
|
|
338
|
+
if days <= 1 or days == 65535:
|
|
339
|
+
return None
|
|
340
|
+
if time == 65535:
|
|
341
|
+
time = 0
|
|
342
|
+
return EPOCH + timedelta(days=days, seconds=time * 6)
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def encode_timestamp(dt: Optional[datetime]) -> bytes:
|
|
346
|
+
if dt is None:
|
|
347
|
+
return b"\xff\xff\xff\xff"
|
|
348
|
+
delta = dt - EPOCH
|
|
349
|
+
days = delta.days
|
|
350
|
+
tenths = delta.seconds // 6
|
|
351
|
+
return struct.pack("<HH", tenths, days)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def encode_unicode(s: str) -> bytes:
|
|
355
|
+
return s.encode("utf-16-le") + b"\0\0"
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def decode_unicode(b: bytes) -> str:
|
|
359
|
+
return b.decode("utf-16-le", errors="replace").split("\0", 1)[0]
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
# duration units (MS Project codes) -> tenths-of-a-minute divisor
|
|
363
|
+
DURATION_UNITS = {3: ("m", 10), 5: ("h", 600), 7: ("d", 4800), 9: ("w", 24000), 11: ("mo", 96000)}
|
pymppwriter/cfb.py
ADDED
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Minimal writer for Microsoft Compound File Binary (OLE2) containers.
|
|
2
|
+
|
|
3
|
+
Implements enough of [MS-CFB] v3 (512-byte sectors, 64-byte mini sectors,
|
|
4
|
+
4096-byte mini-stream cutoff) to produce files that Apache POI / olefile /
|
|
5
|
+
MS Project can read. Written from the public [MS-CFB] specification.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
import struct
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import Dict, List, Optional, Union
|
|
11
|
+
|
|
12
|
+
FREESECT = 0xFFFFFFFF
|
|
13
|
+
ENDOFCHAIN = 0xFFFFFFFE
|
|
14
|
+
FATSECT = 0xFFFFFFFD
|
|
15
|
+
DIFSECT = 0xFFFFFFFC
|
|
16
|
+
NOSTREAM = 0xFFFFFFFF
|
|
17
|
+
|
|
18
|
+
SECTOR = 512
|
|
19
|
+
MINI_SECTOR = 64
|
|
20
|
+
MINI_CUTOFF = 4096
|
|
21
|
+
FAT_PER_SECTOR = SECTOR // 4
|
|
22
|
+
|
|
23
|
+
TYPE_STORAGE, TYPE_STREAM, TYPE_ROOT = 1, 2, 5
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class Storage:
|
|
28
|
+
children: Dict[str, Union["Storage", bytes]] = field(default_factory=dict)
|
|
29
|
+
|
|
30
|
+
def add_stream(self, name: str, data: bytes) -> None:
|
|
31
|
+
self.children[name] = bytes(data)
|
|
32
|
+
|
|
33
|
+
def add_storage(self, name: str) -> "Storage":
|
|
34
|
+
s = self.children.get(name)
|
|
35
|
+
if not isinstance(s, Storage):
|
|
36
|
+
s = Storage()
|
|
37
|
+
self.children[name] = s
|
|
38
|
+
return s
|
|
39
|
+
|
|
40
|
+
def set_path(self, path: str, data: bytes) -> None:
|
|
41
|
+
parts = path.split("/")
|
|
42
|
+
s = self
|
|
43
|
+
for p in parts[:-1]:
|
|
44
|
+
s = s.add_storage(p)
|
|
45
|
+
s.add_stream(parts[-1], data)
|
|
46
|
+
|
|
47
|
+
def storage_path(self, path: str) -> "Storage":
|
|
48
|
+
s = self
|
|
49
|
+
for p in path.split("/"):
|
|
50
|
+
s = s.add_storage(p)
|
|
51
|
+
return s
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class _Entry:
|
|
55
|
+
__slots__ = ("name", "etype", "data", "child", "left", "right", "start", "size", "index", "clsid")
|
|
56
|
+
|
|
57
|
+
def __init__(self, name: str, etype: int, data: Optional[bytes] = None):
|
|
58
|
+
self.name = name
|
|
59
|
+
self.etype = etype
|
|
60
|
+
self.data = data
|
|
61
|
+
self.child = NOSTREAM
|
|
62
|
+
self.left = NOSTREAM
|
|
63
|
+
self.right = NOSTREAM
|
|
64
|
+
self.start = ENDOFCHAIN
|
|
65
|
+
self.size = 0
|
|
66
|
+
self.index = -1
|
|
67
|
+
self.clsid = b"\0" * 16
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _name_key(name: str):
|
|
71
|
+
# [MS-CFB] 2.6.4: compare by UTF-16 length first, then upper-cased code units.
|
|
72
|
+
u = name.encode("utf-16-le")
|
|
73
|
+
return (len(u), name.upper().encode("utf-16-le"))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _build_tree(entries: List[_Entry]) -> int:
|
|
77
|
+
"""Return index of root of a balanced binary tree over sorted entries."""
|
|
78
|
+
if not entries:
|
|
79
|
+
return NOSTREAM
|
|
80
|
+
mid = len(entries) // 2
|
|
81
|
+
node = entries[mid]
|
|
82
|
+
node.left = _build_tree(entries[:mid])
|
|
83
|
+
node.right = _build_tree(entries[mid + 1:])
|
|
84
|
+
return node.index
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _pad(data: bytes, unit: int) -> bytes:
|
|
88
|
+
rem = len(data) % unit
|
|
89
|
+
return data if rem == 0 else data + b"\0" * (unit - rem)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def write_cfb(root: Storage, root_clsid: bytes = b"\0" * 16) -> bytes:
|
|
93
|
+
# ---- flatten directory -------------------------------------------------
|
|
94
|
+
entries: List[_Entry] = []
|
|
95
|
+
root_entry = _Entry("Root Entry", TYPE_ROOT)
|
|
96
|
+
root_entry.clsid = root_clsid
|
|
97
|
+
entries.append(root_entry)
|
|
98
|
+
|
|
99
|
+
def flatten(storage: Storage, parent: _Entry) -> None:
|
|
100
|
+
kids: List[_Entry] = []
|
|
101
|
+
for name, val in storage.children.items():
|
|
102
|
+
if isinstance(val, Storage):
|
|
103
|
+
e = _Entry(name, TYPE_STORAGE)
|
|
104
|
+
else:
|
|
105
|
+
e = _Entry(name, TYPE_STREAM, val)
|
|
106
|
+
entries.append(e)
|
|
107
|
+
kids.append(e)
|
|
108
|
+
for e in kids:
|
|
109
|
+
e.index = entries.index(e)
|
|
110
|
+
kids.sort(key=lambda e: _name_key(e.name))
|
|
111
|
+
parent.child = _build_tree(kids)
|
|
112
|
+
for e in kids:
|
|
113
|
+
if e.etype == TYPE_STORAGE:
|
|
114
|
+
flatten(storage.children[e.name], e)
|
|
115
|
+
|
|
116
|
+
root_entry.index = 0
|
|
117
|
+
flatten(root, root_entry)
|
|
118
|
+
for i, e in enumerate(entries):
|
|
119
|
+
e.index = i
|
|
120
|
+
|
|
121
|
+
# ---- mini stream -------------------------------------------------------
|
|
122
|
+
mini_stream = bytearray()
|
|
123
|
+
minifat: List[int] = []
|
|
124
|
+
for e in entries:
|
|
125
|
+
if e.etype == TYPE_STREAM and 0 < len(e.data) < MINI_CUTOFF:
|
|
126
|
+
e.size = len(e.data)
|
|
127
|
+
e.start = len(minifat)
|
|
128
|
+
padded = _pad(e.data, MINI_SECTOR)
|
|
129
|
+
n = len(padded) // MINI_SECTOR
|
|
130
|
+
mini_stream += padded
|
|
131
|
+
minifat.extend(range(len(minifat) + 1, len(minifat) + n))
|
|
132
|
+
minifat.append(ENDOFCHAIN)
|
|
133
|
+
root_entry.size = len(mini_stream)
|
|
134
|
+
|
|
135
|
+
# ---- regular sectors ---------------------------------------------------
|
|
136
|
+
sectors: List[bytes] = [] # sector payloads (512 bytes each)
|
|
137
|
+
fat: List[int] = [] # FAT entry per sector
|
|
138
|
+
|
|
139
|
+
def add_chain(data: bytes) -> int:
|
|
140
|
+
padded = _pad(data, SECTOR)
|
|
141
|
+
n = len(padded) // SECTOR
|
|
142
|
+
if n == 0:
|
|
143
|
+
return ENDOFCHAIN
|
|
144
|
+
first = len(sectors)
|
|
145
|
+
for i in range(n):
|
|
146
|
+
sectors.append(padded[i * SECTOR:(i + 1) * SECTOR])
|
|
147
|
+
fat.append(first + i + 1 if i < n - 1 else ENDOFCHAIN)
|
|
148
|
+
return first
|
|
149
|
+
|
|
150
|
+
# large streams
|
|
151
|
+
for e in entries:
|
|
152
|
+
if e.etype == TYPE_STREAM and len(e.data) >= MINI_CUTOFF:
|
|
153
|
+
e.size = len(e.data)
|
|
154
|
+
e.start = add_chain(e.data)
|
|
155
|
+
|
|
156
|
+
# mini stream itself lives in regular sectors, pointed to by root entry
|
|
157
|
+
root_entry.start = add_chain(bytes(mini_stream)) if mini_stream else ENDOFCHAIN
|
|
158
|
+
|
|
159
|
+
# minifat
|
|
160
|
+
minifat_start, minifat_count = ENDOFCHAIN, 0
|
|
161
|
+
if minifat:
|
|
162
|
+
raw = b"".join(struct.pack("<I", v) for v in minifat)
|
|
163
|
+
raw = _pad(raw, SECTOR)
|
|
164
|
+
# pad unused minifat slots with FREESECT
|
|
165
|
+
raw = raw[: len(minifat) * 4] + b"\xff" * (len(raw) - len(minifat) * 4)
|
|
166
|
+
minifat_start = add_chain(raw)
|
|
167
|
+
minifat_count = len(raw) // SECTOR
|
|
168
|
+
|
|
169
|
+
# directory
|
|
170
|
+
dir_raw = bytearray()
|
|
171
|
+
for e in entries:
|
|
172
|
+
dir_raw += _dir_entry(e)
|
|
173
|
+
dir_raw = _pad(bytes(dir_raw), SECTOR)
|
|
174
|
+
# fill trailing unused entries with valid empty entries
|
|
175
|
+
n_slots = len(dir_raw) // 128
|
|
176
|
+
for i in range(len(entries), n_slots):
|
|
177
|
+
dir_raw = dir_raw[: i * 128] + _dir_entry(None) + dir_raw[(i + 1) * 128:]
|
|
178
|
+
dir_start = add_chain(bytes(dir_raw))
|
|
179
|
+
|
|
180
|
+
# FAT sectors (iterate: FAT sectors — and any DIFAT sectors listing them
|
|
181
|
+
# beyond the header's 109 slots — need FAT entries too)
|
|
182
|
+
DIFAT_PER_SECTOR = FAT_PER_SECTOR - 1 # last dword chains to the next DIFAT sector
|
|
183
|
+
n_data = len(sectors)
|
|
184
|
+
n_fat = n_difat = 0
|
|
185
|
+
while True:
|
|
186
|
+
needed_fat = -(-(n_data + n_fat + n_difat) // FAT_PER_SECTOR)
|
|
187
|
+
needed_difat = -(-max(0, needed_fat - 109) // DIFAT_PER_SECTOR)
|
|
188
|
+
if (needed_fat, needed_difat) == (n_fat, n_difat):
|
|
189
|
+
break
|
|
190
|
+
n_fat, n_difat = needed_fat, needed_difat
|
|
191
|
+
|
|
192
|
+
fat_full = fat + [FATSECT] * n_fat + [DIFSECT] * n_difat
|
|
193
|
+
fat_full += [FREESECT] * (n_fat * FAT_PER_SECTOR - len(fat_full))
|
|
194
|
+
fat_raw = b"".join(struct.pack("<I", v) for v in fat_full)
|
|
195
|
+
fat_sector_ids = list(range(n_data, n_data + n_fat))
|
|
196
|
+
difat_sector_ids = list(range(n_data + n_fat, n_data + n_fat + n_difat))
|
|
197
|
+
for i in range(n_fat):
|
|
198
|
+
sectors.append(fat_raw[i * SECTOR:(i + 1) * SECTOR])
|
|
199
|
+
for i in range(n_difat):
|
|
200
|
+
chunk = fat_sector_ids[109 + i * DIFAT_PER_SECTOR: 109 + (i + 1) * DIFAT_PER_SECTOR]
|
|
201
|
+
chunk += [FREESECT] * (DIFAT_PER_SECTOR - len(chunk))
|
|
202
|
+
nxt = difat_sector_ids[i + 1] if i + 1 < n_difat else ENDOFCHAIN
|
|
203
|
+
sectors.append(b"".join(struct.pack("<I", v) for v in chunk) + struct.pack("<I", nxt))
|
|
204
|
+
|
|
205
|
+
# ---- header ------------------------------------------------------------
|
|
206
|
+
hdr = bytearray(SECTOR)
|
|
207
|
+
struct.pack_into("<8s16sHHHHH6sIIIIIIIII", hdr, 0,
|
|
208
|
+
b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1", b"\0" * 16,
|
|
209
|
+
0x003E, 0x0003, 0xFFFE, 9, 6, b"\0" * 6,
|
|
210
|
+
0, # number of directory sectors (v3: 0)
|
|
211
|
+
n_fat,
|
|
212
|
+
dir_start,
|
|
213
|
+
0, # transaction signature
|
|
214
|
+
MINI_CUTOFF,
|
|
215
|
+
minifat_start, minifat_count,
|
|
216
|
+
difat_sector_ids[0] if n_difat else ENDOFCHAIN, n_difat)
|
|
217
|
+
difat = fat_sector_ids[:109] + [FREESECT] * max(0, 109 - len(fat_sector_ids))
|
|
218
|
+
struct.pack_into("<109I", hdr, 76, *difat)
|
|
219
|
+
|
|
220
|
+
return bytes(hdr) + b"".join(sectors)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _dir_entry(e: Optional[_Entry]) -> bytes:
|
|
224
|
+
if e is None:
|
|
225
|
+
return struct.pack("<64sHBBIII16sIQQIQ", b"\0" * 64, 0, 0, 0,
|
|
226
|
+
NOSTREAM, NOSTREAM, NOSTREAM, b"\0" * 16, 0, 0, 0, 0, 0)
|
|
227
|
+
name = e.name.encode("utf-16-le") + b"\0\0"
|
|
228
|
+
if len(name) > 64:
|
|
229
|
+
raise ValueError(f"name too long: {e.name!r}")
|
|
230
|
+
return struct.pack("<64sHBBIII16sIQQIQ",
|
|
231
|
+
name.ljust(64, b"\0"), len(name), e.etype, 1, # colour: black
|
|
232
|
+
e.left, e.right, e.child, e.clsid, 0, 0, 0,
|
|
233
|
+
e.start if e.start != ENDOFCHAIN else (0 if e.etype != TYPE_STREAM else ENDOFCHAIN),
|
|
234
|
+
e.size)
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def load_cfb(path: str) -> Storage:
|
|
238
|
+
"""Read an existing compound file into a Storage tree (via olefile)."""
|
|
239
|
+
import olefile
|
|
240
|
+
ole = olefile.OleFileIO(path)
|
|
241
|
+
root = Storage()
|
|
242
|
+
for parts in ole.listdir(streams=True, storages=True):
|
|
243
|
+
p = "/".join(parts)
|
|
244
|
+
if ole.get_type(p) == olefile.STGTY_STORAGE:
|
|
245
|
+
root.storage_path(p)
|
|
246
|
+
else:
|
|
247
|
+
root.set_path(p, ole.openstream(p).read())
|
|
248
|
+
return root
|