@pptx-studio/writer 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/LICENSE +202 -0
- package/NOTICE +43 -0
- package/README.md +184 -0
- package/dist/index.d.ts +818 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +1165 -0
- package/dist/index.js.map +1 -0
- package/package.json +57 -0
package/dist/index.js
ADDED
|
@@ -0,0 +1,1165 @@
|
|
|
1
|
+
import { CONTENT_TYPES_PART, PartStore, ROOT_RELS_PART, deflatedEntry, isContentTypesStreamName, isRelationshipPartName, isXmlContentType, normalizePartName, passthroughEntry, readZip, relsPartNameFor, sha256Hex, sha256HexOfText, sourcePartNameForRels, writeZip } from "@pptx-studio/opc";
|
|
2
|
+
import { assertValid } from "@pptx-studio/validate";
|
|
3
|
+
import { NS, canonicalXml, encodeXmlSource, excerpt, firstDifference, isXmlError, parseXml } from "@pptx-studio/xml";
|
|
4
|
+
//#region src/errors.ts
|
|
5
|
+
/**
|
|
6
|
+
* The writer's own failures.
|
|
7
|
+
*
|
|
8
|
+
* There are only three, and the smallness is deliberate. Almost everything that
|
|
9
|
+
* can go wrong on an export already has an owner: `OpcError` for the container
|
|
10
|
+
* (a part with no content type, a relationship pointing at nothing), and
|
|
11
|
+
* `ValidateError` for the twenty-nine rules. Re-wrapping either would throw away
|
|
12
|
+
* the detail a caller needs - `ValidateError` carries the whole `Report` on its
|
|
13
|
+
* `detail` - in exchange for a uniform class name nobody dispatches on.
|
|
14
|
+
*
|
|
15
|
+
* So this file holds what is genuinely the writer's: a hook that threw, a
|
|
16
|
+
* garbage collection that could not be shown to be safe, and the one assertion
|
|
17
|
+
* the writer makes on its own output.
|
|
18
|
+
*/
|
|
19
|
+
const WRITER_ERROR_CODES = [
|
|
20
|
+
"ERR_PREPARE_FAILED",
|
|
21
|
+
"ERR_COLLECTION_UNSAFE",
|
|
22
|
+
"ERR_PRESERVATION_BROKEN"
|
|
23
|
+
];
|
|
24
|
+
var WriterError = class extends Error {
|
|
25
|
+
name = "WriterError";
|
|
26
|
+
code;
|
|
27
|
+
detail;
|
|
28
|
+
constructor(code, message, detail = {}, options = {}) {
|
|
29
|
+
super(message, options);
|
|
30
|
+
this.code = code;
|
|
31
|
+
this.detail = detail;
|
|
32
|
+
}
|
|
33
|
+
};
|
|
34
|
+
function isWriterError(value) {
|
|
35
|
+
return value instanceof WriterError;
|
|
36
|
+
}
|
|
37
|
+
//#endregion
|
|
38
|
+
//#region src/gc/reachability.ts
|
|
39
|
+
/** Every part reachable by following relationships from the package root. */
|
|
40
|
+
function reachableParts(store) {
|
|
41
|
+
const reachable = /* @__PURE__ */ new Set();
|
|
42
|
+
const blockedBy = [];
|
|
43
|
+
const queue = [];
|
|
44
|
+
const mark = (partName) => {
|
|
45
|
+
const key = normalizePartName(partName);
|
|
46
|
+
if (reachable.has(key)) return;
|
|
47
|
+
reachable.add(key);
|
|
48
|
+
queue.push(partName);
|
|
49
|
+
const rels = relsPartNameFor(partName);
|
|
50
|
+
if (store.has(rels)) reachable.add(normalizePartName(rels));
|
|
51
|
+
};
|
|
52
|
+
const follow = (rels) => {
|
|
53
|
+
for (const rel of rels.all) {
|
|
54
|
+
if (rel.targetMode === "External") continue;
|
|
55
|
+
let target;
|
|
56
|
+
try {
|
|
57
|
+
target = rels.resolve(rel);
|
|
58
|
+
} catch {
|
|
59
|
+
continue;
|
|
60
|
+
}
|
|
61
|
+
if (store.has(target)) mark(target);
|
|
62
|
+
}
|
|
63
|
+
};
|
|
64
|
+
let root;
|
|
65
|
+
try {
|
|
66
|
+
root = store.rootRelationships();
|
|
67
|
+
} catch {
|
|
68
|
+
return {
|
|
69
|
+
reachable,
|
|
70
|
+
complete: false,
|
|
71
|
+
blockedBy: [ROOT_RELS_PART]
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
if (root.size === 0) return {
|
|
75
|
+
reachable,
|
|
76
|
+
complete: false,
|
|
77
|
+
blockedBy: [ROOT_RELS_PART]
|
|
78
|
+
};
|
|
79
|
+
if (store.has(ROOT_RELS_PART)) reachable.add(normalizePartName(ROOT_RELS_PART));
|
|
80
|
+
follow(root);
|
|
81
|
+
for (let part = queue.pop(); part !== void 0; part = queue.pop()) try {
|
|
82
|
+
follow(store.relationships(part));
|
|
83
|
+
} catch {
|
|
84
|
+
blockedBy.push(relsPartNameFor(part));
|
|
85
|
+
}
|
|
86
|
+
return {
|
|
87
|
+
reachable,
|
|
88
|
+
complete: blockedBy.length === 0,
|
|
89
|
+
blockedBy
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Parts nothing in the package can get to.
|
|
94
|
+
*
|
|
95
|
+
* Relationship parts are excluded by construction rather than by filter - see
|
|
96
|
+
* the header. `[Content_Types].xml` never appears because it is not a part.
|
|
97
|
+
*/
|
|
98
|
+
function orphanedParts(store, walk) {
|
|
99
|
+
return store.partNames.filter((name) => !walk.reachable.has(normalizePartName(name)));
|
|
100
|
+
}
|
|
101
|
+
//#endregion
|
|
102
|
+
//#region src/gc/collect.ts
|
|
103
|
+
/** `/ppt/media/...` - the default sweep set. */
|
|
104
|
+
function isMediaPart(partName) {
|
|
105
|
+
return /^\/ppt\/media\//i.test(partName);
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* Plan and apply in one pass.
|
|
109
|
+
*
|
|
110
|
+
* Not split into a pure plan and a separate apply, and the reason is the fixed
|
|
111
|
+
* point below: removing a part can orphan the parts *it* referenced, and the
|
|
112
|
+
* only way to observe that is to remove and look again. A pure planner would
|
|
113
|
+
* have to simulate the store to find the second round, which is a second
|
|
114
|
+
* implementation of the store.
|
|
115
|
+
*
|
|
116
|
+
* Never throws. A package too broken to reason about yields a plan that
|
|
117
|
+
* collects nothing and says why - refusing would make a deck with one
|
|
118
|
+
* unparseable `.rels` permanently unsaveable, and that `.rels` may well have
|
|
119
|
+
* arrived that way. `collectGarbage` is the variant that refuses, for callers
|
|
120
|
+
* who asked for a sweep and would misread silence as "nothing to do".
|
|
121
|
+
*/
|
|
122
|
+
function planCollection(options) {
|
|
123
|
+
const policy = options.policy ?? "orphaned-here";
|
|
124
|
+
if (policy === "none") return {
|
|
125
|
+
collect: [],
|
|
126
|
+
kept: [],
|
|
127
|
+
safe: true,
|
|
128
|
+
blockedBy: []
|
|
129
|
+
};
|
|
130
|
+
const sweepable = options.sweepable ?? isMediaPart;
|
|
131
|
+
const store = options.store;
|
|
132
|
+
let before = null;
|
|
133
|
+
if (policy === "orphaned-here") {
|
|
134
|
+
const baseline = options.baseline ?? null;
|
|
135
|
+
if (baseline === null) {
|
|
136
|
+
const now = reachableParts(store);
|
|
137
|
+
return {
|
|
138
|
+
collect: [],
|
|
139
|
+
kept: candidateOrphans(store, now, sweepable).map((part) => ({
|
|
140
|
+
part,
|
|
141
|
+
why: "it is unreferenced, but with no baseline to compare against there is no way to tell whether this session orphaned it or it arrived that way."
|
|
142
|
+
})),
|
|
143
|
+
safe: now.complete,
|
|
144
|
+
blockedBy: now.blockedBy
|
|
145
|
+
};
|
|
146
|
+
}
|
|
147
|
+
before = reachableParts(baseline);
|
|
148
|
+
if (!before.complete) return {
|
|
149
|
+
collect: [],
|
|
150
|
+
kept: [],
|
|
151
|
+
safe: false,
|
|
152
|
+
blockedBy: before.blockedBy
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
const collect = [];
|
|
156
|
+
const kept = [];
|
|
157
|
+
const decided = /* @__PURE__ */ new Set();
|
|
158
|
+
for (;;) {
|
|
159
|
+
const now = reachableParts(store);
|
|
160
|
+
if (!now.complete) return {
|
|
161
|
+
collect,
|
|
162
|
+
kept,
|
|
163
|
+
safe: false,
|
|
164
|
+
blockedBy: now.blockedBy
|
|
165
|
+
};
|
|
166
|
+
const found = candidateOrphans(store, now, sweepable).filter((part) => !decided.has(normalizePartName(part)));
|
|
167
|
+
if (found.length === 0) break;
|
|
168
|
+
const removing = [];
|
|
169
|
+
for (const part of found) {
|
|
170
|
+
decided.add(normalizePartName(part));
|
|
171
|
+
if (before !== null && !before.reachable.has(normalizePartName(part))) {
|
|
172
|
+
kept.push({
|
|
173
|
+
part,
|
|
174
|
+
why: "it was already unreferenced when this package was opened, so it is not ours to remove."
|
|
175
|
+
});
|
|
176
|
+
continue;
|
|
177
|
+
}
|
|
178
|
+
removing.push(part);
|
|
179
|
+
}
|
|
180
|
+
if (removing.length === 0) break;
|
|
181
|
+
for (const part of removing) {
|
|
182
|
+
store.removePart(part);
|
|
183
|
+
collect.push(part);
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
return {
|
|
187
|
+
collect,
|
|
188
|
+
kept,
|
|
189
|
+
safe: true,
|
|
190
|
+
blockedBy: []
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
function candidateOrphans(store, walk, sweepable) {
|
|
194
|
+
if (!walk.complete) return [];
|
|
195
|
+
return orphanedParts(store, walk).filter(sweepable);
|
|
196
|
+
}
|
|
197
|
+
/**
|
|
198
|
+
* Collect, and refuse rather than guess.
|
|
199
|
+
*
|
|
200
|
+
* The entry point for a deliberate clean-up: a user asking for one, or a
|
|
201
|
+
* command that knows it has just orphaned something. It throws when the graph
|
|
202
|
+
* could not be walked completely, because a caller who asked for a sweep and
|
|
203
|
+
* got silence would reasonably conclude there had been nothing to sweep.
|
|
204
|
+
*
|
|
205
|
+
* `exportPackage` deliberately does not use this. An export that failed because
|
|
206
|
+
* some unrelated `.rels` will not parse would be refusing to save a file over a
|
|
207
|
+
* defect it did not cause and cannot fix.
|
|
208
|
+
*/
|
|
209
|
+
function collectGarbage(options) {
|
|
210
|
+
const plan = planCollection(options);
|
|
211
|
+
if (plan.safe) return plan;
|
|
212
|
+
throw new WriterError("ERR_COLLECTION_UNSAFE", "refusing to collect: " + String(plan.blockedBy.length) + " relationship part(s) would not parse, so no part in this package can be shown to be unreferenced. First: " + (plan.blockedBy[0] ?? "(unknown)") + ". An unreadable edge is an invisible edge, and an invisible edge makes its target look like garbage.", { blockedBy: plan.blockedBy });
|
|
213
|
+
}
|
|
214
|
+
//#endregion
|
|
215
|
+
//#region src/export/prepare.ts
|
|
216
|
+
/**
|
|
217
|
+
* Run the hooks in the order given.
|
|
218
|
+
*
|
|
219
|
+
* The order is the caller's, not ours, and it is not inferred from anything.
|
|
220
|
+
* Some pairs genuinely depend on each other - autofit must settle before a font
|
|
221
|
+
* pass decides which typefaces are used - and a writer that sorted hooks by
|
|
222
|
+
* some notion of priority would be guessing at a dependency the caller knows
|
|
223
|
+
* for certain.
|
|
224
|
+
*
|
|
225
|
+
* A hook that throws stops the export. That is the point: the alternative is
|
|
226
|
+
* handing over a package with five of the six font artifacts in it, which
|
|
227
|
+
* PowerPoint reports as a problem with the content and no way to find out
|
|
228
|
+
* which.
|
|
229
|
+
*/
|
|
230
|
+
function runPrepare(hooks, store, baseline) {
|
|
231
|
+
const records = [];
|
|
232
|
+
for (const hook of hooks) {
|
|
233
|
+
const notes = [];
|
|
234
|
+
const context = {
|
|
235
|
+
store,
|
|
236
|
+
baseline,
|
|
237
|
+
note: (message) => notes.push(message)
|
|
238
|
+
};
|
|
239
|
+
try {
|
|
240
|
+
hook.run(context);
|
|
241
|
+
} catch (error) {
|
|
242
|
+
throw new WriterError("ERR_PREPARE_FAILED", "the prepare hook " + hook.name + " threw, so nothing was written: " + (error instanceof Error ? error.message : String(error)), { hook: hook.name }, { cause: error });
|
|
243
|
+
}
|
|
244
|
+
if (notes.length > 0) records.push({
|
|
245
|
+
hook: hook.name,
|
|
246
|
+
notes
|
|
247
|
+
});
|
|
248
|
+
}
|
|
249
|
+
return records;
|
|
250
|
+
}
|
|
251
|
+
//#endregion
|
|
252
|
+
//#region src/export/preserve.ts
|
|
253
|
+
/** The key an entry is matched on across the two archives. */
|
|
254
|
+
function entryKey(entry) {
|
|
255
|
+
return isContentTypesStreamName(entry.name) ? CONTENT_TYPES_PART : normalizePartName("/" + entry.name);
|
|
256
|
+
}
|
|
257
|
+
/**
|
|
258
|
+
* Compare the emitted archive against the one it came from.
|
|
259
|
+
*
|
|
260
|
+
* Throws `ERR_PRESERVATION_BROKEN` on the first difference, naming the part.
|
|
261
|
+
* There is no "report and continue" mode: one part quietly re-serialised is the
|
|
262
|
+
* whole category of failure this package is built to make impossible, and every
|
|
263
|
+
* later one is likely the same cause.
|
|
264
|
+
*/
|
|
265
|
+
function assertPreserved(output, original, rewritten, zip = {}) {
|
|
266
|
+
if (original === void 0) return {
|
|
267
|
+
checked: 0,
|
|
268
|
+
rewritten: rewritten.size,
|
|
269
|
+
skipped: "the bytes this package was opened from were not supplied, so there is nothing to compare the output against."
|
|
270
|
+
};
|
|
271
|
+
let before, after;
|
|
272
|
+
try {
|
|
273
|
+
before = readZip(original, zip);
|
|
274
|
+
after = readZip(output, zip);
|
|
275
|
+
} catch (error) {
|
|
276
|
+
return {
|
|
277
|
+
checked: 0,
|
|
278
|
+
rewritten: rewritten.size,
|
|
279
|
+
skipped: "one of the two archives could not be read: " + (error instanceof Error ? error.message : String(error))
|
|
280
|
+
};
|
|
281
|
+
}
|
|
282
|
+
const source = /* @__PURE__ */ new Map();
|
|
283
|
+
for (const entry of before.entries) {
|
|
284
|
+
if (entry.isDirectory) continue;
|
|
285
|
+
source.set(entryKey(entry), entry);
|
|
286
|
+
}
|
|
287
|
+
let checked = 0;
|
|
288
|
+
for (const entry of after.entries) {
|
|
289
|
+
const key = entryKey(entry);
|
|
290
|
+
if (rewritten.has(key)) continue;
|
|
291
|
+
const was = source.get(key);
|
|
292
|
+
if (was === void 0) continue;
|
|
293
|
+
const now = after.raw(entry);
|
|
294
|
+
const then = before.raw(was);
|
|
295
|
+
if (entry.method === was.method && entry.crc32 === was.crc32 && entry.uncompressedSize === was.uncompressedSize && sameBytes(now, then)) {
|
|
296
|
+
checked++;
|
|
297
|
+
continue;
|
|
298
|
+
}
|
|
299
|
+
throw new WriterError("ERR_PRESERVATION_BROKEN", entry.name + " was not edited and came out different anyway (" + describe(was) + " in, " + describe(entry) + " out). An untouched part is streamed through still compressed and is never re-serialised: the moment that stops being true, every feature we cannot parse is a feature we can lose.", { part: "/" + entry.name });
|
|
300
|
+
}
|
|
301
|
+
return {
|
|
302
|
+
checked,
|
|
303
|
+
rewritten: rewritten.size,
|
|
304
|
+
skipped: null
|
|
305
|
+
};
|
|
306
|
+
}
|
|
307
|
+
function describe(entry) {
|
|
308
|
+
return String(entry.compressedSize) + "B method " + String(entry.method) + " crc " + entry.crc32.toString(16);
|
|
309
|
+
}
|
|
310
|
+
function sameBytes(a, b) {
|
|
311
|
+
if (a.byteLength !== b.byteLength) return false;
|
|
312
|
+
for (let i = 0; i < a.byteLength; i++) if (a[i] !== b[i]) return false;
|
|
313
|
+
return true;
|
|
314
|
+
}
|
|
315
|
+
//#endregion
|
|
316
|
+
//#region src/export/export.ts
|
|
317
|
+
/** Open a package for editing, with the baseline an export will need. */
|
|
318
|
+
function openPackage(bytes, options = {}) {
|
|
319
|
+
return {
|
|
320
|
+
store: PartStore.open(bytes, options),
|
|
321
|
+
baseline: PartStore.open(bytes, options),
|
|
322
|
+
baselineBytes: bytes
|
|
323
|
+
};
|
|
324
|
+
}
|
|
325
|
+
function exportPackage(options) {
|
|
326
|
+
const store = options.store;
|
|
327
|
+
const baseline = options.baseline ?? null;
|
|
328
|
+
const prepared = runPrepare(options.prepare ?? [], store, baseline);
|
|
329
|
+
const collection = planCollection({
|
|
330
|
+
store,
|
|
331
|
+
baseline,
|
|
332
|
+
...options.collect === void 0 ? {} : { policy: options.collect },
|
|
333
|
+
...options.sweepable === void 0 ? {} : { sweepable: options.sweepable }
|
|
334
|
+
});
|
|
335
|
+
store.materializeRelationships();
|
|
336
|
+
const rewritten = store.rewrittenParts();
|
|
337
|
+
const streamed = store.partNames.length - rewritten.length;
|
|
338
|
+
const bytes = store.write({
|
|
339
|
+
...options.normalizeEntryOrder === void 0 ? {} : { normalizeEntryOrder: options.normalizeEntryOrder },
|
|
340
|
+
...options.deflateLevel === void 0 ? {} : { deflateLevel: options.deflateLevel }
|
|
341
|
+
});
|
|
342
|
+
const preservation = options.verifyPreservation === false ? {
|
|
343
|
+
checked: 0,
|
|
344
|
+
rewritten: rewritten.length,
|
|
345
|
+
skipped: "turned off by the caller."
|
|
346
|
+
} : assertPreserved(bytes, options.baselineBytes, rewrittenKeys(store, rewritten), options.zip);
|
|
347
|
+
return {
|
|
348
|
+
bytes,
|
|
349
|
+
report: options.validate === false ? null : assertValid({
|
|
350
|
+
store,
|
|
351
|
+
bytes,
|
|
352
|
+
...baseline === null ? {} : { baseline },
|
|
353
|
+
...options.baselineBytes === void 0 ? {} : { baselineBytes: options.baselineBytes },
|
|
354
|
+
...options.rules === void 0 ? {} : { rules: options.rules },
|
|
355
|
+
...options.zip === void 0 ? {} : { zip: options.zip }
|
|
356
|
+
}),
|
|
357
|
+
prepared,
|
|
358
|
+
collection,
|
|
359
|
+
rewritten,
|
|
360
|
+
streamed,
|
|
361
|
+
preservation
|
|
362
|
+
};
|
|
363
|
+
}
|
|
364
|
+
/**
|
|
365
|
+
* The parts the preservation check must not compare, in its own key space.
|
|
366
|
+
*
|
|
367
|
+
* `[Content_Types].xml` is in the set whenever the map changed, and only then:
|
|
368
|
+
* it is passed through untouched when nothing altered it, which is what lets a
|
|
369
|
+
* deck that spells it `[content_types].xml` round-trip its original bytes under
|
|
370
|
+
* the canonical entry name.
|
|
371
|
+
*/
|
|
372
|
+
function rewrittenKeys(store, rewritten) {
|
|
373
|
+
const keys = new Set(rewritten.map((name) => normalizePartName(name)));
|
|
374
|
+
if (store.contentTypes.dirty) keys.add(CONTENT_TYPES_PART);
|
|
375
|
+
return keys;
|
|
376
|
+
}
|
|
377
|
+
//#endregion
|
|
378
|
+
//#region src/oracle/bisect.ts
|
|
379
|
+
/** Every change in the forest, parents before children. */
|
|
380
|
+
function flattenChanges(changes) {
|
|
381
|
+
const out = [];
|
|
382
|
+
const stack = [...changes].reverse();
|
|
383
|
+
while (stack.length > 0) {
|
|
384
|
+
const change = stack.pop();
|
|
385
|
+
out.push(change);
|
|
386
|
+
for (let i = change.children.length - 1; i >= 0; i -= 1) stack.push(change.children[i]);
|
|
387
|
+
}
|
|
388
|
+
return out;
|
|
389
|
+
}
|
|
390
|
+
/** The leaves: the finest changes this delta can express. */
|
|
391
|
+
function leafChanges(changes) {
|
|
392
|
+
return flattenChanges(changes).filter((change) => change.children.length === 0);
|
|
393
|
+
}
|
|
394
|
+
function index(archive) {
|
|
395
|
+
const map = /* @__PURE__ */ new Map();
|
|
396
|
+
for (const entry of archive.entries) {
|
|
397
|
+
if (entry.isDirectory) continue;
|
|
398
|
+
map.set(entry.name, {
|
|
399
|
+
archive,
|
|
400
|
+
entry
|
|
401
|
+
});
|
|
402
|
+
}
|
|
403
|
+
return map;
|
|
404
|
+
}
|
|
405
|
+
function equalBytes(a, b) {
|
|
406
|
+
if (a.length !== b.length) return false;
|
|
407
|
+
for (let i = 0; i < a.length; i += 1) if (a[i] !== b[i]) return false;
|
|
408
|
+
return true;
|
|
409
|
+
}
|
|
410
|
+
/**
|
|
411
|
+
* The package with exactly this subset of the changes applied to the original.
|
|
412
|
+
*
|
|
413
|
+
* Entries no change touches are passed through still compressed rather than
|
|
414
|
+
* re-deflated - the same trick `PartStore.write` uses, and it matters more
|
|
415
|
+
* here, because this runs once per oracle call rather than once per export.
|
|
416
|
+
*/
|
|
417
|
+
function buildPackage(context, selected) {
|
|
418
|
+
const applied = /* @__PURE__ */ new Map();
|
|
419
|
+
for (const [name, changes] of context.byEntry) {
|
|
420
|
+
const chosen = changes.filter((change) => selected.has(change.id));
|
|
421
|
+
if (chosen.length > 0) applied.set(name, chosen);
|
|
422
|
+
}
|
|
423
|
+
const entries = [];
|
|
424
|
+
for (const name of context.order) {
|
|
425
|
+
const source = context.original.get(name);
|
|
426
|
+
const chosen = applied.get(name);
|
|
427
|
+
if (chosen === void 0) {
|
|
428
|
+
if (source !== void 0) entries.push(passthroughEntry(source.archive, source.entry));
|
|
429
|
+
continue;
|
|
430
|
+
}
|
|
431
|
+
const whole = chosen.find((change) => change.span === null);
|
|
432
|
+
if (whole !== void 0) {
|
|
433
|
+
if (whole.kind === "entry-removed") continue;
|
|
434
|
+
const target = context.broken.get(name);
|
|
435
|
+
entries.push(passthroughEntry(target.archive, target.entry));
|
|
436
|
+
continue;
|
|
437
|
+
}
|
|
438
|
+
let text = context.text.get(name);
|
|
439
|
+
const ordered = [...chosen].sort((a, b) => b.span.start - a.span.start);
|
|
440
|
+
for (const change of ordered) text = text.slice(0, change.span.start) + change.text + text.slice(change.span.end);
|
|
441
|
+
entries.push(deflatedEntry(name, encodeXmlSource(text)));
|
|
442
|
+
}
|
|
443
|
+
return writeZip(entries);
|
|
444
|
+
}
|
|
445
|
+
function atom(builder, entry, kind, where, depth, span, was, text, children = []) {
|
|
446
|
+
return {
|
|
447
|
+
id: builder.next(),
|
|
448
|
+
entry,
|
|
449
|
+
kind,
|
|
450
|
+
where,
|
|
451
|
+
depth,
|
|
452
|
+
span,
|
|
453
|
+
text,
|
|
454
|
+
was,
|
|
455
|
+
children
|
|
456
|
+
};
|
|
457
|
+
}
|
|
458
|
+
function slice(source, node) {
|
|
459
|
+
return source.slice(node.start, node.end);
|
|
460
|
+
}
|
|
461
|
+
function nameOf(node) {
|
|
462
|
+
switch (node.type) {
|
|
463
|
+
case "element": return node.qname;
|
|
464
|
+
case "text": return "text()";
|
|
465
|
+
case "cdata": return "cdata()";
|
|
466
|
+
case "comment": return "comment()";
|
|
467
|
+
case "processingInstruction": return "processing-instruction()";
|
|
468
|
+
case "declaration": return "xml-declaration()";
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
/** `/p:sld/p:cSld/p:spTree/p:sp[3]` - one-based, and indexed only where it must be. */
|
|
472
|
+
function pathTo(parent, siblings, at) {
|
|
473
|
+
const name = nameOf(siblings[at]);
|
|
474
|
+
let position = 1;
|
|
475
|
+
let total = 0;
|
|
476
|
+
for (let i = 0; i < siblings.length; i += 1) {
|
|
477
|
+
if (nameOf(siblings[i]) !== name) continue;
|
|
478
|
+
total += 1;
|
|
479
|
+
if (i < at) position += 1;
|
|
480
|
+
}
|
|
481
|
+
return parent + "/" + name + (total > 1 ? "[" + String(position) + "]" : "");
|
|
482
|
+
}
|
|
483
|
+
/**
|
|
484
|
+
* Attributes paired by position, or null when they do not line up.
|
|
485
|
+
*
|
|
486
|
+
* Null whenever anything outside the attributes themselves differs - the
|
|
487
|
+
* element name, the number or order of attributes, the whitespace before `/>`.
|
|
488
|
+
* Otherwise the attribute spans would not cover the whole difference, and
|
|
489
|
+
* reverting every one of them would not reproduce the original tag, which is
|
|
490
|
+
* the single property the descent is allowed to assume.
|
|
491
|
+
*/
|
|
492
|
+
function pairAttributes(a, b) {
|
|
493
|
+
if (a.attributes.length === 0 || a.attributes.length !== b.attributes.length) return null;
|
|
494
|
+
if (a.nameEnd - a.start !== b.nameEnd - b.start) return null;
|
|
495
|
+
if (a.attributes[0].start - a.start !== b.attributes[0].start - b.start) return null;
|
|
496
|
+
if (a.attributes.at(-1).end !== a.trailingSpaceStart) return null;
|
|
497
|
+
if (b.attributes.at(-1).end !== b.trailingSpaceStart) return null;
|
|
498
|
+
if (a.openTagEnd - a.trailingSpaceStart !== b.openTagEnd - b.trailingSpaceStart) return null;
|
|
499
|
+
const pairs = [];
|
|
500
|
+
for (let i = 0; i < a.attributes.length; i += 1) {
|
|
501
|
+
const left = a.attributes[i];
|
|
502
|
+
const right = b.attributes[i];
|
|
503
|
+
if (left.qname !== right.qname) return null;
|
|
504
|
+
pairs.push([left, right]);
|
|
505
|
+
}
|
|
506
|
+
return pairs;
|
|
507
|
+
}
|
|
508
|
+
/** The start tag, decomposed into attributes when the two tags line up. */
|
|
509
|
+
function tagChange(builder, entry, where, depth, a, b, ao, bo) {
|
|
510
|
+
const children = [];
|
|
511
|
+
const pairs = pairAttributes(a, b);
|
|
512
|
+
if (pairs !== null) for (const [left, right] of pairs) {
|
|
513
|
+
const was = ao.slice(left.start, left.end);
|
|
514
|
+
const text = bo.slice(right.start, right.end);
|
|
515
|
+
if (was === text) continue;
|
|
516
|
+
children.push(atom(builder, entry, "attribute", where + "/@" + left.qname, depth + 1, {
|
|
517
|
+
start: left.start,
|
|
518
|
+
end: left.end
|
|
519
|
+
}, was, text));
|
|
520
|
+
}
|
|
521
|
+
return atom(builder, entry, "tag", where, depth, {
|
|
522
|
+
start: a.start,
|
|
523
|
+
end: a.openTagEnd
|
|
524
|
+
}, ao.slice(a.start, a.openTagEnd), bo.slice(b.start, b.openTagEnd), children);
|
|
525
|
+
}
|
|
526
|
+
/**
|
|
527
|
+
* Line up two lists of children and describe the difference between them.
|
|
528
|
+
*
|
|
529
|
+
* The two lists are matched from both ends first, and only what is left in the
|
|
530
|
+
* middle is treated as changed. That is what makes an insertion or a deletion
|
|
531
|
+
* nameable: without it, any change to the *number* of children shifts every
|
|
532
|
+
* position after it, nothing pairs, and the answer degrades to "something in
|
|
533
|
+
* this element". The case that forced it is the headline one - a missing
|
|
534
|
+
* `<Default Extension="fntdata"/>` reduces `<Types>` from seventeen children to
|
|
535
|
+
* sixteen, and positional pairing reported the whole content-type stream.
|
|
536
|
+
*
|
|
537
|
+
* When the middles are the same length they pair off one to one, which is the
|
|
538
|
+
* ordinary edit-in-place case. When they are not, the whole middle is one
|
|
539
|
+
* change: a run of children replaced by another run. That is coarse if a part
|
|
540
|
+
* had several separate insertions, and it is exact in the common case of one -
|
|
541
|
+
* which is what a writer produces.
|
|
542
|
+
*/
|
|
543
|
+
function pairChildren(builder, entry, where, depth, a, b, ao, bo) {
|
|
544
|
+
const left = a.children;
|
|
545
|
+
const right = b.children;
|
|
546
|
+
let lo = 0;
|
|
547
|
+
while (lo < left.length && lo < right.length && slice(ao, left[lo]) === slice(bo, right[lo])) lo += 1;
|
|
548
|
+
let hiA = left.length;
|
|
549
|
+
let hiB = right.length;
|
|
550
|
+
while (hiA > lo && hiB > lo && slice(ao, left[hiA - 1]) === slice(bo, right[hiB - 1])) {
|
|
551
|
+
hiA -= 1;
|
|
552
|
+
hiB -= 1;
|
|
553
|
+
}
|
|
554
|
+
if (hiA === lo && hiB === lo) return [];
|
|
555
|
+
const changes = [];
|
|
556
|
+
if (hiA - lo === hiB - lo) {
|
|
557
|
+
for (let i = lo; i < hiA; i += 1) {
|
|
558
|
+
builder.budget.left -= 1;
|
|
559
|
+
changes.push(pairNodes(builder, entry, pathTo(where, left, i), depth + 1, left[i], right[i], ao, bo));
|
|
560
|
+
}
|
|
561
|
+
return changes;
|
|
562
|
+
}
|
|
563
|
+
const start = lo < hiA ? left[lo].start : left[hiA]?.start ?? a.closeTagStart;
|
|
564
|
+
const end = lo < hiA ? left[hiA - 1].end : start;
|
|
565
|
+
const text = lo < hiB ? bo.slice(right[lo].start, right[hiB - 1].end) : "";
|
|
566
|
+
const at = lo < hiA ? pathTo(where, left, lo) : where;
|
|
567
|
+
builder.budget.left -= 1;
|
|
568
|
+
changes.push(atom(builder, entry, "children", at, depth + 1, {
|
|
569
|
+
start,
|
|
570
|
+
end
|
|
571
|
+
}, ao.slice(start, end), text));
|
|
572
|
+
return changes;
|
|
573
|
+
}
|
|
574
|
+
/** One change turning the original's node into the broken package's node. */
|
|
575
|
+
function pairNodes(builder, entry, where, depth, a, b, ao, bo) {
|
|
576
|
+
const span = {
|
|
577
|
+
start: a.start,
|
|
578
|
+
end: a.end
|
|
579
|
+
};
|
|
580
|
+
const was = slice(ao, a);
|
|
581
|
+
const text = slice(bo, b);
|
|
582
|
+
if (!(a.type === "element" && b.type === "element" && a.qname === b.qname && a.selfClosing === b.selfClosing)) return atom(builder, entry, a.type === "element" ? "element" : "node", where, depth, span, was, text);
|
|
583
|
+
if (builder.budget.left <= 0) {
|
|
584
|
+
builder.truncated.yes = true;
|
|
585
|
+
return atom(builder, entry, "element", where, depth, span, was, text);
|
|
586
|
+
}
|
|
587
|
+
const element = a;
|
|
588
|
+
const other = b;
|
|
589
|
+
const children = [];
|
|
590
|
+
if (ao.slice(element.start, element.openTagEnd) !== bo.slice(other.start, other.openTagEnd)) {
|
|
591
|
+
builder.budget.left -= 1;
|
|
592
|
+
children.push(tagChange(builder, entry, where, depth + 1, element, other, ao, bo));
|
|
593
|
+
}
|
|
594
|
+
children.push(...pairChildren(builder, entry, where, depth, element, other, ao, bo));
|
|
595
|
+
if (children.length === 0) return atom(builder, entry, "element", where, depth, span, was, text);
|
|
596
|
+
return atom(builder, entry, "element", where, depth, span, was, text, children);
|
|
597
|
+
}
|
|
598
|
+
function deltaForEntry(builder, name, originalBytes, brokenBytes, text) {
|
|
599
|
+
const whole = () => atom(builder, name, "entry", "/", 0, null, null, null);
|
|
600
|
+
let a;
|
|
601
|
+
let b;
|
|
602
|
+
try {
|
|
603
|
+
a = parseXml(originalBytes);
|
|
604
|
+
b = parseXml(brokenBytes);
|
|
605
|
+
} catch {
|
|
606
|
+
return whole();
|
|
607
|
+
}
|
|
608
|
+
if (a.children.length !== b.children.length) return whole();
|
|
609
|
+
const children = [];
|
|
610
|
+
for (let i = 0; i < a.children.length; i += 1) {
|
|
611
|
+
const left = a.children[i];
|
|
612
|
+
const right = b.children[i];
|
|
613
|
+
if (slice(a.source, left) === slice(b.source, right)) continue;
|
|
614
|
+
builder.budget.left -= 1;
|
|
615
|
+
children.push(pairNodes(builder, name, pathTo("", a.children, i), 1, left, right, a.source, b.source));
|
|
616
|
+
}
|
|
617
|
+
if (children.length === 0) return whole();
|
|
618
|
+
text.set(name, a.source);
|
|
619
|
+
return atom(builder, name, "entry", "/", 0, null, null, null, children);
|
|
620
|
+
}
|
|
621
|
+
/** The forest of changes between two packages. One root per differing entry. */
|
|
622
|
+
function collectDelta(original, broken, maxChanges = 2e4) {
|
|
623
|
+
let counter = 0;
|
|
624
|
+
const builder = {
|
|
625
|
+
next: () => counter++,
|
|
626
|
+
budget: { left: maxChanges },
|
|
627
|
+
truncated: { yes: false }
|
|
628
|
+
};
|
|
629
|
+
const left = index(original);
|
|
630
|
+
const right = index(broken);
|
|
631
|
+
const text = /* @__PURE__ */ new Map();
|
|
632
|
+
const changes = [];
|
|
633
|
+
for (const [name, source] of left) {
|
|
634
|
+
const target = right.get(name);
|
|
635
|
+
if (target === void 0) {
|
|
636
|
+
changes.push(atom(builder, name, "entry-removed", "/", 0, null, null, null));
|
|
637
|
+
continue;
|
|
638
|
+
}
|
|
639
|
+
const a = source.archive.read(source.entry);
|
|
640
|
+
const b = target.archive.read(target.entry);
|
|
641
|
+
if (equalBytes(a, b)) continue;
|
|
642
|
+
changes.push(deltaForEntry(builder, name, a, b, text));
|
|
643
|
+
}
|
|
644
|
+
for (const name of right.keys()) {
|
|
645
|
+
if (left.has(name)) continue;
|
|
646
|
+
changes.push(atom(builder, name, "entry-added", "/", 0, null, null, null));
|
|
647
|
+
}
|
|
648
|
+
return {
|
|
649
|
+
changes,
|
|
650
|
+
text,
|
|
651
|
+
truncated: builder.truncated.yes
|
|
652
|
+
};
|
|
653
|
+
}
|
|
654
|
+
/** Internal. Never escapes `bisectPackages`; see the `maxRuns` handling there. */
|
|
655
|
+
var Exhausted = class extends Error {};
|
|
656
|
+
/**
|
|
657
|
+
* `ddmin`, reducing a set of changes to a 1-minimal failing subset.
|
|
658
|
+
*
|
|
659
|
+
* 1-minimal means every change left in the answer is load-bearing: drop any one
|
|
660
|
+
* and the package stops failing. It does **not** mean no smaller failing set
|
|
661
|
+
* exists - two changes may only fail together, and neither alone - and the
|
|
662
|
+
* difference is worth keeping straight, because a report claiming a minimality
|
|
663
|
+
* it does not have sends someone looking in the wrong place.
|
|
664
|
+
*/
|
|
665
|
+
function ddmin(items, test, record = () => {}) {
|
|
666
|
+
let current = items;
|
|
667
|
+
let n = 2;
|
|
668
|
+
while (current.length >= 2) {
|
|
669
|
+
const size = Math.ceil(current.length / n);
|
|
670
|
+
const subsets = [];
|
|
671
|
+
for (let i = 0; i < current.length; i += size) subsets.push(current.slice(i, i + size));
|
|
672
|
+
const failing = subsets.find((subset) => test(subset) === "fails");
|
|
673
|
+
if (failing !== void 0) {
|
|
674
|
+
current = failing;
|
|
675
|
+
record(current);
|
|
676
|
+
n = 2;
|
|
677
|
+
continue;
|
|
678
|
+
}
|
|
679
|
+
let dropped = false;
|
|
680
|
+
for (const subset of subsets) {
|
|
681
|
+
const complement = current.filter((change) => !subset.includes(change));
|
|
682
|
+
if (complement.length === 0 || complement.length === current.length) continue;
|
|
683
|
+
if (test(complement) !== "fails") continue;
|
|
684
|
+
current = complement;
|
|
685
|
+
record(current);
|
|
686
|
+
n = Math.max(n - 1, 2);
|
|
687
|
+
dropped = true;
|
|
688
|
+
break;
|
|
689
|
+
}
|
|
690
|
+
if (dropped) continue;
|
|
691
|
+
if (n >= current.length) break;
|
|
692
|
+
n = Math.min(current.length, n * 2);
|
|
693
|
+
}
|
|
694
|
+
return current;
|
|
695
|
+
}
|
|
696
|
+
function bisectPackages(originalBytes, brokenBytes, options) {
|
|
697
|
+
const original = readZip(originalBytes, options.zip ?? {});
|
|
698
|
+
const broken = readZip(brokenBytes, options.zip ?? {});
|
|
699
|
+
const { changes, text, truncated } = collectDelta(original, broken, options.maxChanges);
|
|
700
|
+
const order = [];
|
|
701
|
+
for (const entry of original.entries) if (!entry.isDirectory) order.push(entry.name);
|
|
702
|
+
for (const entry of broken.entries) {
|
|
703
|
+
if (entry.isDirectory || order.includes(entry.name)) continue;
|
|
704
|
+
order.push(entry.name);
|
|
705
|
+
}
|
|
706
|
+
const byEntry = /* @__PURE__ */ new Map();
|
|
707
|
+
for (const change of flattenChanges(changes)) {
|
|
708
|
+
const list = byEntry.get(change.entry);
|
|
709
|
+
if (list === void 0) byEntry.set(change.entry, [change]);
|
|
710
|
+
else list.push(change);
|
|
711
|
+
}
|
|
712
|
+
const context = {
|
|
713
|
+
order,
|
|
714
|
+
original: index(original),
|
|
715
|
+
broken: index(broken),
|
|
716
|
+
text,
|
|
717
|
+
byEntry
|
|
718
|
+
};
|
|
719
|
+
const maxRuns = options.maxRuns ?? 2e3;
|
|
720
|
+
let runs = 0;
|
|
721
|
+
let cached = 0;
|
|
722
|
+
let unresolved = 0;
|
|
723
|
+
let exhausted = false;
|
|
724
|
+
const seen = /* @__PURE__ */ new Map();
|
|
725
|
+
const test = (subset) => {
|
|
726
|
+
const ids = new Set(subset.map((change) => change.id));
|
|
727
|
+
const key = [...ids].sort((a, b) => a - b).join(",");
|
|
728
|
+
const remembered = seen.get(key);
|
|
729
|
+
if (remembered !== void 0) {
|
|
730
|
+
cached += 1;
|
|
731
|
+
return remembered;
|
|
732
|
+
}
|
|
733
|
+
if (runs >= maxRuns) throw new Exhausted();
|
|
734
|
+
runs += 1;
|
|
735
|
+
const label = String(subset.length) + "/" + String(changes.length) + " change(s)";
|
|
736
|
+
options.onRun?.({
|
|
737
|
+
run: runs,
|
|
738
|
+
applied: subset.length,
|
|
739
|
+
total: changes.length,
|
|
740
|
+
label
|
|
741
|
+
});
|
|
742
|
+
const verdict = options.oracle(buildPackage(context, ids), label);
|
|
743
|
+
if (verdict === "unresolved") unresolved += 1;
|
|
744
|
+
seen.set(key, verdict);
|
|
745
|
+
return verdict;
|
|
746
|
+
};
|
|
747
|
+
const done = (outcome, minimal) => ({
|
|
748
|
+
outcome,
|
|
749
|
+
changes,
|
|
750
|
+
minimal,
|
|
751
|
+
bytes: buildPackage(context, new Set(minimal.map((change) => change.id))),
|
|
752
|
+
runs,
|
|
753
|
+
cached,
|
|
754
|
+
unresolved,
|
|
755
|
+
exhausted,
|
|
756
|
+
truncated
|
|
757
|
+
});
|
|
758
|
+
if (changes.length === 0) return done("identical", []);
|
|
759
|
+
let best = changes;
|
|
760
|
+
try {
|
|
761
|
+
if (test([]) === "fails") return done("original-fails", []);
|
|
762
|
+
if (test(changes) !== "fails") return done("broken-passes", changes);
|
|
763
|
+
let level = changes;
|
|
764
|
+
for (;;) {
|
|
765
|
+
best = ddmin(level, test, (current) => {
|
|
766
|
+
best = current;
|
|
767
|
+
});
|
|
768
|
+
const expanded = best.flatMap((change) => change.children.length > 0 ? [...change.children] : [change]);
|
|
769
|
+
if (expanded.length === best.length && expanded.every((c, i) => c === best[i])) break;
|
|
770
|
+
level = expanded;
|
|
771
|
+
}
|
|
772
|
+
} catch (error) {
|
|
773
|
+
if (!(error instanceof Exhausted)) throw error;
|
|
774
|
+
exhausted = true;
|
|
775
|
+
}
|
|
776
|
+
return done("localized", best);
|
|
777
|
+
}
|
|
778
|
+
/** One line naming a change, without its text. */
|
|
779
|
+
function describeChange(change) {
|
|
780
|
+
const where = change.where === "/" ? "" : " " + change.where;
|
|
781
|
+
switch (change.kind) {
|
|
782
|
+
case "entry-added": return "added /" + change.entry;
|
|
783
|
+
case "entry-removed": return "removed /" + change.entry;
|
|
784
|
+
case "entry": return "entry /" + change.entry;
|
|
785
|
+
case "element": return "element /" + change.entry + where;
|
|
786
|
+
case "node": return "node /" + change.entry + where;
|
|
787
|
+
case "children":
|
|
788
|
+
if (change.was === "") return "added /" + change.entry + where;
|
|
789
|
+
if (change.text === "") return "removed /" + change.entry + where;
|
|
790
|
+
return "children /" + change.entry + where;
|
|
791
|
+
case "tag": return "tag /" + change.entry + where;
|
|
792
|
+
case "attribute": return "attr /" + change.entry + where;
|
|
793
|
+
}
|
|
794
|
+
}
|
|
795
|
+
/** `2 change(s) in 1 entry(s), from a delta of 91, in 37 oracle run(s)`. */
|
|
796
|
+
function summarizeBisect(result) {
|
|
797
|
+
const entries = new Set(result.minimal.map((change) => change.entry));
|
|
798
|
+
return String(result.minimal.length) + " change(s) in " + String(entries.size) + " entry(s), from a delta of " + String(flattenChanges(result.changes).length) + ", in " + String(result.runs) + " oracle run(s)";
|
|
799
|
+
}
|
|
800
|
+
//#endregion
|
|
801
|
+
//#region src/oracle/roundtrip.ts
|
|
802
|
+
/** The relationships namespace: every attribute in it is a relationship id. */
|
|
803
|
+
const R = NS.r;
|
|
804
|
+
/**
|
|
805
|
+
* What a relationship points at, as a string that does not mention its id.
|
|
806
|
+
*
|
|
807
|
+
* An internal target resolves to an absolute part name, so that `../media/a.png`
|
|
808
|
+
* from a slide and `/ppt/media/a.png` written absolutely are one target. An
|
|
809
|
+
* external one is compared verbatim: it is a URI, we do not own it, and
|
|
810
|
+
* normalising somebody's `http://Example.COM/` would be inventing a rule.
|
|
811
|
+
*/
|
|
812
|
+
function targetKey(rels, rel) {
|
|
813
|
+
if (rel.targetMode === "External") return "External " + rel.target;
|
|
814
|
+
try {
|
|
815
|
+
return "Internal " + normalizePartName(rels.resolve(rel));
|
|
816
|
+
} catch {
|
|
817
|
+
return "Internal? " + rel.target;
|
|
818
|
+
}
|
|
819
|
+
}
|
|
820
|
+
function relationshipKey(rels, rel) {
|
|
821
|
+
return rel.type + " -> " + targetKey(rels, rel);
|
|
822
|
+
}
|
|
823
|
+
/**
|
|
824
|
+
* Match two relationship collections by what they point at.
|
|
825
|
+
*
|
|
826
|
+
* Duplicates are legal - a part may relate to the same target twice under two
|
|
827
|
+
* ids, and `a03-fills` in the corpus has eight fills on one image - so this
|
|
828
|
+
* matches within a key group by document order rather than assuming one
|
|
829
|
+
* relationship per key. Order inside a group is arbitrary in principle; it is
|
|
830
|
+
* the only tiebreak available, and picking the wrong pairing among identical
|
|
831
|
+
* relationships cannot change the answer, because the two ids resolve to the
|
|
832
|
+
* same place.
|
|
833
|
+
*/
|
|
834
|
+
function matchGraphs(source, before, after) {
|
|
835
|
+
const groupsBefore = /* @__PURE__ */ new Map();
|
|
836
|
+
const groupsAfter = /* @__PURE__ */ new Map();
|
|
837
|
+
const collect = (rels, into) => {
|
|
838
|
+
const keys = [];
|
|
839
|
+
for (const rel of rels.all) {
|
|
840
|
+
const key = relationshipKey(rels, rel);
|
|
841
|
+
keys.push(key);
|
|
842
|
+
const bucket = into.get(key);
|
|
843
|
+
if (bucket === void 0) into.set(key, [rel]);
|
|
844
|
+
else bucket.push(rel);
|
|
845
|
+
}
|
|
846
|
+
return keys.sort();
|
|
847
|
+
};
|
|
848
|
+
const signature = collect(before, groupsBefore).join("\n");
|
|
849
|
+
const writtenSignature = collect(after, groupsAfter).join("\n");
|
|
850
|
+
const relabel = /* @__PURE__ */ new Map();
|
|
851
|
+
const relabelled = [];
|
|
852
|
+
const differences = [];
|
|
853
|
+
for (const [key, mine] of groupsBefore) {
|
|
854
|
+
const theirs = groupsAfter.get(key) ?? [];
|
|
855
|
+
const shared = Math.min(mine.length, theirs.length);
|
|
856
|
+
for (let i = 0; i < shared; i++) {
|
|
857
|
+
const from = mine[i].id;
|
|
858
|
+
const to = theirs[i].id;
|
|
859
|
+
relabel.set(from, to);
|
|
860
|
+
if (from !== to) relabelled.push({
|
|
861
|
+
source,
|
|
862
|
+
from,
|
|
863
|
+
to,
|
|
864
|
+
target: key
|
|
865
|
+
});
|
|
866
|
+
}
|
|
867
|
+
for (let i = shared; i < mine.length; i++) differences.push({
|
|
868
|
+
kind: "relationship",
|
|
869
|
+
part: source,
|
|
870
|
+
detail: "the relationship " + mine[i].id + " (" + key + ") is in the original and not in what was written"
|
|
871
|
+
});
|
|
872
|
+
}
|
|
873
|
+
for (const [key, theirs] of groupsAfter) {
|
|
874
|
+
const mine = groupsBefore.get(key) ?? [];
|
|
875
|
+
for (let i = mine.length; i < theirs.length; i++) differences.push({
|
|
876
|
+
kind: "relationship",
|
|
877
|
+
part: source,
|
|
878
|
+
detail: "the relationship " + theirs[i].id + " (" + key + ") was written and is not in the original"
|
|
879
|
+
});
|
|
880
|
+
}
|
|
881
|
+
return {
|
|
882
|
+
relabel,
|
|
883
|
+
relabelled,
|
|
884
|
+
differences,
|
|
885
|
+
signature,
|
|
886
|
+
writtenSignature
|
|
887
|
+
};
|
|
888
|
+
}
|
|
889
|
+
/**
|
|
890
|
+
* Every part that could own relationships, in either package, plus the root.
|
|
891
|
+
*
|
|
892
|
+
* Every part, and not only the ones with a `.rels` in `partNames` - which is
|
|
893
|
+
* the shorter, faster and wrong version of this function, and the same trap
|
|
894
|
+
* `reachability.ts` documents. A relationship collection created in this
|
|
895
|
+
* session exists as a parsed object until `PartStore.write` materialises it, so
|
|
896
|
+
* keying on existing relationship parts makes exactly the edges an editing
|
|
897
|
+
* session added invisible. `PartStore.relationships` returns an empty
|
|
898
|
+
* collection for a part that has none, so asking every part costs a lookup and
|
|
899
|
+
* removes the whole class of miss.
|
|
900
|
+
*/
|
|
901
|
+
function relationshipSources(original, written) {
|
|
902
|
+
const sources = /* @__PURE__ */ new Set(["/"]);
|
|
903
|
+
for (const store of [original, written]) for (const partName of store.partNames) {
|
|
904
|
+
if (!isRelationshipPartName(partName)) {
|
|
905
|
+
sources.add(partName);
|
|
906
|
+
continue;
|
|
907
|
+
}
|
|
908
|
+
try {
|
|
909
|
+
sources.add(sourcePartNameForRels(partName));
|
|
910
|
+
} catch {}
|
|
911
|
+
}
|
|
912
|
+
return [...sources].sort();
|
|
913
|
+
}
|
|
914
|
+
function readRelationships(store, source) {
|
|
915
|
+
try {
|
|
916
|
+
return store.relationships(source);
|
|
917
|
+
} catch (error) {
|
|
918
|
+
return error instanceof Error ? error : new Error(String(error));
|
|
919
|
+
}
|
|
920
|
+
}
|
|
921
|
+
/** The content types are the same type if they differ only in case or spacing. */
|
|
922
|
+
function sameContentType(a, b) {
|
|
923
|
+
return (a ?? "").trim().toLowerCase() === (b ?? "").trim().toLowerCase();
|
|
924
|
+
}
|
|
925
|
+
/**
|
|
926
|
+
* Compare two packages.
|
|
927
|
+
*
|
|
928
|
+
* Never throws. Every failure - a part that will not parse, a `.rels` that will
|
|
929
|
+
* not read - is a difference with a reason, because the caller of a comparison
|
|
930
|
+
* wants the whole list and not the first item on it.
|
|
931
|
+
*/
|
|
932
|
+
function comparePackages(original, written, options = {}) {
|
|
933
|
+
const differences = [];
|
|
934
|
+
const parts = [];
|
|
935
|
+
const relabelled = [];
|
|
936
|
+
const relabelBySource = /* @__PURE__ */ new Map();
|
|
937
|
+
const graphs = /* @__PURE__ */ new Map();
|
|
938
|
+
for (const source of relationshipSources(original, written)) {
|
|
939
|
+
const before = readRelationships(original, source);
|
|
940
|
+
const after = readRelationships(written, source);
|
|
941
|
+
if (before instanceof Error || after instanceof Error) {
|
|
942
|
+
const which = before instanceof Error ? "the original" : "what was written";
|
|
943
|
+
const error = before instanceof Error ? before : after;
|
|
944
|
+
differences.push({
|
|
945
|
+
kind: "unreadable",
|
|
946
|
+
part: source,
|
|
947
|
+
detail: "the relationships of " + which + " would not parse: " + error.message
|
|
948
|
+
});
|
|
949
|
+
continue;
|
|
950
|
+
}
|
|
951
|
+
const match = matchGraphs(source, before, after);
|
|
952
|
+
graphs.set(normalizePartName(source), match);
|
|
953
|
+
relabelBySource.set(normalizePartName(source), match.relabel);
|
|
954
|
+
differences.push(...match.differences);
|
|
955
|
+
relabelled.push(...match.relabelled);
|
|
956
|
+
}
|
|
957
|
+
const names = /* @__PURE__ */ new Map();
|
|
958
|
+
for (const store of [original, written]) for (const partName of store.partNames) names.set(normalizePartName(partName), partName);
|
|
959
|
+
let xmlCount = 0;
|
|
960
|
+
let binaryCount = 0;
|
|
961
|
+
let relsCount = 0;
|
|
962
|
+
for (const key of [...names.keys()].sort()) {
|
|
963
|
+
const part = names.get(key);
|
|
964
|
+
const inOriginal = original.has(part);
|
|
965
|
+
const inWritten = written.has(part);
|
|
966
|
+
if (!inOriginal || !inWritten) {
|
|
967
|
+
differences.push({
|
|
968
|
+
kind: inOriginal ? "part-removed" : "part-added",
|
|
969
|
+
part,
|
|
970
|
+
detail: inOriginal ? "the original has this part and what was written does not" : "this part was written and the original does not have it"
|
|
971
|
+
});
|
|
972
|
+
continue;
|
|
973
|
+
}
|
|
974
|
+
const typeBefore = original.contentTypeOf(part);
|
|
975
|
+
const typeAfter = written.contentTypeOf(part);
|
|
976
|
+
if (!sameContentType(typeBefore, typeAfter)) differences.push({
|
|
977
|
+
kind: "content-type",
|
|
978
|
+
part,
|
|
979
|
+
detail: "the content type was " + (typeBefore ?? "(none)") + " and is now " + (typeAfter ?? "(none)")
|
|
980
|
+
});
|
|
981
|
+
if (isRelationshipPartName(part)) {
|
|
982
|
+
relsCount++;
|
|
983
|
+
parts.push(relationshipComparison(part, graphs));
|
|
984
|
+
continue;
|
|
985
|
+
}
|
|
986
|
+
if (isXmlContentType(typeBefore) && isXmlContentType(typeAfter)) {
|
|
987
|
+
xmlCount++;
|
|
988
|
+
parts.push(compareXmlPart(part, original, written, relabelBySource, differences, options.limits));
|
|
989
|
+
continue;
|
|
990
|
+
}
|
|
991
|
+
binaryCount++;
|
|
992
|
+
parts.push(compareBinaryPart(part, original, written, differences));
|
|
993
|
+
}
|
|
994
|
+
const same = parts.filter((entry) => entry.same).length;
|
|
995
|
+
return {
|
|
996
|
+
ok: differences.length === 0,
|
|
997
|
+
parts,
|
|
998
|
+
differences,
|
|
999
|
+
relabelled,
|
|
1000
|
+
counts: {
|
|
1001
|
+
xml: xmlCount,
|
|
1002
|
+
binary: binaryCount,
|
|
1003
|
+
relationships: relsCount,
|
|
1004
|
+
same
|
|
1005
|
+
}
|
|
1006
|
+
};
|
|
1007
|
+
}
|
|
1008
|
+
/**
|
|
1009
|
+
* A relationship part's verdict, taken from the graph pass rather than from its
|
|
1010
|
+
* bytes.
|
|
1011
|
+
*
|
|
1012
|
+
* The digest is over the graph with the ids omitted, so it is the same on both
|
|
1013
|
+
* sides of a renumbering. That is the whole claim of this comparison stated as
|
|
1014
|
+
* a single value: two `.rels` with one digest describe one package.
|
|
1015
|
+
*/
|
|
1016
|
+
function relationshipComparison(part, graphs) {
|
|
1017
|
+
let source;
|
|
1018
|
+
try {
|
|
1019
|
+
source = sourcePartNameForRels(part);
|
|
1020
|
+
} catch {
|
|
1021
|
+
return {
|
|
1022
|
+
part,
|
|
1023
|
+
how: "relationships",
|
|
1024
|
+
same: true,
|
|
1025
|
+
digest: "",
|
|
1026
|
+
writtenDigest: ""
|
|
1027
|
+
};
|
|
1028
|
+
}
|
|
1029
|
+
const match = graphs.get(normalizePartName(source));
|
|
1030
|
+
if (match === void 0) return {
|
|
1031
|
+
part,
|
|
1032
|
+
how: "relationships",
|
|
1033
|
+
same: false,
|
|
1034
|
+
digest: "",
|
|
1035
|
+
writtenDigest: ""
|
|
1036
|
+
};
|
|
1037
|
+
const digest = sha256HexOfText(match.signature);
|
|
1038
|
+
const writtenDigest = sha256HexOfText(match.writtenSignature);
|
|
1039
|
+
return {
|
|
1040
|
+
part,
|
|
1041
|
+
how: "relationships",
|
|
1042
|
+
same: digest === writtenDigest,
|
|
1043
|
+
digest,
|
|
1044
|
+
writtenDigest
|
|
1045
|
+
};
|
|
1046
|
+
}
|
|
1047
|
+
function compareXmlPart(part, original, written, relabelBySource, differences, limits) {
|
|
1048
|
+
const relabel = relabelBySource.get(normalizePartName(part));
|
|
1049
|
+
const rewriteAttributeValue = relabel === void 0 || relabel.size === 0 ? void 0 : (value, namespaceUri) => namespaceUri === R ? relabel.get(value) ?? value : value;
|
|
1050
|
+
const canonical = (store, which, relabelThis) => {
|
|
1051
|
+
try {
|
|
1052
|
+
const document = parseXml(store.read(part), limits);
|
|
1053
|
+
return canonicalXml(document, relabelThis && rewriteAttributeValue !== void 0 ? { rewriteAttributeValue } : {});
|
|
1054
|
+
} catch (error) {
|
|
1055
|
+
return {
|
|
1056
|
+
kind: "unreadable",
|
|
1057
|
+
part,
|
|
1058
|
+
detail: which + " would not parse as XML: " + (isXmlError(error) ? error.code + " - " + error.message : String(error))
|
|
1059
|
+
};
|
|
1060
|
+
}
|
|
1061
|
+
};
|
|
1062
|
+
const before = canonical(original, "the original", true);
|
|
1063
|
+
const after = canonical(written, "what was written", false);
|
|
1064
|
+
if (typeof before !== "string" || typeof after !== "string") {
|
|
1065
|
+
if (typeof before !== "string") differences.push(before);
|
|
1066
|
+
if (typeof after !== "string") differences.push(after);
|
|
1067
|
+
return {
|
|
1068
|
+
part,
|
|
1069
|
+
how: "xml",
|
|
1070
|
+
same: false,
|
|
1071
|
+
digest: "",
|
|
1072
|
+
writtenDigest: ""
|
|
1073
|
+
};
|
|
1074
|
+
}
|
|
1075
|
+
const digest = sha256HexOfText(before);
|
|
1076
|
+
const writtenDigest = sha256HexOfText(after);
|
|
1077
|
+
if (digest !== writtenDigest) {
|
|
1078
|
+
const at = firstDifference(before, after);
|
|
1079
|
+
differences.push({
|
|
1080
|
+
kind: "xml",
|
|
1081
|
+
part,
|
|
1082
|
+
detail: "the canonical XML differs at character " + String(at) + "\n original: " + excerpt(before, at) + "\n written: " + excerpt(after, at)
|
|
1083
|
+
});
|
|
1084
|
+
}
|
|
1085
|
+
return {
|
|
1086
|
+
part,
|
|
1087
|
+
how: "xml",
|
|
1088
|
+
same: digest === writtenDigest,
|
|
1089
|
+
digest,
|
|
1090
|
+
writtenDigest
|
|
1091
|
+
};
|
|
1092
|
+
}
|
|
1093
|
+
function compareBinaryPart(part, original, written, differences) {
|
|
1094
|
+
let digest;
|
|
1095
|
+
let writtenDigest;
|
|
1096
|
+
try {
|
|
1097
|
+
digest = sha256Hex(original.read(part));
|
|
1098
|
+
writtenDigest = sha256Hex(written.read(part));
|
|
1099
|
+
} catch (error) {
|
|
1100
|
+
differences.push({
|
|
1101
|
+
kind: "unreadable",
|
|
1102
|
+
part,
|
|
1103
|
+
detail: "the part could not be read: " + (error instanceof Error ? error.message : "")
|
|
1104
|
+
});
|
|
1105
|
+
return {
|
|
1106
|
+
part,
|
|
1107
|
+
how: "binary",
|
|
1108
|
+
same: false,
|
|
1109
|
+
digest: "",
|
|
1110
|
+
writtenDigest: ""
|
|
1111
|
+
};
|
|
1112
|
+
}
|
|
1113
|
+
if (digest !== writtenDigest) differences.push({
|
|
1114
|
+
kind: "binary",
|
|
1115
|
+
part,
|
|
1116
|
+
detail: "the bytes differ: sha256 was " + digest.slice(0, 12) + ", is now " + writtenDigest.slice(0, 12) + " (" + String(original.read(part).length) + " bytes then, " + String(written.read(part).length) + " now)"
|
|
1117
|
+
});
|
|
1118
|
+
return {
|
|
1119
|
+
part,
|
|
1120
|
+
how: "binary",
|
|
1121
|
+
same: digest === writtenDigest,
|
|
1122
|
+
digest,
|
|
1123
|
+
writtenDigest
|
|
1124
|
+
};
|
|
1125
|
+
}
|
|
1126
|
+
/** A one-line summary, for a CLI and for a test failure message. */
|
|
1127
|
+
function summarizeRoundTrip(report) {
|
|
1128
|
+
const { xml, binary, relationships, same } = report.counts;
|
|
1129
|
+
const total = xml + binary + relationships;
|
|
1130
|
+
return String(same) + "/" + String(total) + " parts identical (" + String(xml) + " xml, " + String(relationships) + " rels, " + String(binary) + " binary)" + (report.relabelled.length === 0 ? "" : ", " + String(report.relabelled.length) + " relationship id(s) renamed") + (report.ok ? "" : ", " + String(report.differences.length) + " difference(s)");
|
|
1131
|
+
}
|
|
1132
|
+
/**
|
|
1133
|
+
* Read a package, write it back, and compare the two.
|
|
1134
|
+
*
|
|
1135
|
+
* The whole of sub-phase 1.4 in one call, and the shape `cli roundtrip` and the
|
|
1136
|
+
* corpus gate both want. Every option `exportPackage` takes is accepted, so a
|
|
1137
|
+
* caller can round-trip through a different deflate level or a normalised entry
|
|
1138
|
+
* order - which is the point of a comparison that is not byte equality, and
|
|
1139
|
+
* worth having a test do deliberately.
|
|
1140
|
+
*
|
|
1141
|
+
* Throws whatever `exportPackage` throws. A package the firewall refuses is not
|
|
1142
|
+
* a package with a round-trip difference; it is one that never got written, and
|
|
1143
|
+
* flattening the two into one verdict would lose the distinction exactly where
|
|
1144
|
+
* it matters.
|
|
1145
|
+
*/
|
|
1146
|
+
function roundTripPackage(bytes, options = {}) {
|
|
1147
|
+
const { limits, ...exportOptions } = options;
|
|
1148
|
+
const opened = openPackage(bytes, options.zip ?? {});
|
|
1149
|
+
const exported = exportPackage({
|
|
1150
|
+
...exportOptions,
|
|
1151
|
+
...opened
|
|
1152
|
+
});
|
|
1153
|
+
const written = PartStore.open(exported.bytes, options.zip ?? {});
|
|
1154
|
+
const comparison = comparePackages(opened.baseline, written, limits === void 0 ? {} : { limits });
|
|
1155
|
+
return {
|
|
1156
|
+
exported,
|
|
1157
|
+
written,
|
|
1158
|
+
comparison,
|
|
1159
|
+
ok: comparison.ok
|
|
1160
|
+
};
|
|
1161
|
+
}
|
|
1162
|
+
//#endregion
|
|
1163
|
+
export { WRITER_ERROR_CODES, WriterError, assertPreserved, bisectPackages, collectDelta, collectGarbage, comparePackages, describeChange, exportPackage, flattenChanges, isMediaPart, isWriterError, leafChanges, openPackage, orphanedParts, planCollection, reachableParts, roundTripPackage, runPrepare, summarizeBisect, summarizeRoundTrip };
|
|
1164
|
+
|
|
1165
|
+
//# sourceMappingURL=index.js.map
|