single-file-core 1.5.124 → 1.5.126
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/core/helper.js +3 -0
- package/core/index.js +5 -0
- package/core/lib/doctype.js +0 -3
- package/core/lib/processor-helper-common.js +22 -11
- package/core/lib/processor-helper-inline.js +4 -0
- package/core/lib/processor-helper.js +4 -0
- package/core/util.js +11 -16
- package/doc/singlefile-archive.md +117 -40
- package/modules/html-images-alt-minifier.js +4 -4
- package/package.json +2 -2
- package/processors/compression/compression.js +10 -8
- package/processors/frame-tree/content/content-frame-tree.js +54 -17
- package/processors/hooks/content/content-hooks-frames-web.js +51 -34
- package/test/sfz-harness/README.md +2 -1
- package/test/sfz-harness/charset-round-trip.js +161 -0
- package/test/sfz-harness/format-rules.js +60 -1
package/core/helper.js
CHANGED
|
@@ -293,6 +293,9 @@ function preProcessDoc(doc, win, options) {
|
|
|
293
293
|
}
|
|
294
294
|
|
|
295
295
|
function markInvalidNesting(doc) {
|
|
296
|
+
if (!doc.body) {
|
|
297
|
+
return;
|
|
298
|
+
}
|
|
296
299
|
addTrackIds(doc.body);
|
|
297
300
|
const verificationDoc = parseDocContent(serialize(doc));
|
|
298
301
|
const markedMap = buildTrackIdMap(doc.body);
|
package/core/index.js
CHANGED
|
@@ -1361,6 +1361,11 @@ class Processor {
|
|
|
1361
1361
|
}
|
|
1362
1362
|
}
|
|
1363
1363
|
}));
|
|
1364
|
+
frameElements.forEach(frameElement => {
|
|
1365
|
+
if (!frameElement.getAttribute("src") && !frameElement.getAttribute("srcdoc") && !frameElement.getAttribute("data")) {
|
|
1366
|
+
frameElement.removeAttribute("sandbox");
|
|
1367
|
+
}
|
|
1368
|
+
});
|
|
1364
1369
|
|
|
1365
1370
|
async function initializeProcessor(frameData, frameElement, frameWindowId, batchRequest, options) {
|
|
1366
1371
|
options.insertSingleFileComment = false;
|
package/core/lib/doctype.js
CHANGED
|
@@ -38,9 +38,6 @@ function getDoctypeString(doc) {
|
|
|
38
38
|
} else if (docType.systemId) {
|
|
39
39
|
docTypeString += " SYSTEM \"" + docType.systemId + "\"";
|
|
40
40
|
}
|
|
41
|
-
if (docType.internalSubset) {
|
|
42
|
-
docTypeString += " [" + docType.internalSubset + "]";
|
|
43
|
-
}
|
|
44
41
|
docTypeString += ">";
|
|
45
42
|
}
|
|
46
43
|
return docTypeString;
|
|
@@ -243,9 +243,16 @@ class ProcessorHelperCommon {
|
|
|
243
243
|
return serializeSrcset([Object.assign({}, srcsetValue, { url: resourceURL })]);
|
|
244
244
|
}
|
|
245
245
|
}));
|
|
246
|
-
|
|
246
|
+
const newSrcset = srcsetValues.filter(srcsetValue => srcsetValue).join(", ");
|
|
247
|
+
if (newSrcset) {
|
|
248
|
+
resourceElement.setAttribute("srcset", newSrcset);
|
|
249
|
+
} else {
|
|
250
|
+
resourceElement.removeAttribute("srcset");
|
|
251
|
+
resourceElement.removeAttribute("sizes");
|
|
252
|
+
}
|
|
247
253
|
} else {
|
|
248
|
-
resourceElement.
|
|
254
|
+
resourceElement.removeAttribute("srcset");
|
|
255
|
+
resourceElement.removeAttribute("sizes");
|
|
249
256
|
}
|
|
250
257
|
}));
|
|
251
258
|
}
|
|
@@ -267,6 +274,7 @@ class ProcessorHelperCommon {
|
|
|
267
274
|
element.style.setProperty("background-size", style && style["background-size"] ? style["background-size"] : "100% 100%", "important");
|
|
268
275
|
element.style.setProperty("background-origin", "content-box", "important");
|
|
269
276
|
element.style.setProperty("background-repeat", "no-repeat", "important");
|
|
277
|
+
element.style.setProperty("background-attachment", "scroll", "important");
|
|
270
278
|
}
|
|
271
279
|
|
|
272
280
|
async getStylesheetContent(resourceURL, options) {
|
|
@@ -405,7 +413,7 @@ class ProcessorHelperCommon {
|
|
|
405
413
|
}
|
|
406
414
|
sheetIndex++;
|
|
407
415
|
});
|
|
408
|
-
processFontDetails(fontsDetails);
|
|
416
|
+
processFontDetails(fontsDetails, fonts);
|
|
409
417
|
await Promise.all([...stylesheets].map(async ([, stylesheetInfo], sheetIndex) => {
|
|
410
418
|
if (stylesheetInfo.stylesheet) {
|
|
411
419
|
const cssRules = stylesheetInfo.stylesheet.children;
|
|
@@ -443,13 +451,10 @@ class ProcessorHelperCommon {
|
|
|
443
451
|
} else if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
|
|
444
452
|
const key = this.getFontKey(ruleData);
|
|
445
453
|
const fontInfo = fontsDetails.fonts.get(key);
|
|
446
|
-
if (fontInfo) {
|
|
447
|
-
const
|
|
448
|
-
if (
|
|
454
|
+
if (fontInfo && fontsDetails.lastRules.get(key) == ruleData) {
|
|
455
|
+
const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
|
|
456
|
+
if (!keptRule) {
|
|
449
457
|
removedRules.push(cssRule);
|
|
450
|
-
} else {
|
|
451
|
-
fontsDetails.emittedFonts.add(ruleKey);
|
|
452
|
-
await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
|
|
453
458
|
}
|
|
454
459
|
} else {
|
|
455
460
|
removedRules.push(cssRule);
|
|
@@ -487,16 +492,22 @@ class ProcessorHelperCommon {
|
|
|
487
492
|
fontInfo = [];
|
|
488
493
|
mediaFontsDetails.fonts.set(fontKey, fontInfo);
|
|
489
494
|
}
|
|
495
|
+
mediaFontsDetails.lastRules.set(fontKey, ruleData);
|
|
490
496
|
const src = this.getPropertyValue(ruleData, "src");
|
|
491
497
|
if (src) {
|
|
492
498
|
const fontSources = src.match(REGEXP_URL_FUNCTION);
|
|
493
499
|
if (fontSources) {
|
|
500
|
+
const ruleSources = [];
|
|
494
501
|
fontSources.forEach(source => {
|
|
495
502
|
if (fontInfo.includes(source)) {
|
|
496
503
|
fontInfo.splice(fontInfo.indexOf(source), 1);
|
|
497
504
|
}
|
|
498
|
-
|
|
505
|
+
if (ruleSources.includes(source)) {
|
|
506
|
+
ruleSources.splice(ruleSources.indexOf(source), 1);
|
|
507
|
+
}
|
|
508
|
+
ruleSources.unshift(source);
|
|
499
509
|
});
|
|
510
|
+
ruleSources.forEach(source => fontInfo.push(source));
|
|
500
511
|
}
|
|
501
512
|
}
|
|
502
513
|
}
|
|
@@ -509,7 +520,7 @@ class ProcessorHelperCommon {
|
|
|
509
520
|
medias: new Map(),
|
|
510
521
|
supports: new Map(),
|
|
511
522
|
layers: new Map(),
|
|
512
|
-
|
|
523
|
+
lastRules: new Map()
|
|
513
524
|
};
|
|
514
525
|
}
|
|
515
526
|
|
|
@@ -648,6 +648,9 @@ function getProcessorHelperClass(utilInstance) {
|
|
|
648
648
|
}
|
|
649
649
|
}
|
|
650
650
|
stats.fonts.discarded -= fontInfo.length;
|
|
651
|
+
if (!fontInfo.length) {
|
|
652
|
+
return false;
|
|
653
|
+
}
|
|
651
654
|
fontInfo.reverse();
|
|
652
655
|
try {
|
|
653
656
|
srcDeclaration.data.value = cssTree.parse(fontInfo.map(fontSource => fontSource.src).join(","), { context: "value", parseCustomProperty: true });
|
|
@@ -656,6 +659,7 @@ function getProcessorHelperClass(utilInstance) {
|
|
|
656
659
|
// ignored
|
|
657
660
|
}
|
|
658
661
|
}
|
|
662
|
+
return true;
|
|
659
663
|
}
|
|
660
664
|
};
|
|
661
665
|
}
|
|
@@ -574,6 +574,9 @@ function getProcessorHelperClass(utilInstance) {
|
|
|
574
574
|
removedNodes.forEach(node => ruleData.block.children.remove(node));
|
|
575
575
|
const srcDeclaration = ruleData.block.children.filter(node => node.property == "src").tail;
|
|
576
576
|
if (srcDeclaration) {
|
|
577
|
+
if (!fontInfo.length) {
|
|
578
|
+
return false;
|
|
579
|
+
}
|
|
577
580
|
fontInfo.reverse();
|
|
578
581
|
try {
|
|
579
582
|
srcDeclaration.data.value = cssTree.parse(fontInfo.map(fontSource => fontSource.src).join(","), { context: "value", parseCustomProperty: true });
|
|
@@ -582,6 +585,7 @@ function getProcessorHelperClass(utilInstance) {
|
|
|
582
585
|
// ignored
|
|
583
586
|
}
|
|
584
587
|
}
|
|
588
|
+
return true;
|
|
585
589
|
}
|
|
586
590
|
};
|
|
587
591
|
}
|
package/core/util.js
CHANGED
|
@@ -69,11 +69,6 @@ const EXPECTED_TYPES_MEDIA = ["font", "image", "video", "audio"];
|
|
|
69
69
|
const URL = globalThis.URL;
|
|
70
70
|
const DOMParser = globalThis.DOMParser;
|
|
71
71
|
const Blob = globalThis.Blob;
|
|
72
|
-
const fetch = (url, options) => {
|
|
73
|
-
options.cache = "force-cache";
|
|
74
|
-
options.referrerPolicy = "strict-origin-when-cross-origin";
|
|
75
|
-
return globalThis.fetch(url, options);
|
|
76
|
-
};
|
|
77
72
|
const TextDecoder = globalThis.TextDecoder;
|
|
78
73
|
const URLSearchParams = globalThis.URLSearchParams;
|
|
79
74
|
|
|
@@ -83,8 +78,8 @@ export {
|
|
|
83
78
|
|
|
84
79
|
function getInstance(utilOptions) {
|
|
85
80
|
utilOptions = utilOptions || {};
|
|
86
|
-
utilOptions.fetch = utilOptions.fetch || fetch;
|
|
87
|
-
utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch
|
|
81
|
+
utilOptions.fetch = utilOptions.fetch || ((url, options) => globalThis.fetch(url, { ...options, cache: "force-cache", referrerPolicy: "strict-origin-when-cross-origin" }));
|
|
82
|
+
utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch;
|
|
88
83
|
return {
|
|
89
84
|
getDoctypeString,
|
|
90
85
|
getFilenameExtension(resourceURL, replacedCharacters, replacementCharacter, replacementCharacters) {
|
|
@@ -240,7 +235,7 @@ function getInstance(utilOptions) {
|
|
|
240
235
|
startTime = Date.now();
|
|
241
236
|
log(" // STARTED download url =", resourceURL, "asBinary =", options.asBinary);
|
|
242
237
|
}
|
|
243
|
-
if (options.blockMixedContent && /^https:/i.test(options.baseURI) && !/^https:/i.test(resourceURL)) {
|
|
238
|
+
if (options.blockMixedContent && /^https:/i.test(options.baseURI) && !/^https:/i.test(resourceURL) && !/^blob:https:/i.test(resourceURL)) {
|
|
244
239
|
return getFetchResponse(resourceURL, options);
|
|
245
240
|
}
|
|
246
241
|
if (options.networkTimeout) {
|
|
@@ -264,7 +259,7 @@ function getInstance(utilOptions) {
|
|
|
264
259
|
// eslint-disable-next-line no-unused-vars
|
|
265
260
|
} catch (error) {
|
|
266
261
|
response = await Promise.race([
|
|
267
|
-
fetchResource(resourceURL, { headers: { accept } }),
|
|
262
|
+
fetchResource(resourceURL, { referrer: options.resourceReferrer, headers: { accept } }),
|
|
268
263
|
networkTimeoutPromise
|
|
269
264
|
]);
|
|
270
265
|
}
|
|
@@ -390,13 +385,13 @@ function guessMIMEType(expectedType, buffer) {
|
|
|
390
385
|
if (compareBytes([255, 255, 255, 255], [0, 0, 2, 0])) {
|
|
391
386
|
return "image/x-icon";
|
|
392
387
|
}
|
|
393
|
-
if (compareBytes([255, 255], [
|
|
388
|
+
if (compareBytes([255, 255], [66, 77])) {
|
|
394
389
|
return "image/bmp";
|
|
395
390
|
}
|
|
396
391
|
if (compareBytes([255, 255, 255, 255, 255, 255], [71, 73, 70, 56, 57, 97])) {
|
|
397
392
|
return "image/gif";
|
|
398
393
|
}
|
|
399
|
-
if (compareBytes([255, 255, 255, 255, 255, 255], [71, 73, 70, 56,
|
|
394
|
+
if (compareBytes([255, 255, 255, 255, 255, 255], [71, 73, 70, 56, 55, 97])) {
|
|
400
395
|
return "image/gif";
|
|
401
396
|
}
|
|
402
397
|
if (compareBytes([255, 255, 255, 255, 0, 0, 0, 0, 255, 255, 255, 255, 255, 255], [82, 73, 70, 70, 0, 0, 0, 0, 87, 69, 66, 80, 86, 80])) {
|
|
@@ -434,7 +429,7 @@ function guessMIMEType(expectedType, buffer) {
|
|
|
434
429
|
if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 105, 115, 111, 109])) {
|
|
435
430
|
return "video/mp4";
|
|
436
431
|
}
|
|
437
|
-
if (compareBytes([255, 255, 255, 255, 0, 0, 0, 0, 255, 255, 255, 255], [82, 73, 70, 70, 0, 0, 0, 0,
|
|
432
|
+
if (compareBytes([255, 255, 255, 255, 0, 0, 0, 0, 255, 255, 255, 255], [82, 73, 70, 70, 0, 0, 0, 0, 65, 86, 73, 32])) {
|
|
438
433
|
return "video/x-msvideo";
|
|
439
434
|
}
|
|
440
435
|
if (compareBytes([255, 255, 255, 255], [0, 0, 1, 179]) || compareBytes([255, 255, 255, 255], [0, 0, 1, 186])) {
|
|
@@ -443,18 +438,18 @@ function guessMIMEType(expectedType, buffer) {
|
|
|
443
438
|
if (compareBytes([255, 255, 255, 255], [79, 103, 103, 83])) {
|
|
444
439
|
return "video/ogg";
|
|
445
440
|
}
|
|
446
|
-
if (compareBytes([255], [71])) {
|
|
447
|
-
return "video/mp2t";
|
|
448
|
-
}
|
|
449
441
|
if (compareBytes([255, 255, 255, 255], [26, 69, 223, 163])) {
|
|
450
442
|
return "video/webm";
|
|
451
443
|
}
|
|
452
444
|
if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 51, 103])) {
|
|
453
445
|
return "video/3gpp";
|
|
454
446
|
}
|
|
447
|
+
if (compareBytes([255], [71])) {
|
|
448
|
+
return "video/mp2t";
|
|
449
|
+
}
|
|
455
450
|
}
|
|
456
451
|
if (expectedType == "audio") {
|
|
457
|
-
if (compareBytes([255, 255], [255, 249]) || compareBytes([255, 255], [255, 254])) {
|
|
452
|
+
if (compareBytes([255, 255], [255, 241]) || compareBytes([255, 255], [255, 249]) || compareBytes([255, 255], [255, 254])) {
|
|
458
453
|
return "audio/aac";
|
|
459
454
|
}
|
|
460
455
|
if (compareBytes([255, 255, 255, 255], [77, 84, 104, 100])) {
|
|
@@ -136,7 +136,7 @@ Three consequences shape everything below:
|
|
|
136
136
|
| **region** | A byte range with a single producer, named in §3. Regions are the units the rest of this document reasons about; a region can appear in several pieces — `html-prologue` resumes after the embedded PDF document in the PDF variants, and after the `tEXt "ZIP"` chunk header in the PNG ones, so with all four faces it comes in three. |
|
|
137
137
|
| **universal mode** | The variant whose HTML face can extract the archive from the *parsed page text*, the text and comment nodes the HTML parser produced, and therefore needs no access to its own raw bytes. Named "universal" because it works from any location, including the `file:` protocol. |
|
|
138
138
|
| **wrapper tag** | The HTML construct that hides a binary region from the HTML parser, `<!--`…`-->` by default (§5.1). |
|
|
139
|
-
| **appended data** | Bytes after the ZIP End Of Central Directory record.
|
|
139
|
+
| **appended data** | Bytes after the ZIP End Of Central Directory record. A reader tolerates them as far back as its EOCD scan reaches, and how far that is varies by an order of magnitude: 65557 bytes from the end of the file for Python `zipfile` (the 22-byte record plus the 65535-byte maximum comment length), but 16383 for libarchive and 32768 for perl `Archive::Zip` (§8.1). No reader's window is guaranteed, so a writer keeps its own narrower budget (§5.2). The format's one hard limit is the 65535-byte comment field, and it binds only a run the writer declares (§4.2). It may be left undeclared or declared as the archive comment; both forms are valid ZIP and readers MUST accept both (§4.2). The recovery payload can be computed before that choice is made because it stops two bytes short of the record, excluding its comment-length field (see *recovered range* below). |
|
|
140
140
|
| **ZIP region** | The contiguous byte range holding the archive proper: from the first local file header the ZIP writer emitted through the last byte of the End Of Central Directory record. It spans the `zip-entries`, `pdf-central-record` (when present) and `central-directory · eocd` blocks of §3, and in the HTML variants it is the content of the last wrapper, exactly so on the element rungs and preceded by the `sfz-data` identifier on the comment rung, which the extractor steps over. It does **not** include `pdf-local-header` or the PDF document, which sit earlier in the file. |
|
|
141
141
|
| **archive** | The *logical* ZIP file: the set of entries the central directory describes, wherever their bytes lie. This is distinct from the ZIP region above, which is a contiguous byte range. Every entry but one has its bytes inside the region; `page.pdf` is the deliberate exception, an entry of the archive whose local header and data sit before the region (§4.2). "Archive" in this document always means the logical file, "ZIP region" always the byte range, and the two differ only in the PDF-with-HTML variants. |
|
|
142
142
|
| **recovered range** | What the universal extractor reproduces (§4.5): the ZIP region minus its last two bytes, the comment-length field of the End Of Central Directory record. That field is the one part of the record whose value depends on what follows the region, so leaving it out is what lets a writer decide the appended-data form after the recovery payload is final (§4.2). The extractor supplies the two bytes itself, as zeroes — the recovered range carries no comment. |
|
|
@@ -174,8 +174,9 @@ every acquisition path including the ones that read raw bytes and could return i
|
|
|
174
174
|
HTML face, so the option does not apply. The *Specimen* column names the measured
|
|
175
175
|
reference files this document cites; §8 records how to regenerate them.
|
|
176
176
|
|
|
177
|
-
Other writer options shape the file without adding a face: `preventAppendedData
|
|
178
|
-
`declareAppendedData` (§4.2, §5.2), `includeBOM` (§3.1),
|
|
177
|
+
Other writer options shape the file without adding a face: `preventAppendedData`,
|
|
178
|
+
`declareAppendedData` and `maxAppendedDataLength` (§4.2, §5.2), `includeBOM` (§3.1),
|
|
179
|
+
`insertTextBody` (§4.6),
|
|
179
180
|
`password` (§5.6), `createRootDirectory` (§7.1), and the head-element switches
|
|
180
181
|
`insertCanonicalLink`, `insertMetaNoIndex` and `insertMetaCSP` (§3.1).
|
|
181
182
|
|
|
@@ -328,7 +329,8 @@ the same way: only a universal file carries an `<sfz-extra-data>` element.
|
|
|
328
329
|
|
|
329
330
|
Unless a row states otherwise, the layouts below are measured from specimen files
|
|
330
331
|
saved from `example.com` (the generation commands are in §8). The relocated row covers
|
|
331
|
-
two cases with one layout, `preventAppendedData` and a payload over
|
|
332
|
+
two cases with one layout, `preventAppendedData` and a payload over the appended-data
|
|
333
|
+
budget (§5.2): the first is
|
|
332
334
|
measured on the relocated specimen, the second derived from the writer rules, because
|
|
333
335
|
such a payload requires an archive too large for a readable specimen. The figure below shows
|
|
334
336
|
the regions and their order; the glossary of §3.1 is the normative list, and it states
|
|
@@ -353,7 +355,7 @@ face adds, then the regions the PNG face adds.
|
|
|
353
355
|
| `<!--` / `-->` | HTML | HTML face | The wrapper tag pair hiding a binary region from the HTML parser — comment tags by default, another pair when the hidden bytes defeat them — which `-->` is only the commonest way to do, the full test being `<!--`, `--!>`, a trailing `<!-` and, for the PNG payload, a leading `>` or `->` (§5.1). Drawn at each opening and closing position. The close tag is absent whenever the recovery payload is relocated (§5.2): under `preventAppendedData`, when the payload outgrows the appended-data budget, or on the `<plaintext>` wrapper which cannot close. No markup then follows the archive and the wrapper runs to end-of-file. That does not mean the file ends at the EOCD — the PNG face's tail still follows, inside the wrapper, where it parses as text (§5.1). |
|
|
354
356
|
| `zip-entries` | ZIP | always | The archive's local file headers and entry data, written by the ZIP writer. The central directory of an archive written by the reference writer lists `index.html` (the page) first, then `manifest.json` (a JSON description of the archive: original URL, title, save time, resource-to-URL map — informative; the page displays without it), then the resources; the *physical* order of the local headers inside the region is not guaranteed to match, and readers MUST NOT rely on either order — entries are addressed by name (§7.1). |
|
|
355
357
|
| `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
|
|
356
|
-
| `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the
|
|
358
|
+
| `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the appended-data budget (§5.2) or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
|
|
357
359
|
| `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
|
|
358
360
|
| `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag set as on every other entry — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
|
|
359
361
|
| `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
|
|
@@ -366,7 +368,7 @@ face adds, then the regions the PNG face adds.
|
|
|
366
368
|
| `crc · IEND` | PNG | PNG face | The `tEXt "ZIP"` chunk's CRC, computed once the archive bytes are final (§6), followed by the empty `IEND` chunk — the last bytes of the file (PNG requires `IEND` to end the stream, which is why the PNG variants drop the end tags). |
|
|
367
369
|
|
|
368
370
|
The reader-by-reader interpretation of these regions is §4; the mechanics that keep
|
|
369
|
-
them from colliding (wrapper-tag selection, checksums, offsets, the
|
|
371
|
+
them from colliding (wrapper-tag selection, checksums, offsets, the appended-data budget) are
|
|
370
372
|
§5.
|
|
371
373
|
|
|
372
374
|
## 4. Reader lenses
|
|
@@ -447,10 +449,12 @@ path the plain variant's error message describes.
|
|
|
447
449
|
### 4.2 The ZIP reader
|
|
448
450
|
|
|
449
451
|
The ZIP face is read from the end. A reader locates the End Of Central Directory
|
|
450
|
-
record by scanning backward from end-of-file
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
452
|
+
record by scanning backward from end-of-file, and how far back it scans is the one
|
|
453
|
+
reader property the format cannot assume (§1.3, *appended data*). Everything the
|
|
454
|
+
writer emits after the record — wrapper close tag, extra-data, end tags, PNG tail —
|
|
455
|
+
fits the appended-data budget of §5.2, and the reference writer sizes that budget to
|
|
456
|
+
the narrowest scan measured in §8.1, so the record stays reachable for every reader
|
|
457
|
+
listed there. Accepting *undeclared* bytes
|
|
454
458
|
in that window is itself a customary tolerance (§1.1): the ZIP specification
|
|
455
459
|
documents the comment, not trailing junk. From the EOCD
|
|
456
460
|
the reader jumps to the central directory and reads only what it references;
|
|
@@ -510,6 +514,14 @@ readers accept at all: `java.util.zip`, and therefore Android and most JVM tooli
|
|
|
510
514
|
rejects an archive with undeclared trailing bytes outright (§8.1). A writer SHOULD
|
|
511
515
|
offer both and default to raw.
|
|
512
516
|
|
|
517
|
+
The declared form carries a ceiling the raw form does not. The comment length is a
|
|
518
|
+
16-bit field, so a run longer than 65535 bytes cannot be declared at all. A writer
|
|
519
|
+
whose appended-data budget (§5.2) is raised past that ceiling MUST leave such a run
|
|
520
|
+
undeclared rather than write its length back modulo 65536, and readers that accept
|
|
521
|
+
only the declared form then reject the file with no diagnostic. The budget and the
|
|
522
|
+
ceiling are two separate limits, and a writer that exposes the first as an option
|
|
523
|
+
SHOULD say so where it documents it.
|
|
524
|
+
|
|
513
525
|
Neither form constrains the other faces, and universal mode supports both, because
|
|
514
526
|
the recovery payload describes the recovered range rather than the whole region: the
|
|
515
527
|
comment-length field is excluded (§1.3), so its value can be decided after the
|
|
@@ -735,6 +747,23 @@ zip64 locator states an absolute offset in the original file, so it needs the sh
|
|
|
735
747
|
this formula produces and cannot be used to find it. A reader that instead uses the
|
|
736
748
|
sentinels arithmetically gets a shift in the billions, with no diagnostic.
|
|
737
749
|
|
|
750
|
+
Which leaves the record itself to be located without the offset that normally points at
|
|
751
|
+
it. Scan backward from the locator for the `PK\x06\x06` signature and confirm each
|
|
752
|
+
candidate against the record's own size field, the 8 bytes at `p + 4`, which by
|
|
753
|
+
definition excludes the leading 12:
|
|
754
|
+
|
|
755
|
+
```
|
|
756
|
+
p + 12 + size == locatorPosition
|
|
757
|
+
```
|
|
758
|
+
|
|
759
|
+
A well-formed archive puts the record immediately before the locator, where it is 56
|
|
760
|
+
bytes long if it carries no extensible data sector, so `locatorPosition - 56` is worth
|
|
761
|
+
testing before scanning at all. The confirmation matters on the archives that miss:
|
|
762
|
+
past the record the scan walks back through the central directory, whose file names and
|
|
763
|
+
extra fields are arbitrary bytes, and past that through entry data, and a four-byte
|
|
764
|
+
signature turns up in bytes nothing constrains. The test settles each candidate against
|
|
765
|
+
the record's own field, so it needs no offset it does not already have.
|
|
766
|
+
|
|
738
767
|
### 4.6 Text tools
|
|
739
768
|
|
|
740
769
|
The optional text body (`insertTextBody`) addresses one more consumer: software that
|
|
@@ -972,36 +1001,54 @@ limit of what the CDATA rung buys here — its value is that real payloads rarel
|
|
|
972
1001
|
above states a MUST rather than a quality-of-implementation preference — exhaustion is
|
|
973
1002
|
reachable by construction, not only by a payload built to provoke it.
|
|
974
1003
|
|
|
975
|
-
The
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
1004
|
+
The selection test above — both patterns, on every rung the writer considers — MUST
|
|
1005
|
+
be applied to the payload's bytes in their final form, for every payload the writer
|
|
1006
|
+
hides and in every variant that hides one. *Final* is the whole of the requirement:
|
|
1007
|
+
bytes the writer has yet to settle have not been tested. Every rejection restarts the build (§6):
|
|
1008
|
+
the wrapper choice changes the bytes preceding the archive, so the archive must be
|
|
1009
|
+
rewritten at its new position.
|
|
1010
|
+
|
|
1011
|
+
Two fields are patched after that test, and each needs one of its own. The EOCD
|
|
1012
|
+
comment-length field sits at the end of the ZIP region and is patched under the
|
|
1013
|
+
declared form (§6.1, step 11); the writer tests the bytes around it again with the
|
|
1014
|
+
final value in place and keeps the raw form when that value would complete a pattern,
|
|
1015
|
+
since the raw form is always valid. The `tEXt "ZIP"` length field sits inside the pixel-data wrapper,
|
|
1016
|
+
with the fixed `tEXt` type and `ZIP` keyword after it, and is written last (step 12).
|
|
1017
|
+
The header is tested with the rest of the payload, the length as zeros, which cannot
|
|
1018
|
+
join a pattern; the real length is big-endian, so a pattern byte in it would have to be
|
|
1019
|
+
the most significant byte of the chunk's size, and the smallest byte any pattern
|
|
1020
|
+
contains, `-` at 0x2D, puts that size at 0x2D000000 bytes, about 755 MB. The writer
|
|
1021
|
+
refuses to build a self-extracting PNG variant whose chunk reaches that size rather than
|
|
991
1022
|
re-check the field.
|
|
992
1023
|
|
|
993
1024
|
### 5.2 The appended-data budget
|
|
994
1025
|
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
the
|
|
1026
|
+
The run the writer emits after the EOCD record has two limits, and only one of them
|
|
1027
|
+
comes from the format. A run *declared* as the archive comment MUST fit in 65535
|
|
1028
|
+
bytes, the largest value a comment-length field can hold (§4.2). A run left *raw* has
|
|
1029
|
+
no format limit at all: the bytes are outside the archive, and nothing in ZIP bounds
|
|
1030
|
+
them. What bounds both in practice is the reader. Locating the EOCD record means
|
|
1031
|
+
scanning backward from end-of-file, and the searches measured in §8.1 stop at 16383
|
|
1032
|
+
bytes for libarchive, 32768 for perl `Archive::Zip` and 65557 for Python `zipfile`, so
|
|
1033
|
+
a run sized to the comment ceiling is already invisible to the narrowest of them. A
|
|
1034
|
+
writer therefore keeps a *budget*, sized to the readers it means to satisfy rather
|
|
1035
|
+
than to the format. The appended run is:
|
|
998
1036
|
|
|
999
1037
|
```
|
|
1000
1038
|
wrapper close tag + extra-data element + end tags + (PNG face: 4-byte chunk CRC + 12-byte IEND)
|
|
1001
1039
|
```
|
|
1002
1040
|
|
|
1003
|
-
and the writer compares its total against
|
|
1004
|
-
record's own 22 bytes sit inside
|
|
1041
|
+
and the writer compares its total against that budget before committing to it. The
|
|
1042
|
+
EOCD record's own 22 bytes sit inside a reader's window as well, which is what turns a
|
|
1043
|
+
65535-byte run into the 65557 bytes of §1.3 and the reference writer's budget into
|
|
1044
|
+
libarchive's 16383.
|
|
1045
|
+
|
|
1046
|
+
The reference writer exposes the budget as `maxAppendedDataLength` and defaults it to
|
|
1047
|
+
16361 bytes: libarchive's window less the 22 bytes of the record, which is the largest
|
|
1048
|
+
run behind which every reader of §8.1 still finds the record. A writer MAY choose
|
|
1049
|
+
another value. Raising it above 65535 leaves the run undeclarable: it is emitted, and
|
|
1050
|
+
it is still valid ZIP, but no comment length can cover it, and §4.2 says what that
|
|
1051
|
+
costs.
|
|
1005
1052
|
|
|
1006
1053
|
Only the extra-data element can outgrow the budget: it carries one 2-bit code per
|
|
1007
1054
|
newline sequence in the recovered range — the ZIP region without its comment-length
|
|
@@ -1009,9 +1056,9 @@ field (§4.5), CR LF counting once, for two bytes (§5.5) — so it
|
|
|
1009
1056
|
grows with the archive. Newline bytes
|
|
1010
1057
|
occur at their natural density in compressed and STOREd binary data — about two in
|
|
1011
1058
|
every 256 bytes — and the codes are compressed and base64-encoded, which measures at
|
|
1012
|
-
one byte of element per 650 bytes of archive at scale (§8). The budget is
|
|
1013
|
-
exhausted at an archive of roughly
|
|
1014
|
-
practice. That ratio is the large-archive limit and must not be used to size a
|
|
1059
|
+
one byte of element per 650 bytes of archive at scale (§8). The default budget is
|
|
1060
|
+
therefore exhausted at an archive of roughly 10 MB, and the 65535-byte ceiling at
|
|
1061
|
+
roughly 40 MB, so the relocated placement is uncommon in practice. That ratio is the large-archive limit and must not be used to size a
|
|
1015
1062
|
particular file: deflate's overhead is a fixed cost spread over a growing payload, so
|
|
1016
1063
|
small archives are far less efficient. Measured on exact byte counts, a 6099-byte region
|
|
1017
1064
|
needs 69 bytes of element — a ratio of 88 — and a 74057-byte region needs 189, a ratio
|
|
@@ -1032,6 +1079,14 @@ trailing bytes open it (§8.1). The parser closes the open
|
|
|
1032
1079
|
comment or element at end of file, and `</body></html>` are implied, so the page
|
|
1033
1080
|
renders the same.
|
|
1034
1081
|
|
|
1082
|
+
Relocation is not a move at constant size, and it can end either way. Two effects pull
|
|
1083
|
+
against each other: the room the writer sets aside, which in the reference writer is
|
|
1084
|
+
`Math.ceil(length * 1.01) + 32` bytes — a percentage of the payload plus a constant, so
|
|
1085
|
+
the constant dominates a small payload and the percentage a large one — and the 17 bytes
|
|
1086
|
+
of wrapper terminator and end tags it stops emitting. Measured on three files the net ran
|
|
1087
|
+
from 9 bytes saved to 190 bytes spent, so a writer sizing a file should quote that range
|
|
1088
|
+
rather than a single figure.
|
|
1089
|
+
|
|
1035
1090
|
### 5.3 Offset bookkeeping
|
|
1036
1091
|
|
|
1037
1092
|
Three coordinate systems coexist in one file, and the format's job is to keep each
|
|
@@ -1043,6 +1098,15 @@ self-consistent:
|
|
|
1043
1098
|
file positions (§4.2), so a reader of the *whole file* never needs prepended-data
|
|
1044
1099
|
compensation — the repair by which a reader recomputes offsets that disagree with
|
|
1045
1100
|
the file size. A reader of the recovered ZIP region alone does need it (§4.5).
|
|
1101
|
+
|
|
1102
|
+
The alternative, offsets relative to the start of the region, is not a compatibility
|
|
1103
|
+
problem in itself: a reader that compensates arrives at the same entries, and 7-Zip
|
|
1104
|
+
opens such a file when told the type. What absolute offsets buy is the step before
|
|
1105
|
+
that. The file is a valid archive read as it stands, so it survives format
|
|
1106
|
+
auto-detection — 7-Zip reports a base of 0 and a physical size covering the whole
|
|
1107
|
+
file — and the compensation is confined to the one path that cannot avoid it,
|
|
1108
|
+
universal-mode recovery. Nothing in the format depends on the choice; a writer using
|
|
1109
|
+
the other form produces files this document's readers still open.
|
|
1046
1110
|
- **PDF offsets are header-relative.** The document's own cross-reference offsets are
|
|
1047
1111
|
interpreted from the `%PDF-` header, so embedding it needs no rewriting; the writer
|
|
1048
1112
|
only MUST keep the header inside the scan window (§4.3).
|
|
@@ -1439,7 +1503,7 @@ terminate because each of them advances a monotone quantity:
|
|
|
1439
1503
|
There is no converse of the second: a pass that reserved room never discards it,
|
|
1440
1504
|
even when the relocated payload would have fit the appended window. Relocation moves
|
|
1441
1505
|
the archive, which changes the offsets, which changes the payload that made the
|
|
1442
|
-
relocation necessary, so a payload lying on the
|
|
1506
|
+
relocation necessary, so a payload lying on the budget boundary can be too large
|
|
1443
1507
|
appended and small enough relocated, and a writer that dropped the reservation could
|
|
1444
1508
|
rebuild the two placements forever. Relocation is therefore final (§5.2), and the file
|
|
1445
1509
|
keeps at most the reservation's own margin of dead padding.
|
|
@@ -1594,7 +1658,7 @@ only if it affects the bytes the page is built from:
|
|
|
1594
1658
|
| An entry's CRC-32 or AES authentication code does not match | **SHOULD** fail for that entry, and MUST NOT present a page rebuilt from it as intact |
|
|
1595
1659
|
| `page.pdf` was reconstructed from the parsed page and its CRC-32 does not match | **MUST** discard the reconstruction (§4.5). The bytes are a guess about newlines the recovery payload does not describe, and the checksum is the only thing that tests it — unlike the row above, there is no read to have gone wrong, only an inference |
|
|
1596
1660
|
| Bytes outside the archive proper — before the first local file header, after the EOCD record, or between an entry's data and the next header | **MUST** tolerate: they are the other faces (§7.1). The gap in the middle is not hypothetical: with the PDF face the bootstrap lies between `page.pdf`'s data and the ZIP region |
|
|
1597
|
-
| The appended run exceeds the 65535-byte
|
|
1661
|
+
| The appended run exceeds the 65535-byte ceiling (§5.2) | Not a reader's problem: if the EOCD record was found, the archive is readable. Readers MAY warn |
|
|
1598
1662
|
| A `tEXt` chunk CRC does not match, or a chunk holds bytes PNG does not permit (§4.4) | Irrelevant to extraction; a reader of the archive MAY ignore both |
|
|
1599
1663
|
| `page.pdf` is present but its data does not begin with `%PDF-` | Not an error. The entry is data like any other |
|
|
1600
1664
|
| `index.html` is present without `manifest.json` | **MUST** still extract (§7.1) |
|
|
@@ -1634,11 +1698,20 @@ and is class C.
|
|
|
1634
1698
|
| Info-ZIP `unzip`, `zipinfo` | ✔ | ✔ | ✔ | Lists and extracts every variant. AES entries are skipped — `need PK compat. v5.1 (can do v4.5)` — a limitation of the tool, not of the file; `page.pdf` still extracts because it is never encrypted |
|
|
1635
1699
|
| Python `zipfile` | ✔ | ✔ | ✔ | Lists and extracts every variant |
|
|
1636
1700
|
| 7-Zip (`7zz`) | ✔ | ✔ | ✔ | Lists and extracts every variant, AES included |
|
|
1637
|
-
| libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant |
|
|
1701
|
+
| libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant. Its EOCD scan is the narrowest measured, so a class-C file whose appended run pushes the record past 16383 bytes from the end is rejected with `Unrecognized archive format`; §5.2's default budget is sized to this window, and the ✔ holds for files that respect it |
|
|
1638
1702
|
| libarchive `bsdtar`, piped input | ✔ | ✘ | ✘ | `Unrecognized archive format` — the forward-only case of §1.2, measured |
|
|
1639
1703
|
| Java `java.util.zip` (`jar tf`) | ✔ | ✔ | ✘ | `zip END header not found` whenever bytes follow the EOCD undeclared. Declaring them as the archive comment makes the same file open, measured on every class-C variant (§4.2) |
|
|
1640
1704
|
| macOS `ditto -x -k` | ✔ | ✘ | ✘ | `Couldn't read PKZip signature` — requires a local file header at offset 0, so prepended data alone defeats it |
|
|
1641
1705
|
|
|
1706
|
+
The backward scans behind the class-C column differ by an order of magnitude, and they
|
|
1707
|
+
are what §5.2's budget is sized against. Measured by padding a working archive until
|
|
1708
|
+
the record fell out of reach, the largest distance from end-of-file at which each
|
|
1709
|
+
reader still finds the EOCD record is: libarchive 16383, perl `Archive::Zip` 32768,
|
|
1710
|
+
Python `zipfile` 65557, zip.js 65536, Info-ZIP `unzip` 68000, and 7-Zip beyond 1 MiB,
|
|
1711
|
+
which scans the whole file. libarchive binds, and its window less the 22-byte record
|
|
1712
|
+
is the 16361-byte default budget of §5.2. macOS `ditto` is not on this axis at all: it
|
|
1713
|
+
requires a local file header at offset 0 whatever the tail looks like.
|
|
1714
|
+
|
|
1642
1715
|
The cost of the declared form was measured on the same tools: it is a display cost, not
|
|
1643
1716
|
a compatibility one. An archive whose trailing bytes are declared as the comment has
|
|
1644
1717
|
them printed back on ordinary listings; `unzip -l` reproduces the whole run — in
|
|
@@ -1713,7 +1786,7 @@ specimen without a network.
|
|
|
1713
1786
|
| 123006 | `<sfz-extra-data>` … `</sfz-extra-data>` | recovery payload, appended placement (§5.2); 24 base64 characters for this archive |
|
|
1714
1787
|
| 123063 | `</body></html>` | end tags; end of file at 123077 |
|
|
1715
1788
|
|
|
1716
|
-
The appended run is 74 bytes, well inside the
|
|
1789
|
+
The appended run is 74 bytes, well inside the 16361-byte default budget (§5.2). The ZIP region
|
|
1717
1790
|
is the 998 bytes from 122005 to 123003; the universal extractor reproduces the first 996
|
|
1718
1791
|
of them and supplies the last two itself (§1.3).
|
|
1719
1792
|
|
|
@@ -1747,7 +1820,7 @@ These specimens are deliberately small, and a reader tested only against them is
|
|
|
1747
1820
|
undertested: they are all flat archives of two or three entries. None
|
|
1748
1821
|
exercises a root directory, `frames/<n>/` nesting, a second `index.html`, a `data:`-URL
|
|
1749
1822
|
entry comment, the optional text body (§4.6), a UTF-8 BOM, zip64
|
|
1750
|
-
(§5.7), a payload past the
|
|
1823
|
+
(§5.7), a payload past the appended-data budget, or a relocated reservation with padding left
|
|
1751
1824
|
in it. Two omissions matter more than the rest, because they are the parts of §5.1 a
|
|
1752
1825
|
writer is most likely to get wrong: no specimen defeats a rung by its **start**
|
|
1753
1826
|
pattern, and none defeats one with an **upper-case** pattern. A writer that tested only
|
|
@@ -1766,7 +1839,10 @@ of §5.7.
|
|
|
1766
1839
|
|
|
1767
1840
|
### 8.4 The charset round trip, measured
|
|
1768
1841
|
|
|
1769
|
-
Two claims of §2.1 were verified.
|
|
1842
|
+
Two claims of §2.1 were verified. The first is re-derived on every run by
|
|
1843
|
+
`test/sfz-harness/charset-round-trip.js`, which reads the tables below out of the
|
|
1844
|
+
runtime's own decoders rather than trusting this section, and checks the reverse table
|
|
1845
|
+
the extractor ships against the one the rule of §5.5 produces.
|
|
1770
1846
|
|
|
1771
1847
|
**Which encodings qualify.** Decoding all 256 byte values through each encoding
|
|
1772
1848
|
defined by the WHATWG standard shows 20 that are injective and never produce U+FFFD:
|
|
@@ -1816,6 +1892,7 @@ predicts.
|
|
|
1816
1892
|
| August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
|
|
1817
1893
|
| August 2026 | Core 1.5.119: the inlined ZIP library is built ASCII-only, and §2.1 now requires it of any bootstrap. Its CP437 table had been emitted as literal characters, which the page re-decoded as windows-1252, growing the table from 256 entries to 508 and shifting every lookup by 60 — so the one entry read without the UTF-8 flag, `page.pdf`, came back mangled and no archive with a PDF face extracted in any engine (§5.8) |
|
|
1818
1894
|
| August 2026 | Core 1.5.120: the hand-built `page.pdf` records set the language encoding flag, like every entry the ZIP writer produces (§5.8). Its name is ASCII, so no decoded name changes; what changes is that no entry in an archive is read through CP437 any more, closing the path the 1.5.119 defect surfaced on |
|
|
1895
|
+
| September 2026 | Core 1.5.126: the appended-data budget becomes the `maxAppendedDataLength` writer option and its default drops from 65535 to 16361 bytes, so the EOCD record stays inside libarchive's scan and `bsdtar` opens archives it used to reject (§5.2, §8.1). The 65535-byte comment ceiling is now a separate limit, stated in §4.2: a budget raised past it produces a run that cannot be declared |
|
|
1819
1896
|
|
|
1820
1897
|
This document was itself revised in August 2026, against core 1.5.108, after several
|
|
1821
1898
|
independent reviews. One of them was a reader built from this specification alone, with
|
|
@@ -83,15 +83,15 @@ function getSourceSrcData(sources) {
|
|
|
83
83
|
function setSrc(srcData, imgElement, pictureElement) {
|
|
84
84
|
if (srcData.src) {
|
|
85
85
|
imgElement.setAttribute("src", srcData.src);
|
|
86
|
-
imgElement.
|
|
87
|
-
imgElement.
|
|
86
|
+
imgElement.removeAttribute("srcset");
|
|
87
|
+
imgElement.removeAttribute("sizes");
|
|
88
88
|
} else {
|
|
89
89
|
imgElement.setAttribute("src", EMPTY_RESOURCE);
|
|
90
90
|
if (srcData.srcset) {
|
|
91
91
|
imgElement.setAttribute("srcset", srcData.srcset);
|
|
92
92
|
} else {
|
|
93
|
-
imgElement.
|
|
94
|
-
imgElement.
|
|
93
|
+
imgElement.removeAttribute("srcset");
|
|
94
|
+
imgElement.removeAttribute("sizes");
|
|
95
95
|
}
|
|
96
96
|
}
|
|
97
97
|
if (pictureElement) {
|
package/package.json
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "single-file-core",
|
|
3
|
-
"version": "1.5.
|
|
3
|
+
"version": "1.5.126",
|
|
4
4
|
"description": "SingleFile Core",
|
|
5
5
|
"author": "Gildas Lormeau",
|
|
6
6
|
"license": "AGPL-3.0-or-later",
|
|
7
7
|
"scripts": {
|
|
8
|
-
"test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js",
|
|
8
|
+
"test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
|
|
9
9
|
"bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
|
|
10
10
|
"bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
|
|
11
11
|
"bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
|
|
@@ -89,7 +89,8 @@ const PNG_CHUNK_CRC_LENGTH = 4;
|
|
|
89
89
|
const PNG_SIGNATURE_LENGTH = 8;
|
|
90
90
|
const PNG_IHDR_LENGTH = 25;
|
|
91
91
|
const COMMENT_LENGTH_FIELD_LENGTH = 2;
|
|
92
|
-
const
|
|
92
|
+
const MAX_ZIP_COMMENT_LENGTH = 65535;
|
|
93
|
+
const DEFAULT_MAX_APPENDED_DATA_LENGTH = 16361;
|
|
93
94
|
const PDF_ENTRY_FILENAME = "page.pdf";
|
|
94
95
|
const PRESCAN_WINDOW_LENGTH = 1024;
|
|
95
96
|
const PNG_TEXT_CHUNK_HEADER_LENGTH = 12;
|
|
@@ -121,6 +122,7 @@ const PROCESS_OPTION_NAMES = [
|
|
|
121
122
|
"insertMetaCSP",
|
|
122
123
|
"insertMetaNoIndex",
|
|
123
124
|
"insertTextBody",
|
|
125
|
+
"maxAppendedDataLength",
|
|
124
126
|
"password",
|
|
125
127
|
"preventAppendedData",
|
|
126
128
|
"selfExtractingArchive",
|
|
@@ -278,7 +280,7 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
|
|
|
278
280
|
const payloadView = new DataView(payload.buffer);
|
|
279
281
|
words.forEach((word, indexWord) => payloadView.setUint32(indexWord * 4, word, true));
|
|
280
282
|
extraData = "<sfz-extra-data>" + base64Encode(deflateRaw(payload)) + "</sfz-extra-data>";
|
|
281
|
-
if (options.preventAppendedData || extraData.length >
|
|
283
|
+
if (options.preventAppendedData || extraData.length > getMaxAppendedDataLength(options) - pageContent.length - endTags.length - (options.embeddedImage ? PNG_IEND_LENGTH + PNG_CHUNK_CRC_LENGTH : 0)) {
|
|
282
284
|
if (!options.extraDataSize) {
|
|
283
285
|
options.preventAppendedData = true;
|
|
284
286
|
options.extraDataSize = getReservationSize(extraData.length);
|
|
@@ -304,7 +306,7 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
|
|
|
304
306
|
if (options.declareAppendedData) {
|
|
305
307
|
const appendedDataLength = pageContent.length - data.length +
|
|
306
308
|
(options.embeddedImage ? PNG_CHUNK_CRC_LENGTH + PNG_IEND_LENGTH : 0);
|
|
307
|
-
if (appendedDataLength && appendedDataLength <=
|
|
309
|
+
if (appendedDataLength && appendedDataLength <= MAX_ZIP_COMMENT_LENGTH && isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options)) {
|
|
308
310
|
new DataView(pageContent.buffer, pageContent.byteOffset).setUint16(zipDataEnd, appendedDataLength, true);
|
|
309
311
|
}
|
|
310
312
|
}
|
|
@@ -324,6 +326,10 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
|
|
|
324
326
|
}
|
|
325
327
|
}
|
|
326
328
|
|
|
329
|
+
function getMaxAppendedDataLength(options) {
|
|
330
|
+
return options.maxAppendedDataLength === undefined ? DEFAULT_MAX_APPENDED_DATA_LENGTH : options.maxAppendedDataLength;
|
|
331
|
+
}
|
|
332
|
+
|
|
327
333
|
function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options) {
|
|
328
334
|
if (options.extractDataFromPageTags && options.extractDataFromPageTags[0] == "<plaintext>") {
|
|
329
335
|
return true;
|
|
@@ -747,11 +753,7 @@ async function getContent() {
|
|
|
747
753
|
});
|
|
748
754
|
return new Promise((resolve, reject) => {
|
|
749
755
|
let aborted = false;
|
|
750
|
-
|
|
751
|
-
extractDataFromDocument();
|
|
752
|
-
} else {
|
|
753
|
-
getPageData();
|
|
754
|
-
}
|
|
756
|
+
getPageData();
|
|
755
757
|
|
|
756
758
|
async function extractDataFromDocument() {
|
|
757
759
|
try {
|
|
@@ -177,7 +177,7 @@ function initRequestSync(message) {
|
|
|
177
177
|
if (!TOP_WINDOW) {
|
|
178
178
|
windowId = globalThis.frameId = message.windowId;
|
|
179
179
|
}
|
|
180
|
-
processFrames(document, message.options, windowId, sessionId);
|
|
180
|
+
processFrames(document, message.options, windowId, sessionId, false);
|
|
181
181
|
if (!TOP_WINDOW) {
|
|
182
182
|
sendInitResponse({ frames: [getFrameData(document, globalThis, windowId, message.options, message.scrolling)], sessionId, requestedFrameId: document.documentElement.dataset.requestedFrameId && windowId });
|
|
183
183
|
delete document.documentElement.dataset.requestedFrameId;
|
|
@@ -190,7 +190,7 @@ async function initRequestAsync(message) {
|
|
|
190
190
|
if (!TOP_WINDOW) {
|
|
191
191
|
windowId = globalThis.frameId = message.windowId;
|
|
192
192
|
}
|
|
193
|
-
processFrames(document, message.options, windowId, sessionId);
|
|
193
|
+
processFrames(document, message.options, windowId, sessionId, message.waitForFrames !== false);
|
|
194
194
|
if (!TOP_WINDOW) {
|
|
195
195
|
sendInitResponse({ frames: [getFrameData(document, globalThis, windowId, message.options, message.scrolling)], sessionId, requestedFrameId: document.documentElement.dataset.requestedFrameId && windowId });
|
|
196
196
|
delete document.documentElement.dataset.requestedFrameId;
|
|
@@ -254,15 +254,15 @@ function initResponse(message) {
|
|
|
254
254
|
}
|
|
255
255
|
}
|
|
256
256
|
}
|
|
257
|
-
function processFrames(doc, options, parentWindowId, sessionId) {
|
|
257
|
+
function processFrames(doc, options, parentWindowId, sessionId, waitForFrames) {
|
|
258
258
|
const frameElements = getFrames(doc);
|
|
259
|
-
processFramesAsync(doc, frameElements, options, parentWindowId, sessionId);
|
|
259
|
+
processFramesAsync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames);
|
|
260
260
|
if (frameElements.length) {
|
|
261
|
-
processFramesSync(doc, frameElements, options, parentWindowId, sessionId);
|
|
261
|
+
processFramesSync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames);
|
|
262
262
|
}
|
|
263
263
|
}
|
|
264
264
|
|
|
265
|
-
function processFramesAsync(doc, frameElements, options, parentWindowId, sessionId) {
|
|
265
|
+
function processFramesAsync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames) {
|
|
266
266
|
const frames = [];
|
|
267
267
|
let requestTimeouts;
|
|
268
268
|
if (sessions.get(sessionId)) {
|
|
@@ -280,17 +280,18 @@ function processFramesAsync(doc, frameElements, options, parentWindowId, session
|
|
|
280
280
|
frameElements.forEach((frameElement, frameIndex) => {
|
|
281
281
|
const windowId = parentWindowId + WINDOW_ID_SEPARATOR + frameIndex;
|
|
282
282
|
try {
|
|
283
|
-
sendMessage(frameElement.contentWindow, { method: INIT_REQUEST_MESSAGE, windowId, sessionId, options, scrolling: frameElement.scrolling });
|
|
283
|
+
sendMessage(frameElement.contentWindow, { method: INIT_REQUEST_MESSAGE, windowId, sessionId, options, scrolling: frameElement.scrolling, waitForFrames });
|
|
284
284
|
// eslint-disable-next-line no-unused-vars
|
|
285
285
|
} catch (error) {
|
|
286
286
|
// ignored
|
|
287
287
|
}
|
|
288
|
-
|
|
288
|
+
setFrameFallback(sessionId, windowId, () => getSrcdocFrameData(frameElement, windowId, options, sessionId));
|
|
289
|
+
requestTimeouts[windowId] = globalThis.setTimeout(() => sendInitResponse({ frames: [getFrameFallback(sessionId, windowId) || { windowId, processed: true }], sessionId }), TIMEOUT_INIT_REQUEST_MESSAGE);
|
|
289
290
|
});
|
|
290
291
|
delete doc.documentElement.dataset.requestedFrameId;
|
|
291
292
|
}
|
|
292
293
|
|
|
293
|
-
function processFramesSync(doc, frameElements, options, parentWindowId, sessionId) {
|
|
294
|
+
function processFramesSync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames) {
|
|
294
295
|
const frames = [];
|
|
295
296
|
frameElements.forEach((frameElement, frameIndex) => {
|
|
296
297
|
const windowId = parentWindowId + WINDOW_ID_SEPARATOR + frameIndex;
|
|
@@ -303,21 +304,25 @@ function processFramesSync(doc, frameElements, options, parentWindowId, sessionI
|
|
|
303
304
|
} catch (error) {
|
|
304
305
|
// ignored
|
|
305
306
|
}
|
|
306
|
-
const srcdoc = frameElement.getAttribute("srcdoc");
|
|
307
|
-
if (!frameDoc && srcdoc) {
|
|
308
|
-
const doc = new DOMParser().parseFromString(srcdoc, "text/html");
|
|
309
|
-
frameDoc = doc;
|
|
310
|
-
frameWindow = globalThis;
|
|
311
|
-
}
|
|
312
307
|
if (frameDoc) {
|
|
313
308
|
try {
|
|
314
309
|
clearFrameTimeout("requestTimeouts", sessionId, windowId);
|
|
315
|
-
processFrames(frameDoc, options, windowId, sessionId);
|
|
310
|
+
processFrames(frameDoc, options, windowId, sessionId, waitForFrames);
|
|
316
311
|
frames.push(getFrameData(frameDoc, frameWindow, windowId, options, frameElement.scrolling));
|
|
317
312
|
// eslint-disable-next-line no-unused-vars
|
|
318
313
|
} catch (error) {
|
|
319
314
|
frames.push({ windowId, processed: true });
|
|
320
315
|
}
|
|
316
|
+
} else if (!waitForFrames) {
|
|
317
|
+
// the frame is cross-origin or sandboxed, so its document is out of reach. Re-parsing
|
|
318
|
+
// srcdoc is the only source left here, and it is markup only: no script has run and
|
|
319
|
+
// nothing is rendered. When there is time to wait, the fallback is kept for the frames
|
|
320
|
+
// that never answer instead, so a frame that does answer wins with its rendered data
|
|
321
|
+
const fallbackFrameData = getFrameFallback(sessionId, windowId);
|
|
322
|
+
if (fallbackFrameData) {
|
|
323
|
+
clearFrameTimeout("requestTimeouts", sessionId, windowId);
|
|
324
|
+
frames.push(fallbackFrameData);
|
|
325
|
+
}
|
|
321
326
|
}
|
|
322
327
|
});
|
|
323
328
|
sendInitResponse({ frames, sessionId, requestedFrameId: doc.documentElement.dataset.requestedFrameId && parentWindowId });
|
|
@@ -338,7 +343,39 @@ function clearFrameTimeout(type, sessionId, windowId) {
|
|
|
338
343
|
function createFrameResponseTimeout(sessionId, windowId) {
|
|
339
344
|
const session = sessions.get(sessionId);
|
|
340
345
|
if (session && session.responseTimeouts) {
|
|
341
|
-
session.responseTimeouts[windowId] = globalThis.setTimeout(() => sendInitResponse({ frames: [
|
|
346
|
+
session.responseTimeouts[windowId] = globalThis.setTimeout(() => sendInitResponse({ frames: [getFrameFallback(sessionId, windowId) || { windowId, processed: true }], sessionId: sessionId }), TIMEOUT_INIT_RESPONSE_MESSAGE);
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
function setFrameFallback(sessionId, windowId, getFallbackFrameData) {
|
|
351
|
+
const session = sessions.get(sessionId);
|
|
352
|
+
if (session) {
|
|
353
|
+
if (!session.frameFallbacks) {
|
|
354
|
+
session.frameFallbacks = {};
|
|
355
|
+
}
|
|
356
|
+
session.frameFallbacks[windowId] = getFallbackFrameData;
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
function getFrameFallback(sessionId, windowId) {
|
|
361
|
+
const session = sessions.get(sessionId);
|
|
362
|
+
const getFallbackFrameData = session && session.frameFallbacks && session.frameFallbacks[windowId];
|
|
363
|
+
if (getFallbackFrameData) {
|
|
364
|
+
return getFallbackFrameData();
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
function getSrcdocFrameData(frameElement, windowId, options, sessionId) {
|
|
369
|
+
const srcdoc = frameElement.getAttribute("srcdoc");
|
|
370
|
+
if (srcdoc) {
|
|
371
|
+
try {
|
|
372
|
+
const frameDoc = new DOMParser().parseFromString(srcdoc, "text/html");
|
|
373
|
+
processFrames(frameDoc, options, windowId, sessionId, false);
|
|
374
|
+
return getFrameData(frameDoc, globalThis, windowId, options, frameElement.scrolling);
|
|
375
|
+
// eslint-disable-next-line no-unused-vars
|
|
376
|
+
} catch (error) {
|
|
377
|
+
// ignored
|
|
378
|
+
}
|
|
342
379
|
}
|
|
343
380
|
}
|
|
344
381
|
|
|
@@ -242,39 +242,46 @@
|
|
|
242
242
|
}
|
|
243
243
|
return boundingRect;
|
|
244
244
|
};
|
|
245
|
+
Element.prototype.getBoundingClientRect.toString = function () { return "function getBoundingClientRect() { [native code] }"; };
|
|
246
|
+
setFunctionName(Element.prototype.getBoundingClientRect, "getBoundingClientRect");
|
|
245
247
|
}
|
|
246
248
|
}
|
|
247
249
|
if (!globalThis._singleFileImage) {
|
|
248
|
-
const
|
|
249
|
-
globalThis._singleFileImage =
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
};
|
|
275
|
-
|
|
250
|
+
const NativeImage = globalThis.Image;
|
|
251
|
+
globalThis._singleFileImage = NativeImage;
|
|
252
|
+
const ImageWrapper = function Image() {
|
|
253
|
+
const image = new NativeImage(...arguments);
|
|
254
|
+
const result = new NativeImage(...arguments);
|
|
255
|
+
result.__defineSetter__("src", value => {
|
|
256
|
+
image.src = value;
|
|
257
|
+
document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT, { detail: image.src }));
|
|
258
|
+
});
|
|
259
|
+
result.__defineGetter__("src", () => image.src);
|
|
260
|
+
result.__defineSetter__("srcset", value => {
|
|
261
|
+
document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT));
|
|
262
|
+
image.srcset = value;
|
|
263
|
+
});
|
|
264
|
+
result.__defineGetter__("srcset", () => image.srcset);
|
|
265
|
+
result.__defineGetter__("height", () => image.height);
|
|
266
|
+
result.__defineGetter__("width", () => image.width);
|
|
267
|
+
result.__defineGetter__("naturalHeight", () => image.naturalHeight);
|
|
268
|
+
result.__defineGetter__("naturalWidth", () => image.naturalWidth);
|
|
269
|
+
if (image.decode) {
|
|
270
|
+
const decode = function decode() { return image.decode(); };
|
|
271
|
+
decode.toString = function () { return "function decode() { [native code] }"; };
|
|
272
|
+
setFunctionName(decode, "decode");
|
|
273
|
+
result.__defineGetter__("decode", () => decode);
|
|
274
|
+
}
|
|
275
|
+
image.onload = image.onloadend = image.onerror = event => {
|
|
276
|
+
document.dispatchEvent(new CustomEvent(IMAGE_LOADED_EVENT, { detail: image.src }));
|
|
277
|
+
result.dispatchEvent(new Event(event.type, event));
|
|
276
278
|
};
|
|
277
|
-
|
|
279
|
+
return result;
|
|
280
|
+
};
|
|
281
|
+
ImageWrapper.prototype = NativeImage.prototype;
|
|
282
|
+
ImageWrapper.toString = function () { return "function Image() { [native code] }"; };
|
|
283
|
+
setFunctionName(ImageWrapper, "Image");
|
|
284
|
+
globalThis.__defineGetter__("Image", () => ImageWrapper);
|
|
278
285
|
}
|
|
279
286
|
const verticalZoomFactor = clientHeight / scrollHeight;
|
|
280
287
|
const horizontalZoomFactor = clientWidth / scrollWidth;
|
|
@@ -371,9 +378,10 @@
|
|
|
371
378
|
|
|
372
379
|
if (globalThis.CSS && globalThis.CSS.paintWorklet && globalThis.CSS.paintWorklet.addModule) {
|
|
373
380
|
const addModule = globalThis.CSS.paintWorklet.addModule;
|
|
374
|
-
globalThis.CSS.paintWorklet.addModule = function (moduleURL
|
|
381
|
+
globalThis.CSS.paintWorklet.addModule = function (moduleURL) {
|
|
375
382
|
try {
|
|
376
383
|
const result = addModule.apply(globalThis.CSS.paintWorklet, arguments);
|
|
384
|
+
const options = arguments[1];
|
|
377
385
|
moduleURL = new URL(moduleURL, document.baseURI).href;
|
|
378
386
|
document.dispatchEvent(new CustomEvent(NEW_WORKLET_EVENT, { detail: { moduleURL, options } }));
|
|
379
387
|
return result;
|
|
@@ -382,6 +390,8 @@
|
|
|
382
390
|
throw error;
|
|
383
391
|
}
|
|
384
392
|
};
|
|
393
|
+
globalThis.CSS.paintWorklet.addModule.toString = function () { return "function addModule() { [native code] }"; };
|
|
394
|
+
setFunctionName(globalThis.CSS.paintWorklet.addModule, "addModule");
|
|
385
395
|
}
|
|
386
396
|
|
|
387
397
|
if (globalThis.FontFace) {
|
|
@@ -412,6 +422,7 @@
|
|
|
412
422
|
}
|
|
413
423
|
};
|
|
414
424
|
document.fonts.delete.toString = function () { return "function delete() { [native code] }"; };
|
|
425
|
+
setFunctionName(document.fonts.delete, "delete");
|
|
415
426
|
const clearFonts = document.fonts.clear;
|
|
416
427
|
document.fonts.clear = function () {
|
|
417
428
|
try {
|
|
@@ -423,16 +434,16 @@
|
|
|
423
434
|
}
|
|
424
435
|
};
|
|
425
436
|
document.fonts.clear.toString = function () { return "function clear() { [native code] }"; };
|
|
437
|
+
setFunctionName(document.fonts.clear, "clear");
|
|
426
438
|
}
|
|
427
439
|
|
|
428
440
|
if (globalThis.IntersectionObserver) {
|
|
429
441
|
const origIntersectionObserver = globalThis.IntersectionObserver;
|
|
430
|
-
globalThis.IntersectionObserver = function IntersectionObserver() {
|
|
442
|
+
globalThis.IntersectionObserver = function IntersectionObserver(callback) {
|
|
431
443
|
try {
|
|
432
444
|
const intersectionObserver = new origIntersectionObserver(...arguments);
|
|
433
445
|
const observeIntersection = origIntersectionObserver.prototype.observe || intersectionObserver.observe;
|
|
434
446
|
const unobserveIntersection = origIntersectionObserver.prototype.unobserve || intersectionObserver.unobserve;
|
|
435
|
-
const callback = arguments[0];
|
|
436
447
|
const options = arguments[1];
|
|
437
448
|
if (observeIntersection) {
|
|
438
449
|
intersectionObserver.observe = function (targetElement) {
|
|
@@ -450,6 +461,7 @@
|
|
|
450
461
|
}
|
|
451
462
|
};
|
|
452
463
|
intersectionObserver.observe.toString = function () { return "function observe() { [native code] }"; };
|
|
464
|
+
setFunctionName(intersectionObserver.observe, "observe");
|
|
453
465
|
}
|
|
454
466
|
if (unobserveIntersection) {
|
|
455
467
|
intersectionObserver.unobserve = function (targetElement) {
|
|
@@ -471,6 +483,7 @@
|
|
|
471
483
|
}
|
|
472
484
|
};
|
|
473
485
|
intersectionObserver.unobserve.toString = function () { return "function unobserve() { [native code] }"; };
|
|
486
|
+
setFunctionName(intersectionObserver.unobserve, "unobserve");
|
|
474
487
|
}
|
|
475
488
|
observers.set(intersectionObserver, { callback, options });
|
|
476
489
|
return intersectionObserver;
|
|
@@ -496,6 +509,7 @@
|
|
|
496
509
|
}
|
|
497
510
|
};
|
|
498
511
|
CSSStyleSheet.prototype.replaceSync.toString = function () { return "function replaceSync() { [native code] }"; };
|
|
512
|
+
setFunctionName(CSSStyleSheet.prototype.replaceSync, "replaceSync");
|
|
499
513
|
const orginalReplace = CSSStyleSheet.prototype.replace;
|
|
500
514
|
CSSStyleSheet.prototype.replace = async function (text) {
|
|
501
515
|
try {
|
|
@@ -508,10 +522,11 @@
|
|
|
508
522
|
}
|
|
509
523
|
};
|
|
510
524
|
CSSStyleSheet.prototype.replace.toString = function () { return "function replace() { [native code] }"; };
|
|
525
|
+
setFunctionName(CSSStyleSheet.prototype.replace, "replace");
|
|
511
526
|
const originalInsertRule = CSSStyleSheet.prototype.insertRule;
|
|
512
|
-
CSSStyleSheet.prototype.insertRule = function (rule
|
|
527
|
+
CSSStyleSheet.prototype.insertRule = function (rule) {
|
|
513
528
|
try {
|
|
514
|
-
const result = originalInsertRule.apply(this, [rule,
|
|
529
|
+
const result = originalInsertRule.apply(this, [rule, arguments[1]]);
|
|
515
530
|
adoptedStylesheetsData.delete(this);
|
|
516
531
|
return result;
|
|
517
532
|
} catch (error) {
|
|
@@ -520,6 +535,7 @@
|
|
|
520
535
|
}
|
|
521
536
|
};
|
|
522
537
|
CSSStyleSheet.prototype.insertRule.toString = function () { return "function insertRule() { [native code] }"; };
|
|
538
|
+
setFunctionName(CSSStyleSheet.prototype.insertRule, "insertRule");
|
|
523
539
|
const originalDeleteRule = CSSStyleSheet.prototype.deleteRule;
|
|
524
540
|
CSSStyleSheet.prototype.deleteRule = function (index) {
|
|
525
541
|
try {
|
|
@@ -532,6 +548,7 @@
|
|
|
532
548
|
}
|
|
533
549
|
};
|
|
534
550
|
CSSStyleSheet.prototype.deleteRule.toString = function () { return "function deleteRule() { [native code] }"; };
|
|
551
|
+
setFunctionName(CSSStyleSheet.prototype.deleteRule, "deleteRule");
|
|
535
552
|
|
|
536
553
|
// the listener below is reached through the host element, and a closed shadow root is
|
|
537
554
|
// not reachable from it, so the roots are recorded as they are created
|
|
@@ -27,7 +27,7 @@ any check failed.
|
|
|
27
27
|
|
|
28
28
|
| Script | What it covers |
|
|
29
29
|
|---|---|
|
|
30
|
-
| `format-rules.js` | The rules of the format: charset
|
|
30
|
+
| `format-rules.js` | The rules of the format: the charset declaration and the doctype cap that keeps it inside the scan window, the wrapper-tag ladder and its selection tests, the identifier, appended-data placement and declaration, password scope, the PDF and PNG faces. |
|
|
31
31
|
| `stored-trigger.js` | That a stored (uncompressed) entry whose bytes contain a rung's pattern moves the writer to the right rung. |
|
|
32
32
|
| `check-determinism.js` | That the same inputs produce the same bytes, and that the levers which should change the output do. |
|
|
33
33
|
| `option-wiring.js` | That every option `compression.js` reads is either declared as a caller option or classified as internal, and that `single-file.js` still builds its argument from that declaration. Guards the layer the other suites sit below. |
|
|
@@ -40,6 +40,7 @@ any check failed.
|
|
|
40
40
|
| `filename-characters.js` | That `getValidFilename` maps a full-width lookalike one character at a time — `C++` used to be saved as `C+` — while a run of characters with no lookalike still collapses to a single replacement. |
|
|
41
41
|
| `zip64.js` | That the `page.pdf` record injection accounts for the zip64 end of central directory record (§5.7): all four EOCD fields left at their sentinels, the entry counts and directory size carried in the zip64 record, the directory offset pointing at the injected record, and the archive still readable. The branch runs only past 4 GiB or 65535 entries, so nothing reached it before; the suite forces zip64 through `zipWriter.options` from inside the `writeEntries` callback, with no production lever. |
|
|
42
42
|
| `byte-map.js` | That the byte offsets §8.2 of the specification prints still describe what the writer emits: the prologue order, the doctype and root tag with nothing between them, the identifier's length ahead of the region, absolute EOCD offsets, and the entry order. The specimen §8.2 documents is saved from a live URL and has never been in this repository, so none of its numbers could be checked; three of them were wrong. This builds an equivalent with no network. |
|
|
43
|
+
| `charset-round-trip.js` | That the encoding tables §8.4 prints still describe the WHATWG index: which 20 of the 38 encodings carry all 256 byte values through a decode injectively, the sizes of the reverse tables they need, and the five windows-1252 positions a platform codec of the same name leaves undefined. It also re-derives the reverse table the extractor ships as a literal, which no build step checks and which corrupts one byte per occurrence when wrong. |
|
|
43
44
|
| `css-fonts-minifier.js` | That `removeUnusedFonts` reads the font families it prunes on correctly: a `var()` family resolved from the values the document declares and not only from the ones the body inherits, every font kept when the value is genuinely undetermined, and a multi-word family name that does not also claim a font named after its own tail. |
|
|
44
45
|
|
|
45
46
|
## The tools
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright 2010-2026 Gildas Lormeau
|
|
3
|
+
* contact : gildas.lormeau <at> gmail.com
|
|
4
|
+
*
|
|
5
|
+
* This file is part of SingleFile.
|
|
6
|
+
*
|
|
7
|
+
* The code in this file is free software: you can redistribute it and/or
|
|
8
|
+
* modify it under the terms of the GNU Affero General Public License
|
|
9
|
+
* (GNU AGPL) as published by the Free Software Foundation, either version 3
|
|
10
|
+
* of the License, or (at your option) any later version.
|
|
11
|
+
*
|
|
12
|
+
* The code in this file is distributed in the hope that it will be useful,
|
|
13
|
+
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
14
|
+
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU Affero
|
|
15
|
+
* General Public License for more details.
|
|
16
|
+
*
|
|
17
|
+
* As additional permission under GNU AGPL version 3 section 7, you may
|
|
18
|
+
* distribute UNMODIFIED VERSIONS OF THIS file without the copy of the GNU
|
|
19
|
+
* AGPL normally required by section 4, provided you include this license
|
|
20
|
+
* notice and a URL through which recipients can access the Corresponding
|
|
21
|
+
* Source.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
// Universal mode recovers the ZIP region from the characters the HTML parser produced, so the
|
|
25
|
+
// declared charset has to carry all 256 byte values through a decode injectively (§2.1). Which
|
|
26
|
+
// encodings do is a property of the WHATWG index, not of this repository, and §8.4 prints the
|
|
27
|
+
// answer as a table: 20 qualify, 18 do not, and each qualifying one needs a reverse table of a
|
|
28
|
+
// stated size. Nothing re-derived that table -- it was measured once, by hand, outside the repo,
|
|
29
|
+
// and would go stale silently if an index changed or the prose were edited.
|
|
30
|
+
//
|
|
31
|
+
// The last check is the one with teeth. §5.5 requires the reverse table to be derived from the
|
|
32
|
+
// WHATWG index and NOT from a platform codec of the same name, because most platform codecs
|
|
33
|
+
// leave five windows-1252 positions undefined and those bytes occur in ordinary compressed data.
|
|
34
|
+
// The extractor ships that table as a literal, so it is derived once at authoring time and never
|
|
35
|
+
// again; here it is re-derived from the runtime's own decoder and compared entry by entry.
|
|
36
|
+
|
|
37
|
+
const BYTES = new Uint8Array(256).map((_, index) => index);
|
|
38
|
+
|
|
39
|
+
// the 20 of §8.4, in the order the section lists them
|
|
40
|
+
const QUALIFYING = [
|
|
41
|
+
"windows-1252", "iso-8859-2", "iso-8859-4", "iso-8859-5", "iso-8859-10", "iso-8859-13",
|
|
42
|
+
"iso-8859-14", "iso-8859-15", "iso-8859-16", "koi8-r", "koi8-u", "macintosh", "windows-1250",
|
|
43
|
+
"windows-1251", "windows-1254", "windows-1256", "windows-1258", "x-mac-cyrillic", "ibm866",
|
|
44
|
+
"x-user-defined"
|
|
45
|
+
];
|
|
46
|
+
// the 18 that do not: eight single-byte encodings with undefined positions in their index, then
|
|
47
|
+
// the multi-byte ones, which decode a lone byte sequence to U+FFFD or to fewer than 256 characters
|
|
48
|
+
const DISQUALIFIED = [
|
|
49
|
+
"iso-8859-3", "iso-8859-6", "iso-8859-7", "iso-8859-8", "windows-874", "windows-1253",
|
|
50
|
+
"windows-1255", "windows-1257",
|
|
51
|
+
"utf-8", "utf-16le", "utf-16be", "gbk", "gb18030", "big5", "euc-jp", "shift_jis", "euc-kr",
|
|
52
|
+
"iso-2022-jp"
|
|
53
|
+
];
|
|
54
|
+
|
|
55
|
+
let failures = 0;
|
|
56
|
+
|
|
57
|
+
function describe(label) {
|
|
58
|
+
const points = Array.from(new TextDecoder(label).decode(BYTES));
|
|
59
|
+
if (points.length != 256) {
|
|
60
|
+
return { qualifies: false, reason: points.length + " characters" };
|
|
61
|
+
}
|
|
62
|
+
const table = new Map();
|
|
63
|
+
const seen = new Set();
|
|
64
|
+
let identity = 0;
|
|
65
|
+
for (let byte = 0; byte < 256; byte++) {
|
|
66
|
+
const codePoint = points[byte].codePointAt(0);
|
|
67
|
+
if (codePoint == 0xFFFD) {
|
|
68
|
+
return { qualifies: false, reason: "U+FFFD at 0x" + byte.toString(16) };
|
|
69
|
+
}
|
|
70
|
+
if (seen.has(codePoint)) {
|
|
71
|
+
return { qualifies: false, reason: "collision at 0x" + byte.toString(16) };
|
|
72
|
+
}
|
|
73
|
+
seen.add(codePoint);
|
|
74
|
+
if (codePoint == byte) {
|
|
75
|
+
identity++;
|
|
76
|
+
} else {
|
|
77
|
+
table.set(codePoint, byte);
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
return { qualifies: true, identity, table, points };
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const described = new Map([...QUALIFYING, ...DISQUALIFIED].map(label => [label, describe(label)]));
|
|
84
|
+
|
|
85
|
+
check("§8.4 covers the whole standard: 20 qualifying + 18 disqualified",
|
|
86
|
+
QUALIFYING.length + DISQUALIFIED.length == 38 && new Set([...QUALIFYING, ...DISQUALIFIED]).size == 38);
|
|
87
|
+
|
|
88
|
+
const wrongVerdict = [...described].filter(([label, result]) =>
|
|
89
|
+
result.qualifies != QUALIFYING.includes(label));
|
|
90
|
+
check("every encoding falls on the side of the table §8.4 puts it on", wrongVerdict.length == 0,
|
|
91
|
+
wrongVerdict.map(([label, result]) => label + " " + (result.reason || "qualifies")).join(", "));
|
|
92
|
+
|
|
93
|
+
// §8.4 quotes the extremes of the reverse-table sizes; they bound what an implementation has to
|
|
94
|
+
// carry to support any qualifying charset rather than only the one the reference writer declares
|
|
95
|
+
const qualifying = QUALIFYING.filter(label => described.get(label).qualifies);
|
|
96
|
+
const sizes = qualifying.map(label => [label, described.get(label).table.size]);
|
|
97
|
+
const smallest = Math.min(...sizes.map(([, size]) => size));
|
|
98
|
+
const largest = Math.max(...sizes.map(([, size]) => size));
|
|
99
|
+
check("the smallest reverse table is 8 entries, iso-8859-15", smallest == 8 &&
|
|
100
|
+
sizes.filter(([, size]) => size == smallest).map(([label]) => label).join() == "iso-8859-15");
|
|
101
|
+
check("the largest is 128, for koi8-r, koi8-u, ibm866 and x-user-defined", largest == 128 &&
|
|
102
|
+
sizes.filter(([, size]) => size == largest).map(([label]) => label).sort().join() ==
|
|
103
|
+
"ibm866,koi8-r,koi8-u,x-user-defined");
|
|
104
|
+
|
|
105
|
+
const windows1252 = described.get("windows-1252");
|
|
106
|
+
check("windows-1252 decodes 229 of the 256 values to themselves (§5.5 rule 1)",
|
|
107
|
+
windows1252.identity == 229, String(windows1252.identity));
|
|
108
|
+
check("its reverse table is the remaining 27 (§5.5 rule 2)", windows1252.table.size == 27,
|
|
109
|
+
String(windows1252.table.size));
|
|
110
|
+
|
|
111
|
+
// the trap of §5.5: iso-8859-1 is a LABEL of windows-1252, not the identity mapping its name
|
|
112
|
+
// suggests, so a reader that treats it as latin-1 builds a table with no entries at all
|
|
113
|
+
check("iso-8859-1 is a label of windows-1252, not a separate identity encoding",
|
|
114
|
+
new TextDecoder("iso-8859-1").encoding == "windows-1252" &&
|
|
115
|
+
new TextDecoder("latin1").encoding == "windows-1252");
|
|
116
|
+
|
|
117
|
+
// the five positions of the §5.5 table: the WHATWG index assigns them, most platform codecs do not
|
|
118
|
+
check("the WHATWG index assigns 0x81, 0x8D, 0x8F, 0x90 and 0x9D (§5.5)",
|
|
119
|
+
[0x81, 0x8D, 0x8F, 0x90, 0x9D].every(byte =>
|
|
120
|
+
windows1252.points[byte].codePointAt(0) == byte && !windows1252.table.has(byte)));
|
|
121
|
+
|
|
122
|
+
// §8.4's caveat on x-user-defined: it qualifies on the criterion and is still a poor choice
|
|
123
|
+
check("x-user-defined maps 0x80-0xFF into the Private Use Area, U+F780-U+F7FF",
|
|
124
|
+
[...Array(128).keys()].every(index =>
|
|
125
|
+
described.get("x-user-defined").points[128 + index].codePointAt(0) == 0xF780 + index));
|
|
126
|
+
|
|
127
|
+
// the round trip of §5.5 rules 1 and 2, on every byte value and every qualifying charset. NUL and
|
|
128
|
+
// the newlines need rules 3 and 4 in a browser, where the parser has replaced and normalized them;
|
|
129
|
+
// through a decoder alone they arrive intact, so the mapping is exact for all 256 values here
|
|
130
|
+
const broken = qualifying.filter(label => {
|
|
131
|
+
const { table, points } = described.get(label);
|
|
132
|
+
return points.some((character, byte) => {
|
|
133
|
+
const codePoint = character.codePointAt(0);
|
|
134
|
+
return (table.has(codePoint) ? table.get(codePoint) : codePoint) != byte;
|
|
135
|
+
});
|
|
136
|
+
});
|
|
137
|
+
check("all 256 byte values survive decode and reverse mapping, under every qualifying charset",
|
|
138
|
+
broken.length == 0, broken.join(", "));
|
|
139
|
+
|
|
140
|
+
// the extractor's own table, re-derived. It is inlined into every archive, so an error here is
|
|
141
|
+
// not caught by any build step and corrupts one byte per occurrence in the recovered region
|
|
142
|
+
const compression = await Deno.readTextFile(new URL("../../processors/compression/compression.js", import.meta.url));
|
|
143
|
+
const literal = compression.slice(compression.indexOf("const characterMap = new Map(["));
|
|
144
|
+
const shipped = new Map([...literal.slice(0, literal.indexOf("]);")).matchAll(/\[(\d+),\s*(\d+)\]/g)]
|
|
145
|
+
.map(([, codePoint, byte]) => [Number(codePoint), Number(byte)]));
|
|
146
|
+
const expected = new Map([[0xFFFD, 0], ...windows1252.table]);
|
|
147
|
+
const wrongEntries = [...expected].filter(([codePoint, byte]) => shipped.get(codePoint) !== byte);
|
|
148
|
+
const extraEntries = [...shipped].filter(([codePoint]) => !expected.has(codePoint));
|
|
149
|
+
check("the extractor's characterMap is the derived windows-1252 table plus U+FFFD (28 entries)",
|
|
150
|
+
shipped.size == 28 && wrongEntries.length == 0 && extraEntries.length == 0,
|
|
151
|
+
"missing/wrong " + JSON.stringify(wrongEntries) + " extra " + JSON.stringify(extraEntries));
|
|
152
|
+
|
|
153
|
+
console.log(failures ? `\n${failures} check(s) FAILED` : "\nall checks passed");
|
|
154
|
+
Deno.exit(failures ? 1 : 0);
|
|
155
|
+
|
|
156
|
+
function check(label, condition, detail) {
|
|
157
|
+
if (!condition) {
|
|
158
|
+
failures++;
|
|
159
|
+
}
|
|
160
|
+
console.log((condition ? "PASS" : "FAIL") + " " + label + (condition || !detail ? "" : ": " + detail));
|
|
161
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import "./dom-stub.js";
|
|
2
|
-
import { makePageData, makeOptions, runProcess, mulberry32 } from "./common.js";
|
|
2
|
+
import { makePageData, makeOptions, runProcess, mulberry32, freezeDate } from "./common.js";
|
|
3
3
|
import { ZipReader, ZipWriter, BlobReader } from "../../vendor/zip/zip.js";
|
|
4
4
|
|
|
5
5
|
// the quote is there because the escaper the title shares with the table of contents encodes
|
|
@@ -239,6 +239,65 @@ function countIdentifiers(text) {
|
|
|
239
239
|
check("a relocated archive declares no comment", view.getUint16(bytes.length - 2, true), 0);
|
|
240
240
|
}
|
|
241
241
|
|
|
242
|
+
function makeNewlinePageData(seed, newlineCount) {
|
|
243
|
+
const pageData = makePageData(seed, 4 * 1024);
|
|
244
|
+
const rand = mulberry32(seed);
|
|
245
|
+
const newlines = ["\n", "\r", "\r\n"];
|
|
246
|
+
let content = "";
|
|
247
|
+
for (let index = 0; index < newlineCount; index++) {
|
|
248
|
+
content += newlines[(rand() * 3) | 0];
|
|
249
|
+
}
|
|
250
|
+
pageData.resources.stylesheets.push({ name: "newlines.txt", extension: ".txt", content, url: "https://example.com/newlines.txt" });
|
|
251
|
+
return pageData;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// libarchive gives up looking for the end of central directory record 16383 bytes from the end of
|
|
255
|
+
// the file, so the default budget is 16361 appended bytes: that window minus the 22-byte record.
|
|
256
|
+
// This fixture's payload lands between the default and the 65535-byte comment ceiling, which is
|
|
257
|
+
// the range the old budget kept appended and bsdtar could not open.
|
|
258
|
+
// The Date must be frozen: `archiveTime` (compression.js:155) puts an ISO timestamp in the
|
|
259
|
+
// archive, whose milliseconds move a few newline bytes in and out of the recovered range and
|
|
260
|
+
// therefore change the payload length build to build. The boundary checks below compare a budget
|
|
261
|
+
// against a run measured in an EARLIER build, so without freezing they are off by a few bytes one
|
|
262
|
+
// run in three. check-determinism.js asserts both halves of that
|
|
263
|
+
{
|
|
264
|
+
const restoreDate = freezeDate();
|
|
265
|
+
try {
|
|
266
|
+
const wide = makeOptions({ disableCompression: true, maxAppendedDataLength: 65535 });
|
|
267
|
+
const { bytes: wideBytes } = await runProcess(makeNewlinePageData(43, 80 * 1000), wide);
|
|
268
|
+
const wideTail = readAppendedData(wideBytes).trailing;
|
|
269
|
+
check("a wider budget keeps the payload appended", wide.extraDataSize, undefined);
|
|
270
|
+
check("the fixture overflows the default budget", wideTail > 16361, true);
|
|
271
|
+
check("the fixture fits the comment ceiling", wideTail <= 65535, true);
|
|
272
|
+
|
|
273
|
+
const byDefault = makeOptions({ disableCompression: true });
|
|
274
|
+
const { bytes } = await runProcess(makeNewlinePageData(43, 80 * 1000), byDefault);
|
|
275
|
+
check("the default budget relocates the payload", byDefault.extraDataSize > 0, true);
|
|
276
|
+
check("the default budget keeps the record in libarchive's window", readAppendedData(bytes).trailing <= 16361, true);
|
|
277
|
+
|
|
278
|
+
const fitting = makeOptions({ disableCompression: true, maxAppendedDataLength: wideTail });
|
|
279
|
+
await runProcess(makeNewlinePageData(43, 80 * 1000), fitting);
|
|
280
|
+
check("a budget matching the run to the byte keeps it appended", fitting.extraDataSize, undefined);
|
|
281
|
+
|
|
282
|
+
const tight = makeOptions({ disableCompression: true, maxAppendedDataLength: wideTail - 1 });
|
|
283
|
+
await runProcess(makeNewlinePageData(43, 80 * 1000), tight);
|
|
284
|
+
check("one byte below the run relocates it", tight.extraDataSize > 0, true);
|
|
285
|
+
} finally {
|
|
286
|
+
restoreDate();
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
// the budget and the comment ceiling are two different limits, and only the second is a property
|
|
291
|
+
// of the format. A budget raised past 65535 produces a run the comment-length field cannot hold,
|
|
292
|
+
// which setUint16 would write back modulo 65536: the writer leaves it undeclared instead
|
|
293
|
+
{
|
|
294
|
+
const options = makeOptions({ disableCompression: true, declareAppendedData: true, maxAppendedDataLength: Number.MAX_SAFE_INTEGER });
|
|
295
|
+
const { bytes } = await runProcess(makeNewlinePageData(44, 320 * 1000), options);
|
|
296
|
+
const { declared, trailing } = readAppendedData(bytes);
|
|
297
|
+
check("a run past the comment ceiling stays appended", trailing > 65535, true);
|
|
298
|
+
check("a run past the comment ceiling is left undeclared", declared, 0);
|
|
299
|
+
}
|
|
300
|
+
|
|
242
301
|
{
|
|
243
302
|
const options = makeOptions({ embeddedPdf: PDF });
|
|
244
303
|
const pageData = makePageData(15, 4 * 1024);
|