single-file-core 1.5.126 → 1.5.128
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/core/lib/processor-helper-common.js +16 -20
- package/core/util.js +34 -6
- package/doc/singlefile-archive.md +21 -12
- package/package.json +24 -24
- package/processors/compression/compression-packager.js +16 -19
- package/processors/compression/compression-router.js +1 -1
- package/processors/compression/compression.js +101 -44
- package/test/sfz-harness/README.md +3 -1
- package/test/sfz-harness/content-type-sniffing.js +83 -0
- package/test/sfz-harness/font-face-composite.js +135 -0
- package/test/sfz-harness/format-rules.js +14 -7
- package/test/sfz-harness/pages-archive.js +107 -2
- package/test/sfz-harness/pages-router.js +117 -0
- package/test/sfz-harness/stored-trigger.js +46 -1
- package/test/sfz-harness/trigger-seeds.json +46 -46
- package/vendor/zip/z-worker.js +1 -1
- package/vendor/zip/zip.js +649 -165
- package/vendor/zip/zip.min.js +1 -1
- package/zip-build/package-lock.json +4 -4
- package/zip-build/package.json +1 -1
- package/zip-build/reserved-property-names.json +23 -2
|
@@ -449,12 +449,17 @@ class ProcessorHelperCommon {
|
|
|
449
449
|
await this.processFontFaceRules(ruleData.block.children, sheetIndex, fontsDetails.layers.get("layer-" + sheetIndex + "-" + layerIndex + "-" + layerText), fonts, fontTests, stats);
|
|
450
450
|
layerIndex++;
|
|
451
451
|
} else if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
|
|
452
|
-
const
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
if (!keptRule) {
|
|
452
|
+
const fontInfo = fontsDetails.fonts.get(ruleData);
|
|
453
|
+
if (fontInfo) {
|
|
454
|
+
const ruleKey = this.getFontKey(ruleData) + " " + this.getPropertyValue(ruleData, "src");
|
|
455
|
+
if (fontsDetails.emittedFonts.has(ruleKey)) {
|
|
457
456
|
removedRules.push(cssRule);
|
|
457
|
+
} else {
|
|
458
|
+
fontsDetails.emittedFonts.add(ruleKey);
|
|
459
|
+
const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
|
|
460
|
+
if (!keptRule) {
|
|
461
|
+
removedRules.push(cssRule);
|
|
462
|
+
}
|
|
458
463
|
}
|
|
459
464
|
} else {
|
|
460
465
|
removedRules.push(cssRule);
|
|
@@ -486,28 +491,19 @@ class ProcessorHelperCommon {
|
|
|
486
491
|
layerIndex++;
|
|
487
492
|
this.getFontsDetails(doc, ruleData.block.children, sheetIndex, fontsDetails);
|
|
488
493
|
} else if (ruleData.type == "Atrule" && ruleData.name == "font-face" && ruleData.block && ruleData.block.children) {
|
|
489
|
-
const
|
|
490
|
-
|
|
491
|
-
if (!fontInfo) {
|
|
492
|
-
fontInfo = [];
|
|
493
|
-
mediaFontsDetails.fonts.set(fontKey, fontInfo);
|
|
494
|
-
}
|
|
495
|
-
mediaFontsDetails.lastRules.set(fontKey, ruleData);
|
|
494
|
+
const fontInfo = [];
|
|
495
|
+
mediaFontsDetails.fonts.set(ruleData, fontInfo);
|
|
496
496
|
const src = this.getPropertyValue(ruleData, "src");
|
|
497
497
|
if (src) {
|
|
498
498
|
const fontSources = src.match(REGEXP_URL_FUNCTION);
|
|
499
499
|
if (fontSources) {
|
|
500
|
-
|
|
501
|
-
|
|
500
|
+
fontSources.forEach(fontSource => {
|
|
501
|
+
const source = fontSource.match(REGEXP_FONT_SRC)[1];
|
|
502
502
|
if (fontInfo.includes(source)) {
|
|
503
503
|
fontInfo.splice(fontInfo.indexOf(source), 1);
|
|
504
504
|
}
|
|
505
|
-
|
|
506
|
-
ruleSources.splice(ruleSources.indexOf(source), 1);
|
|
507
|
-
}
|
|
508
|
-
ruleSources.unshift(source);
|
|
505
|
+
fontInfo.unshift(source);
|
|
509
506
|
});
|
|
510
|
-
ruleSources.forEach(source => fontInfo.push(source));
|
|
511
507
|
}
|
|
512
508
|
}
|
|
513
509
|
}
|
|
@@ -520,7 +516,7 @@ class ProcessorHelperCommon {
|
|
|
520
516
|
medias: new Map(),
|
|
521
517
|
supports: new Map(),
|
|
522
518
|
layers: new Map(),
|
|
523
|
-
|
|
519
|
+
emittedFonts: new Set()
|
|
524
520
|
};
|
|
525
521
|
}
|
|
526
522
|
|
package/core/util.js
CHANGED
|
@@ -63,6 +63,8 @@ const CONTENT_TYPE_EXTENSIONS = {
|
|
|
63
63
|
"font/collection": ".ttc"
|
|
64
64
|
};
|
|
65
65
|
const CONTENT_TYPE_OCTET_STREAM = "application/octet-stream";
|
|
66
|
+
const TRANSPORT_STREAM_SYNC_BYTE = 71;
|
|
67
|
+
const TRANSPORT_STREAM_PACKET_SIZE = 188;
|
|
66
68
|
const CONTENT_TYPES_HTML = ["text/html", "application/xhtml+xml"];
|
|
67
69
|
const EXPECTED_TYPES_MEDIA = ["font", "image", "video", "audio"];
|
|
68
70
|
|
|
@@ -295,11 +297,11 @@ function getInstance(utilOptions) {
|
|
|
295
297
|
} catch (error) {
|
|
296
298
|
// ignored
|
|
297
299
|
}
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
300
|
+
const guessedContentType = guessMIMEType(options.expectedType, buffer);
|
|
301
|
+
if (guessedContentType) {
|
|
302
|
+
contentType = guessedContentType;
|
|
303
|
+
} else if (!contentType || (contentType == CONTENT_TYPE_OCTET_STREAM && options.asBinary)) {
|
|
304
|
+
contentType = options.contentType ? options.contentType : options.asBinary ? CONTENT_TYPE_OCTET_STREAM : "";
|
|
303
305
|
}
|
|
304
306
|
if (!charset && options.charset) {
|
|
305
307
|
charset = options.charset;
|
|
@@ -403,6 +405,24 @@ function guessMIMEType(expectedType, buffer) {
|
|
|
403
405
|
if (compareBytes([255, 255, 255], [255, 216, 255])) {
|
|
404
406
|
return "image/jpeg";
|
|
405
407
|
}
|
|
408
|
+
if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 97, 118, 105, 102]) ||
|
|
409
|
+
compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 97, 118, 105, 115])) {
|
|
410
|
+
return "image/avif";
|
|
411
|
+
}
|
|
412
|
+
if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 105, 99]) ||
|
|
413
|
+
compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 105, 120]) ||
|
|
414
|
+
compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 118, 99]) ||
|
|
415
|
+
compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 118, 120])) {
|
|
416
|
+
return "image/heic";
|
|
417
|
+
}
|
|
418
|
+
if (compareBytes([255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 12, 74, 88, 76, 32, 13, 10, 135, 10]) ||
|
|
419
|
+
compareBytes([255, 255], [255, 10])) {
|
|
420
|
+
return "image/jxl";
|
|
421
|
+
}
|
|
422
|
+
if (compareBytes([255, 255, 255, 255], [73, 73, 42, 0]) ||
|
|
423
|
+
compareBytes([255, 255, 255, 255], [77, 77, 0, 42])) {
|
|
424
|
+
return "image/tiff";
|
|
425
|
+
}
|
|
406
426
|
}
|
|
407
427
|
if (expectedType == "font") {
|
|
408
428
|
if (compareBytes([0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 255, 255],
|
|
@@ -444,7 +464,7 @@ function guessMIMEType(expectedType, buffer) {
|
|
|
444
464
|
if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 51, 103])) {
|
|
445
465
|
return "video/3gpp";
|
|
446
466
|
}
|
|
447
|
-
if (
|
|
467
|
+
if (isTransportStream()) {
|
|
448
468
|
return "video/mp2t";
|
|
449
469
|
}
|
|
450
470
|
}
|
|
@@ -475,6 +495,14 @@ function guessMIMEType(expectedType, buffer) {
|
|
|
475
495
|
}
|
|
476
496
|
}
|
|
477
497
|
|
|
498
|
+
function isTransportStream() {
|
|
499
|
+
const value = new Uint8Array(buffer);
|
|
500
|
+
return value.length > TRANSPORT_STREAM_PACKET_SIZE * 2 &&
|
|
501
|
+
value[0] == TRANSPORT_STREAM_SYNC_BYTE &&
|
|
502
|
+
value[TRANSPORT_STREAM_PACKET_SIZE] == TRANSPORT_STREAM_SYNC_BYTE &&
|
|
503
|
+
value[TRANSPORT_STREAM_PACKET_SIZE * 2] == TRANSPORT_STREAM_SYNC_BYTE;
|
|
504
|
+
}
|
|
505
|
+
|
|
478
506
|
function compareBytes(mask, pattern) {
|
|
479
507
|
let patternMatch = true, index = 0;
|
|
480
508
|
if (buffer.byteLength >= pattern.length) {
|
|
@@ -357,7 +357,7 @@ face adds, then the regions the PNG face adds.
|
|
|
357
357
|
| `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
|
|
358
358
|
| `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the appended-data budget (§5.2) or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
|
|
359
359
|
| `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
|
|
360
|
-
| `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag
|
|
360
|
+
| `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag left clear as on every other ASCII name — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
|
|
361
361
|
| `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
|
|
362
362
|
| `pdf-central-record` | ZIP | PDF face with HTML | The central-directory record for `page.pdf`, injected *before* the writer's own central directory. The start of the central directory is the one place a record can be added without moving any offset the writer already committed, and it makes `page.pdf` the first entry ZIP tools list (§6). |
|
|
363
363
|
| `png-signature · IHDR` | PNG | PNG face | The 8-byte PNG signature and the `IHDR` chunk declaring the source image's dimensions — the first 33 bytes of the file. |
|
|
@@ -1291,18 +1291,26 @@ How a name is encoded is ZIP's own business, not this format's: bit 11 of the ge
|
|
|
1291
1291
|
purpose bit flag selects UTF-8, and its absence selects the legacy code page. This
|
|
1292
1292
|
document adds two requirements to that and specifies nothing else about it.
|
|
1293
1293
|
|
|
1294
|
-
**A writer MUST set bit 11
|
|
1295
|
-
it
|
|
1296
|
-
|
|
1294
|
+
**A writer MUST set bit 11 whenever a name or a comment needs it**, and the rule for
|
|
1295
|
+
when it does is ZIP's, not this format's: an encoded name or comment holding a byte
|
|
1296
|
+
outside printable ASCII needs it, one holding only printable ASCII does not, since the
|
|
1297
|
+
two encodings agree there. Control characters count as needing it, the legacy code page
|
|
1298
|
+
mapping them to graphic characters rather than to themselves. Setting it on names that
|
|
1299
|
+
do not need it is allowed and used to be required here; it was dropped because readers
|
|
1300
|
+
disagree about the flag more than they disagree about ASCII, so the safest name is the
|
|
1301
|
+
one that does not exercise the question. A writer MUST NOT set it on a name it then
|
|
1302
|
+
encodes in the legacy code page, which is the one combination that is simply wrong.
|
|
1297
1303
|
|
|
1298
1304
|
**A reader MUST honor the flag** rather than assume one encoding, and MUST expect to
|
|
1299
|
-
meet a clear one
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
|
|
1303
|
-
|
|
1304
|
-
|
|
1305
|
-
|
|
1305
|
+
meet a clear one — which, in an archive from the reference writer, is most of them:
|
|
1306
|
+
that writer percent-encodes every name it produces, so every name is printable ASCII
|
|
1307
|
+
and carries no flag, while an entry comment holding the original URL of a resource can
|
|
1308
|
+
carry one when that URL is not ASCII. The hand-built `page.pdf` records (§3.1, §6) are
|
|
1309
|
+
the only ones the reference writer does not produce through its ZIP writer, and they
|
|
1310
|
+
follow the same rule: `page.pdf` is ASCII, so they carry no flag either, and no entry
|
|
1311
|
+
in the archive is decoded differently from the rest. Archives written before this rule
|
|
1312
|
+
was relaxed carry the flag on every entry instead. Both decode identically, which is
|
|
1313
|
+
the point, but a reader that hardcodes either answer meets the other one eventually.
|
|
1306
1314
|
|
|
1307
1315
|
A name is not a path. §7.3's rule that entry names are untrusted applies to the decoded
|
|
1308
1316
|
name, and decoding is the step before that check, not a substitute for it.
|
|
@@ -1399,7 +1407,7 @@ pages can stop at the first row; the files it produces are accepted by every rea
|
|
|
1399
1407
|
wrapper start tag chosen for the PDF payload (§5.1), the hand-built `page.pdf`
|
|
1400
1408
|
local file header, the PDF document, the wrapper end tag, and record the local
|
|
1401
1409
|
header's absolute position; then resume the prologue. The reference writer's
|
|
1402
|
-
header declares version 2.0,
|
|
1410
|
+
header declares version 2.0, no general purpose bit flag, method STORE, the
|
|
1403
1411
|
build's modification date in DOS form, the precomputed CRC-32, the document's
|
|
1404
1412
|
length as both sizes, and no extra field; its central record adds a Unix
|
|
1405
1413
|
"made by" version and external attributes of a regular file, mode 0644.
|
|
@@ -1892,6 +1900,7 @@ predicts.
|
|
|
1892
1900
|
| August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
|
|
1893
1901
|
| August 2026 | Core 1.5.119: the inlined ZIP library is built ASCII-only, and §2.1 now requires it of any bootstrap. Its CP437 table had been emitted as literal characters, which the page re-decoded as windows-1252, growing the table from 256 entries to 508 and shifting every lookup by 60 — so the one entry read without the UTF-8 flag, `page.pdf`, came back mangled and no archive with a PDF face extracted in any engine (§5.8) |
|
|
1894
1902
|
| August 2026 | Core 1.5.120: the hand-built `page.pdf` records set the language encoding flag, like every entry the ZIP writer produces (§5.8). Its name is ASCII, so no decoded name changes; what changes is that no entry in an archive is read through CP437 any more, closing the path the 1.5.119 defect surfaced on |
|
|
1903
|
+
| September 2026 | §5.8 no longer requires bit 11 on every entry, deferring to ZIP's own rule: the flag is set when a name or a comment holds a byte outside printable ASCII, and left clear otherwise, because readers disagree about the flag more than they disagree about ASCII. The reference writer's names are all percent-encoded, so in practice none of them carries it now, and the hand-built `page.pdf` records follow the writer instead of overriding it — reversing the 1.5.120 row below, whose reason was that `page.pdf` would otherwise be the only entry read through the legacy path. It no longer is: every name in the archive takes the same path again, the other one |
|
|
1895
1904
|
| September 2026 | Core 1.5.126: the appended-data budget becomes the `maxAppendedDataLength` writer option and its default drops from 65535 to 16361 bytes, so the EOCD record stays inside libarchive's scan and `bsdtar` opens archives it used to reject (§5.2, §8.1). The 65535-byte comment ceiling is now a separate limit, stated in §4.2: a budget raised past it produces a run that cannot be declared |
|
|
1896
1905
|
|
|
1897
1906
|
This document was itself revised in August 2026, against core 1.5.108, after several
|
package/package.json
CHANGED
|
@@ -1,26 +1,26 @@
|
|
|
1
1
|
{
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
2
|
+
"name": "single-file-core",
|
|
3
|
+
"version": "1.5.128",
|
|
4
|
+
"description": "SingleFile Core",
|
|
5
|
+
"author": "Gildas Lormeau",
|
|
6
|
+
"license": "AGPL-3.0-or-later",
|
|
7
|
+
"scripts": {
|
|
8
|
+
"test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/font-face-composite.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/pages-router.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/content-type-sniffing.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
|
|
9
|
+
"bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
|
|
10
|
+
"bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
|
|
11
|
+
"bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
|
|
12
|
+
"bump-commit": "git commit -m \"bump up version\" package.json package-lock.json"
|
|
13
|
+
},
|
|
14
|
+
"repository": {
|
|
15
|
+
"type": "git",
|
|
16
|
+
"url": "git+https://github.com/gildas-lormeau/single-file-core.git"
|
|
17
|
+
},
|
|
18
|
+
"bugs": {
|
|
19
|
+
"url": "https://github.com/gildas-lormeau/single-file-core/issues"
|
|
20
|
+
},
|
|
21
|
+
"homepage": "https://github.com/gildas-lormeau/single-file-core#readme",
|
|
22
|
+
"devDependencies": {
|
|
23
|
+
"@eslint/js": "^9.39.5",
|
|
24
|
+
"eslint": "^10.9.1"
|
|
25
|
+
}
|
|
26
26
|
}
|
|
@@ -32,7 +32,8 @@ import {
|
|
|
32
32
|
} from "./../../vendor/zip/zip.js";
|
|
33
33
|
import {
|
|
34
34
|
createArchive,
|
|
35
|
-
escapeHTML
|
|
35
|
+
escapeHTML,
|
|
36
|
+
PROCESS_OPTION_NAMES
|
|
36
37
|
} from "./compression.js";
|
|
37
38
|
|
|
38
39
|
const browser = globalThis.browser;
|
|
@@ -46,6 +47,13 @@ const TOC_STYLE = "body{font-family:system-ui,sans-serif;margin:2em auto;max-wid
|
|
|
46
47
|
"summary{cursor:pointer;font-weight:bold;margin:.5em 0}" +
|
|
47
48
|
"details{padding-left:1em}ul{margin:.25em 0;padding-left:1.5em}" +
|
|
48
49
|
"@media(prefers-color-scheme:dark){body{background-color:#111;color:#eee}a{color:#8ab4f8}a:visited{color:#c58af9}}";
|
|
50
|
+
const ARCHIVE_EXCLUDED_OPTION_NAMES = [
|
|
51
|
+
"createRootDirectory",
|
|
52
|
+
"disableCompression",
|
|
53
|
+
"insertTextBody",
|
|
54
|
+
"password",
|
|
55
|
+
"url"
|
|
56
|
+
];
|
|
49
57
|
const COMMENT_HEADER = "Page saved with SingleFile";
|
|
50
58
|
const SYMLINK_UNIX_MODE = 0o120777;
|
|
51
59
|
|
|
@@ -80,18 +88,11 @@ async function createPagesArchive(pages, options) {
|
|
|
80
88
|
};
|
|
81
89
|
const archiveOptions = {
|
|
82
90
|
url: pages[0].url,
|
|
83
|
-
multiPageArchive: true
|
|
84
|
-
selfExtractingArchive: options.selfExtractingArchive,
|
|
85
|
-
extractDataFromPage: options.extractDataFromPage,
|
|
86
|
-
preventAppendedData: options.preventAppendedData,
|
|
87
|
-
declareAppendedData: options.declareAppendedData,
|
|
88
|
-
embeddedPdf: options.embeddedPdf,
|
|
89
|
-
embeddedImage: options.embeddedImage,
|
|
90
|
-
includeBOM: options.includeBOM,
|
|
91
|
-
insertMetaCSP: options.insertMetaCSP,
|
|
92
|
-
insertCanonicalLink: options.insertCanonicalLink,
|
|
93
|
-
insertMetaNoIndex: options.insertMetaNoIndex
|
|
91
|
+
multiPageArchive: true
|
|
94
92
|
};
|
|
93
|
+
PROCESS_OPTION_NAMES
|
|
94
|
+
.filter(name => !ARCHIVE_EXCLUDED_OPTION_NAMES.includes(name) && !(name in archiveOptions))
|
|
95
|
+
.forEach(name => archiveOptions[name] = options[name]);
|
|
95
96
|
const writtenEntries = options.dedupPages ? new Map() : undefined;
|
|
96
97
|
const aliases = {};
|
|
97
98
|
const blob = await createArchive(pageData, archiveOptions, options.zipScript, async zipWriter => {
|
|
@@ -100,16 +101,12 @@ async function createPagesArchive(pages, options) {
|
|
|
100
101
|
const zipReader = new ZipReader(new Uint8ArrayReader(await pages[pageIndex].getData()));
|
|
101
102
|
for (const entry of await zipReader.getEntries()) {
|
|
102
103
|
const filename = pagePath + entry.filename;
|
|
103
|
-
const rawData = await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
|
|
104
|
+
const rawData = entry.directory ? undefined : await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
|
|
104
105
|
const canonicalFilename = writtenEntries && findDuplicate(writtenEntries, filename, entry, rawData);
|
|
105
106
|
if (canonicalFilename === undefined) {
|
|
106
|
-
await zipWriter.add(filename, new Uint8ArrayReader(rawData), {
|
|
107
|
+
await zipWriter.add(filename, entry.directory ? null : new Uint8ArrayReader(rawData), {
|
|
107
108
|
passThrough: true,
|
|
108
|
-
|
|
109
|
-
uncompressedSize: entry.uncompressedSize,
|
|
110
|
-
crc32: entry.crc32,
|
|
111
|
-
comment: entry.comment,
|
|
112
|
-
lastModDate: entry.lastModDate
|
|
109
|
+
entry
|
|
113
110
|
});
|
|
114
111
|
} else {
|
|
115
112
|
aliases[filename] = canonicalFilename;
|
|
@@ -202,7 +202,7 @@ async function router(content, { extract, display }) {
|
|
|
202
202
|
function parseRoute() {
|
|
203
203
|
const hash = location.hash;
|
|
204
204
|
const routed = !hash || hash.startsWith(ROUTE_PREFIX);
|
|
205
|
-
let path = pages[0].path;
|
|
205
|
+
let path = routed && tocEntry ? TOC_ROUTE : pages[0].path;
|
|
206
206
|
let fragment;
|
|
207
207
|
if (routed && hash) {
|
|
208
208
|
({ path, fragment } = parseRouteHash(hash));
|
|
@@ -40,9 +40,7 @@ import {
|
|
|
40
40
|
router
|
|
41
41
|
} from "./compression-router.js";
|
|
42
42
|
|
|
43
|
-
const { Blob, fetch, TextEncoder,
|
|
44
|
-
|
|
45
|
-
const TEXT_DECODER = new TextDecoder("windows-1252");
|
|
43
|
+
const { Blob, fetch, TextEncoder, DOMParser } = globalThis;
|
|
46
44
|
|
|
47
45
|
const COMPRESSIBLE_CONTENT_TYPES = ["application/javascript", "application/x-javascript", "application/ecmascript", "application/json", "application/ld+json", "application/manifest+json", "application/xml", "application/xhtml+xml", "application/rss+xml", "application/atom+xml", "image/svg+xml"];
|
|
48
46
|
const TEXT_CONTENT_TYPE_PREFIX = "text/";
|
|
@@ -63,19 +61,20 @@ const EMBEDDED_DATA_TAGS = [
|
|
|
63
61
|
...EXTRA_DATA_TAGS,
|
|
64
62
|
];
|
|
65
63
|
const DATA_IDENTIFIER = "sfz-data";
|
|
66
|
-
const
|
|
67
|
-
|
|
68
|
-
[
|
|
69
|
-
[
|
|
70
|
-
[
|
|
71
|
-
[
|
|
72
|
-
[
|
|
73
|
-
[
|
|
74
|
-
[
|
|
64
|
+
const TAG_NAME_TERMINATORS = "\t\n\f\r />";
|
|
65
|
+
const EXTRA_DATA_PATTERNS = [
|
|
66
|
+
[["<script"], ["</script", TAG_NAME_TERMINATORS]],
|
|
67
|
+
[["<style"], ["</style", TAG_NAME_TERMINATORS]],
|
|
68
|
+
[["<noframes"], ["</noframes", TAG_NAME_TERMINATORS]],
|
|
69
|
+
[["<noembed"], ["</noembed", TAG_NAME_TERMINATORS]],
|
|
70
|
+
[["<iframe"], ["</iframe", TAG_NAME_TERMINATORS]],
|
|
71
|
+
[["<xmp"], ["</xmp", TAG_NAME_TERMINATORS]],
|
|
72
|
+
[["<![CDATA["], ["]]>"]],
|
|
73
|
+
[["<plaintext"], ["</plaintext", TAG_NAME_TERMINATORS]]
|
|
75
74
|
];
|
|
76
|
-
const
|
|
77
|
-
[
|
|
78
|
-
...
|
|
75
|
+
const EMBEDDED_DATA_PATTERNS = [
|
|
76
|
+
[["<!--"], ["-->"], ["--!>"], ["<!-", undefined, true]],
|
|
77
|
+
...EXTRA_DATA_PATTERNS,
|
|
79
78
|
];
|
|
80
79
|
const CRC32_TABLE = new Uint32Array(256).map((_, indexTable) => {
|
|
81
80
|
let crc = indexTable;
|
|
@@ -106,7 +105,6 @@ const CENTRAL_FILE_HEADER_SIGNATURE = 0x02014b50;
|
|
|
106
105
|
const END_OF_CENTRAL_DIR_SIGNATURE = 0x06054b50;
|
|
107
106
|
const ZIP64_END_OF_CENTRAL_DIR_SIGNATURE = 0x06064b50;
|
|
108
107
|
const ZIP64_END_OF_CENTRAL_DIR_LOCATOR_SIGNATURE = 0x07064b50;
|
|
109
|
-
const LANGUAGE_ENCODING_FLAG = 0x0800;
|
|
110
108
|
|
|
111
109
|
const browser = globalThis.browser;
|
|
112
110
|
|
|
@@ -229,18 +227,14 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
|
|
|
229
227
|
const lfCodes = [];
|
|
230
228
|
let crc32 = -1;
|
|
231
229
|
if (!options.extractDataFromPageTags || options.extractDataFromPageTags[0] != "<plaintext>") {
|
|
232
|
-
const
|
|
230
|
+
const zipData = data.subarray(startOffset);
|
|
233
231
|
if (options.extractDataFromPageTags) {
|
|
234
232
|
const tagIndex = getExtraDataTagIndex(options.extractDataFromPageTags);
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, tagIndex + 1);
|
|
238
|
-
}
|
|
239
|
-
} else {
|
|
240
|
-
const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[0];
|
|
241
|
-
if (textContent.match(startRegExp) || textContent.match(endRegExp)) {
|
|
242
|
-
return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions);
|
|
233
|
+
if (containsDataPattern(zipData, EXTRA_DATA_PATTERNS[tagIndex])) {
|
|
234
|
+
return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, tagIndex + 1);
|
|
243
235
|
}
|
|
236
|
+
} else if (containsDataPattern(zipData, EMBEDDED_DATA_PATTERNS[0])) {
|
|
237
|
+
return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions);
|
|
244
238
|
}
|
|
245
239
|
}
|
|
246
240
|
if (options.extractDataFromPage) {
|
|
@@ -337,9 +331,7 @@ function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, opt
|
|
|
337
331
|
const tail = pageContent.slice(zipDataEnd - WRAPPER_PATTERN_WINDOW_LENGTH, zipDataEnd + COMMENT_LENGTH_FIELD_LENGTH);
|
|
338
332
|
new DataView(tail.buffer).setUint16(WRAPPER_PATTERN_WINDOW_LENGTH, appendedDataLength, true);
|
|
339
333
|
const tagIndex = options.extractDataFromPageTags ? getExtraDataTagIndex(options.extractDataFromPageTags) + 1 : 0;
|
|
340
|
-
|
|
341
|
-
const tailText = TEXT_DECODER.decode(tail);
|
|
342
|
-
return !tailText.match(startRegExp) && !tailText.match(endRegExp);
|
|
334
|
+
return !containsDataPattern(tail, EMBEDDED_DATA_PATTERNS[tagIndex]);
|
|
343
335
|
}
|
|
344
336
|
|
|
345
337
|
function getCRC32(data, indexData = 0) {
|
|
@@ -365,7 +357,6 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
|
|
|
365
357
|
const localHeaderView = new DataView(localHeader.buffer);
|
|
366
358
|
localHeaderView.setUint32(0, LOCAL_FILE_HEADER_SIGNATURE, true);
|
|
367
359
|
localHeaderView.setUint16(4, 20, true);
|
|
368
|
-
localHeaderView.setUint16(6, LANGUAGE_ENCODING_FLAG, true);
|
|
369
360
|
localHeaderView.setUint16(10, dosTime, true);
|
|
370
361
|
localHeaderView.setUint16(12, dosDate, true);
|
|
371
362
|
localHeaderView.setUint32(14, crc32, true);
|
|
@@ -378,7 +369,6 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
|
|
|
378
369
|
centralRecordView.setUint32(0, CENTRAL_FILE_HEADER_SIGNATURE, true);
|
|
379
370
|
centralRecordView.setUint16(4, 0x0300, true);
|
|
380
371
|
centralRecordView.setUint16(6, 20, true);
|
|
381
|
-
centralRecordView.setUint16(8, LANGUAGE_ENCODING_FLAG, true);
|
|
382
372
|
centralRecordView.setUint16(12, dosTime, true);
|
|
383
373
|
centralRecordView.setUint16(14, dosDate, true);
|
|
384
374
|
centralRecordView.setUint32(16, crc32, true);
|
|
@@ -552,8 +542,7 @@ function getStartHTMLArray(pageData, options, lastModDate, startTag = "") {
|
|
|
552
542
|
const embeddedPdf = new Uint8Array(options.embeddedPdf);
|
|
553
543
|
pdfEntry = options.preventEmbeddedPdfEntry ? undefined : getPDFEntry(embeddedPdf, lastModDate);
|
|
554
544
|
const localHeader = pdfEntry ? pdfEntry.localHeader : new Uint8Array(0);
|
|
555
|
-
const
|
|
556
|
-
const pdfTagIndex = findEmbeddedDataTagIndex(embeddedPdfText);
|
|
545
|
+
const pdfTagIndex = findEmbeddedDataTagIndex(concatArrays(localHeader, embeddedPdf));
|
|
557
546
|
if (pdfTagIndex == -1) {
|
|
558
547
|
dropUnhiddenFace(options, "embeddedPdf", EMBEDDED_PDF_LABEL);
|
|
559
548
|
pdfEntry = undefined;
|
|
@@ -629,12 +618,11 @@ function getExtraDataTagIndex(extractDataFromPageTags) {
|
|
|
629
618
|
return tagIndex;
|
|
630
619
|
}
|
|
631
620
|
|
|
632
|
-
function findExtraDataTags(
|
|
633
|
-
const regExpsTag = EXTRA_DATA_REGEXPS[indexExtractDataFromPageTags];
|
|
621
|
+
function findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags = 0) {
|
|
634
622
|
const plaintextTag = EXTRA_DATA_TAGS[indexExtractDataFromPageTags][0] == "<plaintext>";
|
|
635
|
-
const matchTag = !plaintextTag && (
|
|
623
|
+
const matchTag = !plaintextTag && containsDataPattern(zipData, EXTRA_DATA_PATTERNS[indexExtractDataFromPageTags]);
|
|
636
624
|
if (matchTag) {
|
|
637
|
-
return findExtraDataTags(
|
|
625
|
+
return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags + 1);
|
|
638
626
|
} else {
|
|
639
627
|
options.extractDataFromPageTags = EXTRA_DATA_TAGS[indexExtractDataFromPageTags];
|
|
640
628
|
if (options.extractDataFromPageTags[0] == "<plaintext>") {
|
|
@@ -644,24 +632,93 @@ function findExtraDataTags(textContent, pageData, options, script, entriesData,
|
|
|
644
632
|
}
|
|
645
633
|
}
|
|
646
634
|
|
|
647
|
-
function findEmbeddedDataTagIndex(
|
|
648
|
-
const tagIndex =
|
|
635
|
+
function findEmbeddedDataTagIndex(data, fromIndex = 0) {
|
|
636
|
+
const tagIndex = EMBEDDED_DATA_PATTERNS.slice(fromIndex, -1).findIndex(patterns => !containsDataPattern(data, patterns));
|
|
649
637
|
return tagIndex == -1 ? -1 : tagIndex + fromIndex;
|
|
650
638
|
}
|
|
651
639
|
|
|
640
|
+
function containsDataPattern(data, patterns) {
|
|
641
|
+
const patternsByCharCode = new Map();
|
|
642
|
+
for (const pattern of patterns) {
|
|
643
|
+
const [text, , atEnd] = pattern;
|
|
644
|
+
if (atEnd) {
|
|
645
|
+
if (matchesDataPatternAt(data, pattern, data.length - text.length)) {
|
|
646
|
+
return true;
|
|
647
|
+
}
|
|
648
|
+
} else {
|
|
649
|
+
const charCode = text.charCodeAt(0);
|
|
650
|
+
indexPatternByCharCode(patternsByCharCode, charCode, pattern);
|
|
651
|
+
const alternateCharCode = getAlternateCharCode(charCode);
|
|
652
|
+
if (alternateCharCode != -1) {
|
|
653
|
+
indexPatternByCharCode(patternsByCharCode, alternateCharCode, pattern);
|
|
654
|
+
}
|
|
655
|
+
}
|
|
656
|
+
}
|
|
657
|
+
for (const [charCode, candidates] of patternsByCharCode) {
|
|
658
|
+
for (let index = data.indexOf(charCode); index != -1; index = data.indexOf(charCode, index + 1)) {
|
|
659
|
+
for (let indexCandidate = 0; indexCandidate < candidates.length; indexCandidate++) {
|
|
660
|
+
if (matchesDataPatternAt(data, candidates[indexCandidate], index)) {
|
|
661
|
+
return true;
|
|
662
|
+
}
|
|
663
|
+
}
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
return false;
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
function indexPatternByCharCode(patternsByCharCode, charCode, pattern) {
|
|
670
|
+
if (!patternsByCharCode.has(charCode)) {
|
|
671
|
+
patternsByCharCode.set(charCode, []);
|
|
672
|
+
}
|
|
673
|
+
patternsByCharCode.get(charCode).push(pattern);
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
function matchesDataPatternAt(data, [text, terminators], index) {
|
|
677
|
+
const textLength = text.length;
|
|
678
|
+
if (index < 0 || index + textLength > data.length) {
|
|
679
|
+
return false;
|
|
680
|
+
}
|
|
681
|
+
for (let indexText = 0; indexText < textLength; indexText++) {
|
|
682
|
+
const charCode = text.charCodeAt(indexText);
|
|
683
|
+
const code = data[index + indexText];
|
|
684
|
+
if (code != charCode && code != getAlternateCharCode(charCode)) {
|
|
685
|
+
return false;
|
|
686
|
+
}
|
|
687
|
+
}
|
|
688
|
+
return terminators === undefined || isTerminatorCode(terminators, data[index + textLength]);
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
function isTerminatorCode(terminators, code) {
|
|
692
|
+
return code !== undefined && terminators.includes(String.fromCharCode(code));
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
function getAlternateCharCode(charCode) {
|
|
696
|
+
const lowerCharCode = charCode | 0x20;
|
|
697
|
+
return lowerCharCode >= 0x61 && lowerCharCode <= 0x7a ? charCode ^ 0x20 : -1;
|
|
698
|
+
}
|
|
699
|
+
|
|
700
|
+
function concatArrays(...arrays) {
|
|
701
|
+
const result = new Uint8Array(arrays.reduce((length, array) => length + array.length, 0));
|
|
702
|
+
let offset = 0;
|
|
703
|
+
arrays.forEach(array => {
|
|
704
|
+
result.set(array, offset);
|
|
705
|
+
offset += array.length;
|
|
706
|
+
});
|
|
707
|
+
return result;
|
|
708
|
+
}
|
|
709
|
+
|
|
652
710
|
function getImageHTMLChunk(pageData, options, lastModDate) {
|
|
653
|
-
const
|
|
654
|
-
|
|
655
|
-
let tagIndex = findEmbeddedDataTagIndex(embeddedImageText);
|
|
711
|
+
const embeddedImageData = concatArrays(getEmbeddedImageData(options.embeddedImage), new Uint8Array(4), PNG_ZIP_CHUNK_TYPE_KEYWORD);
|
|
712
|
+
let tagIndex = findEmbeddedDataTagIndex(embeddedImageData);
|
|
656
713
|
while (tagIndex != -1) {
|
|
657
714
|
const [startTag, endTag] = EMBEDDED_DATA_TAGS[tagIndex];
|
|
658
715
|
const startHTMLData = getStartHTMLArray(pageData, options, lastModDate, startTag);
|
|
659
716
|
const htmlData = new Uint8Array([...getLength(startHTMLData.htmlArray.length + 4), ...[0x74, 0x45, 0x58, 0x74, 0x50, 0x4e, 0x47, 0], ...startHTMLData.htmlArray]);
|
|
660
717
|
const htmlDataCRC = getCRC32(htmlData, 4);
|
|
661
|
-
const
|
|
718
|
+
const wrappedData = concatArrays(htmlDataCRC, embeddedImageData);
|
|
662
719
|
if ((tagIndex == 0 && (htmlDataCRC[0] == 0x3e || (htmlDataCRC[0] == 0x2d && htmlDataCRC[1] == 0x3e))) ||
|
|
663
|
-
findEmbeddedDataTagIndex(
|
|
664
|
-
tagIndex = findEmbeddedDataTagIndex(
|
|
720
|
+
findEmbeddedDataTagIndex(wrappedData, tagIndex) != tagIndex) {
|
|
721
|
+
tagIndex = findEmbeddedDataTagIndex(embeddedImageData, tagIndex + 1);
|
|
665
722
|
} else {
|
|
666
723
|
return { endTag, startHTMLData, htmlData, htmlDataCRC };
|
|
667
724
|
}
|
|
@@ -34,7 +34,9 @@ any check failed.
|
|
|
34
34
|
| `css-property-filter.js` | That the declaration filter keeps a property css-tree's dictionary does not know (`stop-color`, `flood-opacity`, anything newer than the pinned build) and still drops a genuinely invalid value. |
|
|
35
35
|
| `adopted-stylesheets-hook.js` | That the page-world hook answers the adopted-stylesheets request for a CLOSED shadow root, which its host does not expose. |
|
|
36
36
|
| `inlined-functions.js` | That a function serialized into a self-extracting archive names nothing outside itself. An import survives bundling and still reads correctly, and the archive then throws a bare `ReferenceError` and renders nothing. |
|
|
37
|
-
| `pages-archive.js` | That `createPagesArchive` packs several pages into one archive correctly: the first page at the root and the others in folders, the manifest, the symlink a deduplicated entry leaves behind, and the escaping of crawled titles in both tables of contents. |
|
|
37
|
+
| `pages-archive.js` | That `createPagesArchive` packs several pages into one archive correctly: the first page at the root and the others in folders, the manifest, the symlink a deduplicated entry leaves behind, and the escaping of crawled titles in both tables of contents. Also that the options reaching the archive writer are derived from `PROCESS_OPTION_NAMES` rather than hand-listed — a hand copy dropped `maxAppendedDataLength` for months — and that `password` stays out of that derivation, since forwarding it would make the writer withhold the prologue's title as if the archive were encrypted while the table of contents and every entry comment still rode in that same cleartext prologue. |
|
|
38
|
+
| `pages-router.js` | That the router opens a multi-page archive on the page a reader expects: the table of contents when the archive stores one, the first page when it does not, and the page a route in the hash names whatever else is stored. The router is inlined into the archive as source text and only ever ran inside a saved page, so nothing drove it before; the table of contents shipped stored, routable and unreachable. |
|
|
39
|
+
| `content-type-sniffing.js` | That a magic-byte match beats the `Content-Type` header, and that the header survives when nothing matches. science.org serves its woff2 files as `text/plain`, which used to be trusted, so the fonts were embedded as `data:text/plain` and the SFZ writer deflated a file that is already Brotli-compressed. Two cases guard the rules themselves rather than the outcome: a video whose first byte is `G` must not be relabelled `video/mp2t`, and the generic `mif1` HEIF brand must identify nothing, because an AVIF can carry it and calling it HEIC would name a format no browser decodes. |
|
|
38
40
|
| `entry-compression.js` | That an entry is deflated or stored on the content type the server sent, not on the extension alone — a module served as `text/javascript` from a `.ts` URL used to go in uncompressed — and that an unrecognized `application/octet-stream` still stays stored. |
|
|
39
41
|
| `filename-max-length.js` | That `formatFilename` counts the ellipsis as well as the extension in the budget it truncates to, so a filename at `filenameMaxLength` stays at it, and that a limit shorter than the extension does not reach `Blob.slice` with a negative start. |
|
|
40
42
|
| `filename-characters.js` | That `getValidFilename` maps a full-width lookalike one character at a time — `C++` used to be saved as `C+` — while a run of characters with no lookalike still collapses to a single replacement. |
|