single-file-core 1.5.126 → 1.5.128
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/core/lib/processor-helper-common.js +16 -20
- package/core/util.js +34 -6
- package/doc/singlefile-archive.md +21 -12
- package/package.json +24 -24
- package/processors/compression/compression-packager.js +16 -19
- package/processors/compression/compression-router.js +1 -1
- package/processors/compression/compression.js +101 -44
- package/test/sfz-harness/README.md +3 -1
- package/test/sfz-harness/content-type-sniffing.js +83 -0
- package/test/sfz-harness/font-face-composite.js +135 -0
- package/test/sfz-harness/format-rules.js +14 -7
- package/test/sfz-harness/pages-archive.js +107 -2
- package/test/sfz-harness/pages-router.js +117 -0
- package/test/sfz-harness/stored-trigger.js +46 -1
- package/test/sfz-harness/trigger-seeds.json +46 -46
- package/vendor/zip/z-worker.js +1 -1
- package/vendor/zip/zip.js +649 -165
- package/vendor/zip/zip.min.js +1 -1
- package/zip-build/package-lock.json +4 -4
- package/zip-build/package.json +1 -1
- package/zip-build/reserved-property-names.json +23 -2
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
// `core/util.js` used to trust any Content-Type the server sent and sniff the bytes only when the
|
|
2
|
+
// header was missing or `application/octet-stream`. science.org serves its woff2 files as
|
|
3
|
+
// `text/plain;charset=UTF-8` — 12 of them — and every one was embedded as `data:text/plain`, which
|
|
4
|
+
// costs twice: a browser with no `format()` hint in the `@font-face` rule has nothing left to
|
|
5
|
+
// identify the font by, and the SFZ writer deflates a file that is already Brotli-compressed
|
|
6
|
+
// because it decides compression from the content type.
|
|
7
|
+
//
|
|
8
|
+
// The rule now: a magic-byte match wins over the header, and the header is kept only when nothing
|
|
9
|
+
// matches. That raises the bar for the sniffer's own rules, which is why `video/mp2t` no longer
|
|
10
|
+
// matches on a single `0x47` byte — as a last resort behind a missing header that was tolerable,
|
|
11
|
+
// as an override of a correct header it would relabel any video whose first byte is `G`. It now
|
|
12
|
+
// wants the sync byte at the 188-byte packet stride, and the case below is the regression guard.
|
|
13
|
+
/* global Response */
|
|
14
|
+
|
|
15
|
+
import "./dom-stub.js";
|
|
16
|
+
// util.js pulls in the page-world hooks, which register a document listener as they are evaluated.
|
|
17
|
+
// None of it is exercised here; the stubs exist so that importing util.js is possible at all
|
|
18
|
+
globalThis.window = globalThis.window || {};
|
|
19
|
+
globalThis.document = globalThis.document || {};
|
|
20
|
+
globalThis.Document = globalThis.Document || class { };
|
|
21
|
+
globalThis.MutationObserver = globalThis.MutationObserver || class {
|
|
22
|
+
observe() { }
|
|
23
|
+
};
|
|
24
|
+
const { getInstance } = await import("./../../core/util.js");
|
|
25
|
+
|
|
26
|
+
const FONT_URL = "https://example.com/font";
|
|
27
|
+
const WOFF2 = bytes([0x77, 0x4F, 0x46, 0x32], 64);
|
|
28
|
+
const PNG = bytes([0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A], 64);
|
|
29
|
+
const AVIF = bytes([0, 0, 0, 0x20, 0x66, 0x74, 0x79, 0x70, 0x61, 0x76, 0x69, 0x66], 64);
|
|
30
|
+
const WEBP2 = bytes([0x77, 0x70, 0x32, 0x20], 64);
|
|
31
|
+
const HEIF_MIF1 = bytes([0, 0, 0, 0x20, 0x66, 0x74, 0x79, 0x70, 0x6D, 0x69, 0x66, 0x31], 64);
|
|
32
|
+
const NOT_A_TRANSPORT_STREAM = bytes([0x47, 0x53, 0x54, 0x00], 512);
|
|
33
|
+
|
|
34
|
+
let failed = false;
|
|
35
|
+
|
|
36
|
+
// [label, bytes, expectedType, the Content-Type the server sent, the type that must reach the data URI]
|
|
37
|
+
const CASES = [
|
|
38
|
+
["a woff2 served as text/plain is embedded as font/woff2", WOFF2, "font", "text/plain;charset=UTF-8", "font/woff2"],
|
|
39
|
+
["a woff2 served as font/woff2 is unchanged", WOFF2, "font", "font/woff2", "font/woff2"],
|
|
40
|
+
["a woff2 served with no type at all is embedded as font/woff2", WOFF2, "font", undefined, "font/woff2"],
|
|
41
|
+
["a png served as text/plain is embedded as image/png", PNG, "image", "text/plain", "image/png"],
|
|
42
|
+
["an avif served as application/octet-stream is embedded as image/avif", AVIF, "image", "application/octet-stream", "image/avif"],
|
|
43
|
+
// "mif1" is the generic HEIF brand and an AVIF may carry it too, so it identifies nothing on
|
|
44
|
+
// its own: claiming HEIC here would relabel an AVIF as a format no browser decodes
|
|
45
|
+
["an ambiguous HEIF brand keeps the type the server sent", HEIF_MIF1, "image", "image/avif", "image/avif"],
|
|
46
|
+
// nothing matches these bytes, so the header is all there is and it has to survive
|
|
47
|
+
["a format the sniffer does not know keeps the type the server sent", WEBP2, "image", "image/webp2", "image/webp2"],
|
|
48
|
+
["a format the sniffer does not know and no type falls back to octet-stream", WEBP2, "image", undefined, "application/octet-stream"],
|
|
49
|
+
// the byte is 0x47, the packet stride is not, so this is not a transport stream
|
|
50
|
+
["a video starting with G is not relabelled as mp2t", NOT_A_TRANSPORT_STREAM, "video", "video/quicktime", "video/quicktime"]
|
|
51
|
+
];
|
|
52
|
+
|
|
53
|
+
for (const [label, data, expectedType, sentContentType, expectedContentType] of CASES) {
|
|
54
|
+
const { data: dataURI } = await fetchContent(data, expectedType, sentContentType);
|
|
55
|
+
check(label, readDataURIType(dataURI), expectedContentType);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
console.log(failed ? "\nsome checks FAILED" : "\nall checks passed");
|
|
59
|
+
Deno.exit(failed ? 1 : 0);
|
|
60
|
+
|
|
61
|
+
function fetchContent(data, expectedType, contentType) {
|
|
62
|
+
const util = getInstance({
|
|
63
|
+
fetch: async () => new Response(data, { headers: contentType ? { "content-type": contentType } : {} })
|
|
64
|
+
});
|
|
65
|
+
return util.getContent(FONT_URL, { asBinary: true, inline: true, expectedType });
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function readDataURIType(dataURI) {
|
|
69
|
+
const indexSeparator = dataURI.indexOf(";");
|
|
70
|
+
return dataURI.substring("data:".length, indexSeparator == -1 ? dataURI.indexOf(",") : indexSeparator);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function bytes(signature, length) {
|
|
74
|
+
const value = new Uint8Array(length);
|
|
75
|
+
value.set(signature);
|
|
76
|
+
return value;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function check(label, actual, expected) {
|
|
80
|
+
const ok = actual === expected;
|
|
81
|
+
console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
|
|
82
|
+
failed ||= !ok;
|
|
83
|
+
}
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
// Several @font-face rules that declare the same family with the same style descriptors are one
|
|
2
|
+
// composite face, not a stack where the last rule wins. CSS Fonts 4 §5.2: "When the matched face is
|
|
3
|
+
// a composite face, user agents must use the procedure above on each of the faces in the composite
|
|
4
|
+
// face in reverse order of @font-face rule definition", selecting a face only "if the effective
|
|
5
|
+
// character map supports the character in question". §4.5.1 says the same for the case where the
|
|
6
|
+
// ranges are identical rather than merely overlapping: "If the unicode ranges overlap for a set of
|
|
7
|
+
// @font-face rules with the same family and style descriptor values, the rules are ordered in the
|
|
8
|
+
// reverse order they were defined; the last rule defined is the first to be checked for a given
|
|
9
|
+
// character." So the later rule wins per character, and a character it has no glyph for falls back
|
|
10
|
+
// to the earlier rule of the same family.
|
|
11
|
+
//
|
|
12
|
+
// Dropping the earlier rule therefore loses glyphs. science.org declares icomoon twice, weight 400
|
|
13
|
+
// style normal in both, first a 103-codepoint icon set and then a 30-codepoint one holding only the
|
|
14
|
+
// slideshow arrows. The banner close button is content:"\e928", which lives in the first font only,
|
|
15
|
+
// so keeping the last rule alone rendered a tofu box reading "E9 28" where the X had been.
|
|
16
|
+
//
|
|
17
|
+
// Pooling the sources of every rule sharing a key breaks it the same way even when both rules are
|
|
18
|
+
// kept: one winning source gets written into all of them, so both rules end up naming the same font
|
|
19
|
+
// and the other one is gone just as surely. Each rule keeps its own sources.
|
|
20
|
+
//
|
|
21
|
+
// Rules that are duplicates outright, same key and same src, are still emitted once: they are the
|
|
22
|
+
// same member of the composite declared twice, so dropping one changes nothing.
|
|
23
|
+
import * as cssTree from "../../vendor/css-tree.js";
|
|
24
|
+
|
|
25
|
+
// helper.js reaches the frame hooks, which install themselves against window and document as they
|
|
26
|
+
// are evaluated: the stubs go in before the dynamic import
|
|
27
|
+
globalThis.window = globalThis;
|
|
28
|
+
globalThis.document = {};
|
|
29
|
+
globalThis.Document = class Document { };
|
|
30
|
+
globalThis.MutationObserver = class MutationObserver { observe() { } };
|
|
31
|
+
const { getProcessorHelperCommonClass } = await import("../../core/lib/processor-helper-common.js");
|
|
32
|
+
|
|
33
|
+
const ProcessorHelperCommon = getProcessorHelperCommonClass({}, cssTree);
|
|
34
|
+
|
|
35
|
+
// the real subclasses pick one source out of the list and rewrite the rule with it; the contract
|
|
36
|
+
// under test is which rules survive and which sources each one is handed, so this records that and
|
|
37
|
+
// keeps every rule
|
|
38
|
+
class TestProcessorHelper extends ProcessorHelperCommon {
|
|
39
|
+
constructor() {
|
|
40
|
+
super();
|
|
41
|
+
this.processedRules = [];
|
|
42
|
+
}
|
|
43
|
+
async processFontFaceRule(ruleData, fontInfo) {
|
|
44
|
+
this.processedRules.push({
|
|
45
|
+
family: this.getPropertyValue(ruleData, "font-family"),
|
|
46
|
+
sources: fontInfo.map(source => source.src)
|
|
47
|
+
});
|
|
48
|
+
return true;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
async function run(css) {
|
|
53
|
+
const helper = new TestProcessorHelper();
|
|
54
|
+
const stylesheets = new Map([[0, { stylesheet: cssTree.parse(css) }]]);
|
|
55
|
+
await helper.removeAlternativeFonts({}, stylesheets, new Map(), new Map());
|
|
56
|
+
const remaining = [];
|
|
57
|
+
stylesheets.get(0).stylesheet.children.forEach(ruleData => {
|
|
58
|
+
if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
|
|
59
|
+
remaining.push(helper.getPropertyValue(ruleData, "src"));
|
|
60
|
+
}
|
|
61
|
+
});
|
|
62
|
+
return { processed: helper.processedRules, remaining };
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
let failures = 0;
|
|
66
|
+
|
|
67
|
+
// a rule that was dropped when it should have been kept leaves a hole in the list, and reporting
|
|
68
|
+
// that as a failed comparison is more use than throwing on the way to the assertion
|
|
69
|
+
function sourcesOf(processedRules, index) {
|
|
70
|
+
const rule = processedRules[index];
|
|
71
|
+
return rule ? rule.sources : null;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function check(label, actual, expected) {
|
|
75
|
+
const pass = JSON.stringify(actual) == JSON.stringify(expected);
|
|
76
|
+
console.log((pass ? "PASS " : "FAIL ") + label + ": " + JSON.stringify(actual));
|
|
77
|
+
if (!pass) {
|
|
78
|
+
console.log(" expected: " + JSON.stringify(expected));
|
|
79
|
+
failures++;
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const TWO_FACES = `
|
|
84
|
+
@font-face{font-family:icomoon;src:url(big.ttf) format("truetype"),url(big.woff) format("woff");font-weight:400;font-style:normal}
|
|
85
|
+
@font-face{font-family:icomoon;src:url(small.ttf) format("truetype"),url(small.woff) format("woff");font-weight:400;font-style:normal}`;
|
|
86
|
+
|
|
87
|
+
const two = await run(TWO_FACES);
|
|
88
|
+
check("both members of the composite face are kept", two.processed.length, 2);
|
|
89
|
+
check("the earlier rule keeps its own sources", sourcesOf(two.processed, 0).sort(), ["url(big.ttf)format(\"truetype\")", "url(big.woff)format(\"woff\")"]);
|
|
90
|
+
check("the later rule keeps its own sources", (sourcesOf(two.processed, 1) || []).sort(), ["url(small.ttf)format(\"truetype\")", "url(small.woff)format(\"woff\")"]);
|
|
91
|
+
check("neither rule is removed from the stylesheet", two.remaining.length, 2);
|
|
92
|
+
|
|
93
|
+
const DUPLICATE = `
|
|
94
|
+
@font-face{font-family:icomoon;src:url(one.woff) format("woff");font-weight:400;font-style:normal}
|
|
95
|
+
@font-face{font-family:icomoon;src:url(one.woff) format("woff");font-weight:400;font-style:normal}`;
|
|
96
|
+
|
|
97
|
+
const duplicate = await run(DUPLICATE);
|
|
98
|
+
check("an outright duplicate rule is emitted once", duplicate.processed.length, 1);
|
|
99
|
+
check("the duplicate is removed from the stylesheet", duplicate.remaining.length, 1);
|
|
100
|
+
|
|
101
|
+
// unicode-range is part of the font key, so the subsetting idiom was never affected by the
|
|
102
|
+
// shadowing bug; it is pinned here because the fix moved what the key is used for
|
|
103
|
+
const RANGES = `
|
|
104
|
+
@font-face{font-family:sub;src:url(latin.woff2) format("woff2");unicode-range:U+0-7F}
|
|
105
|
+
@font-face{font-family:sub;src:url(greek.woff2) format("woff2");unicode-range:U+370-3FF}`;
|
|
106
|
+
|
|
107
|
+
const ranges = await run(RANGES);
|
|
108
|
+
check("faces split by unicode-range are all kept", ranges.processed.length, 2);
|
|
109
|
+
check("each range keeps its own source", ranges.processed.map(rule => rule.sources), [["url(latin.woff2)format(\"woff2\")"], ["url(greek.woff2)format(\"woff2\")"]]);
|
|
110
|
+
|
|
111
|
+
// a rule declaring the same source twice contributes it once, in the position its later declaration
|
|
112
|
+
// gives it. The sources are compared after the separating comma is stripped, so a repeat is
|
|
113
|
+
// recognised wherever it sits: the value is split by a regexp that keeps that comma, and comparing
|
|
114
|
+
// the raw pieces made the last source in a list unequal to the same source anywhere before it
|
|
115
|
+
const REPEATED = `
|
|
116
|
+
@font-face{font-family:repeat;src:url(a.woff) format("woff"),url(b.woff) format("woff"),url(a.woff) format("woff"),url(c.woff) format("woff");font-weight:400}`;
|
|
117
|
+
|
|
118
|
+
const repeated = await run(REPEATED);
|
|
119
|
+
check("a source repeated inside one rule is listed once", (sourcesOf(repeated.processed, 0) || []).length, 3);
|
|
120
|
+
|
|
121
|
+
const REPEATED_LAST = `
|
|
122
|
+
@font-face{font-family:repeat;src:url(a.woff) format("woff"),url(b.woff) format("woff"),url(a.woff) format("woff");font-weight:400}`;
|
|
123
|
+
|
|
124
|
+
const repeatedLast = await run(REPEATED_LAST);
|
|
125
|
+
check("a repeat in last position is recognised too", (sourcesOf(repeatedLast.processed, 0) || []).length, 2);
|
|
126
|
+
// the list is held in reverse of the order it is written back in, so the entry the rule declares
|
|
127
|
+
// last comes first here: a.woff keeps the position its second declaration gives it
|
|
128
|
+
check("the repeat keeps the position of its later declaration", sourcesOf(repeatedLast.processed, 0), ["url(a.woff)format(\"woff\")", "url(b.woff)format(\"woff\")"]);
|
|
129
|
+
|
|
130
|
+
if (failures) {
|
|
131
|
+
console.log("\n" + failures + " check(s) failed");
|
|
132
|
+
Deno.exit(1);
|
|
133
|
+
} else {
|
|
134
|
+
console.log("\nall checks passed");
|
|
135
|
+
}
|
|
@@ -432,9 +432,13 @@ const ALL_FACE_RUNGS = "<!--sfz-data<script<style<noframes<noembed<iframe<xmp<![
|
|
|
432
432
|
check("no page.pdf entry is left behind", entries.some(entry => entry.filename.endsWith("page.pdf")), false);
|
|
433
433
|
}
|
|
434
434
|
|
|
435
|
-
// page.pdf is the only record the writer builds by hand, so it is the only place the
|
|
436
|
-
//
|
|
437
|
-
// through
|
|
435
|
+
// page.pdf is the only record the writer builds by hand, so it is the only place the language
|
|
436
|
+
// encoding flag can disagree with the rest of the archive, and a reader would then decode that one
|
|
437
|
+
// name through a different path than every other name in the same file. Which way the flag goes is
|
|
438
|
+
// the ZIP writer's business, not this format's: it sets bit 11 only when a name or a comment holds
|
|
439
|
+
// a byte outside printable ASCII, so every name in this fixture is flagless. What is asserted here
|
|
440
|
+
// is the agreement, so this still fails if the hand-built record diverges and still passes if the
|
|
441
|
+
// writer changes its rule again
|
|
438
442
|
{
|
|
439
443
|
const options = makeOptions({ embeddedPdf: PDF });
|
|
440
444
|
const pageData = makePageData(23, 4 * 1024);
|
|
@@ -443,11 +447,14 @@ const ALL_FACE_RUNGS = "<!--sfz-data<script<style<noframes<noembed<iframe<xmp<![
|
|
|
443
447
|
const zipReader = new ZipReader(new BlobReader(new Blob([bytes])));
|
|
444
448
|
const entries = await zipReader.getEntries();
|
|
445
449
|
await zipReader.close();
|
|
446
|
-
|
|
447
|
-
check("
|
|
450
|
+
const pdfEntry = entries.find(entry => entry.filename == "page.pdf");
|
|
451
|
+
check("the pdf entry is listed", Boolean(pdfEntry), true);
|
|
452
|
+
check("the hand-built central record declares its name like the records the writer produces",
|
|
453
|
+
entries.every(entry => Boolean(entry.filenameUTF8) == Boolean(pdfEntry.filenameUTF8)), true);
|
|
448
454
|
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
449
|
-
const localFlags = entries.map(entry => view.getUint16(entry.offset + 6, true));
|
|
450
|
-
check("
|
|
455
|
+
const localFlags = new Map(entries.map(entry => [entry.filename, view.getUint16(entry.offset + 6, true) & 0x0800]));
|
|
456
|
+
check("the hand-built local header declares its name like the headers the writer produces",
|
|
457
|
+
[...localFlags.values()].every(flag => flag == localFlags.get("page.pdf")), true);
|
|
451
458
|
}
|
|
452
459
|
|
|
453
460
|
{
|
|
@@ -13,14 +13,15 @@
|
|
|
13
13
|
// - the titles written into the table of contents are CRAWLED, so they are attacker-controlled
|
|
14
14
|
// text going into an href attribute and into element content. Both escapers are checked here.
|
|
15
15
|
import "./dom-stub.js";
|
|
16
|
-
import { makePageData, makeOptions, runProcess } from "./common.js";
|
|
16
|
+
import { makePageData, makeOptions, runProcess, freezeDate } from "./common.js";
|
|
17
17
|
import { createPagesArchive } from "../../processors/compression/compression-packager.js";
|
|
18
|
-
import { ZipReader, BlobReader, TextWriter } from "../../vendor/zip/zip.js";
|
|
18
|
+
import { ZipReader, ZipWriter, BlobReader, TextReader, TextWriter, Uint8ArrayWriter } from "../../vendor/zip/zip.js";
|
|
19
19
|
|
|
20
20
|
// a title as it comes back from a crawl: the quote closes the href it is written into, the angle
|
|
21
21
|
// bracket opens an element, and the ampersand is what a naive escaper double-encodes
|
|
22
22
|
const HOSTILE_TITLE = "Intro & \"start\" <b>";
|
|
23
23
|
const SYMLINK_UNIX_MODE = 0o120777;
|
|
24
|
+
const SOURCE_DATE = new Date("2021-03-04T05:06:08Z");
|
|
24
25
|
|
|
25
26
|
let failed = false;
|
|
26
27
|
|
|
@@ -89,6 +90,42 @@ const pages = [
|
|
|
89
90
|
(await readEntry(entries, "pages/2/styles.css")).includes("font-family"), true);
|
|
90
91
|
}
|
|
91
92
|
|
|
93
|
+
// Every entry is copied with passThrough, i.e. its stored bytes are written back without being
|
|
94
|
+
// decompressed, so everything that DESCRIBES those bytes has to travel with them. Forwarding a
|
|
95
|
+
// subset does not make a partial copy, it makes a corrupt one: the writer cannot tell that a
|
|
96
|
+
// compression method it did not choose describes data it is about to store verbatim.
|
|
97
|
+
//
|
|
98
|
+
// The pages the rest of this file uses cannot show that, because they are written by the same
|
|
99
|
+
// writer with the same defaults as the archive they are copied into, so every value the copy
|
|
100
|
+
// drops is replaced by the one it had. This page carries values the packager's own defaults do
|
|
101
|
+
// not produce. It is the first page, so it is copied to the root under its own names.
|
|
102
|
+
{
|
|
103
|
+
const metadataPages = [await makeMetadataPage(), pages[1]];
|
|
104
|
+
const sourceEntries = await readArchive(await metadataPages[0].getData());
|
|
105
|
+
const entries = await readArchive(await createPagesArchive(metadataPages, packagerOptions()));
|
|
106
|
+
const copied = [...sourceEntries.keys()].filter(filename => entries.has(filename));
|
|
107
|
+
check("every entry of the first page is copied", copied.length, sourceEntries.size);
|
|
108
|
+
for (const property of ["comment", "compressionMethod", "uncompressedSize", "crc32", "filenameUTF8", "externalFileAttributes", "versionMadeBy", "internalFileAttributes", "uid", "gid", "directory"]) {
|
|
109
|
+
check("a copied entry keeps its " + property,
|
|
110
|
+
copied.every(filename => entries.get(filename)[property] === sourceEntries.get(filename)[property]), true);
|
|
111
|
+
}
|
|
112
|
+
for (const property of ["lastModDate", "creationDate", "lastAccessDate"]) {
|
|
113
|
+
check("a copied entry keeps its " + property,
|
|
114
|
+
copied.every(filename => dateOf(entries.get(filename)[property]) === dateOf(sourceEntries.get(filename)[property])), true);
|
|
115
|
+
}
|
|
116
|
+
// the level bits say how hard the deflater tried, and a copy that drops them reports the
|
|
117
|
+
// packager's default instead of the level the entry was actually written at
|
|
118
|
+
check("a copied entry keeps the deflate level it was written at",
|
|
119
|
+
copied.every(filename => entries.get(filename).bitFlag.level === sourceEntries.get(filename).bitFlag.level), true);
|
|
120
|
+
// zip.js rebuilds the fields it interprets itself, so what has to survive is the rest
|
|
121
|
+
check("a copied entry keeps an extra field zip.js does not interpret",
|
|
122
|
+
extraFieldOf(entries.get("styles.css")), extraFieldOf(sourceEntries.get("styles.css")));
|
|
123
|
+
const sameBytes = await Promise.all(copied.map(async filename => equalData(
|
|
124
|
+
await readRawData(entries.get(filename)),
|
|
125
|
+
await readRawData(sourceEntries.get(filename)))));
|
|
126
|
+
check("a copied entry holds the bytes it was read from", sameBytes.every(Boolean), true);
|
|
127
|
+
}
|
|
128
|
+
|
|
92
129
|
{
|
|
93
130
|
const entries = await readArchive(await createPagesArchive(pages, packagerOptions({ tocPage: true })));
|
|
94
131
|
const toc = await readEntry(entries, "sfz-toc.html");
|
|
@@ -138,6 +175,33 @@ const pages = [
|
|
|
138
175
|
new TextDecoder("windows-1252").decode(await createPagesArchive(pages, packagerOptions())).includes("<nav><ul>"), false);
|
|
139
176
|
}
|
|
140
177
|
|
|
178
|
+
// the options handed to the archive writer are DERIVED from PROCESS_OPTION_NAMES, not hand-listed.
|
|
179
|
+
// The hand copy carried eleven names and silently dropped maxAppendedDataLength, so
|
|
180
|
+
// --max-appended-data-length did nothing on any multi-page save and nothing failed for months.
|
|
181
|
+
// A one-byte budget has to reach the writer, where it is indistinguishable from refusing to append
|
|
182
|
+
{
|
|
183
|
+
const unfreeze = freezeDate();
|
|
184
|
+
try {
|
|
185
|
+
const budgeted = await createPagesArchive(pages, packagerOptions({ maxAppendedDataLength: 1 }));
|
|
186
|
+
const prevented = await createPagesArchive(pages, packagerOptions({ preventAppendedData: true }));
|
|
187
|
+
const unbudgeted = await createPagesArchive(pages, packagerOptions());
|
|
188
|
+
check("a one-byte appended-data budget reaches the archive writer", equalData(budgeted, prevented), true);
|
|
189
|
+
check("and appending is what the writer does without one", equalData(unbudgeted, prevented), false);
|
|
190
|
+
} finally {
|
|
191
|
+
unfreeze();
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// `password` is the one name the derivation must NOT forward. An encrypted multi-page archive
|
|
196
|
+
// cannot be written yet, and forwarding the password would half-ship it: the writer would start
|
|
197
|
+
// withholding the prologue's title as if the archive were encrypted, while the table of contents
|
|
198
|
+
// and every entry comment — each one a resource URL — kept riding in that same cleartext prologue
|
|
199
|
+
{
|
|
200
|
+
const prologue = new TextDecoder("windows-1252").decode(await createPagesArchive(pages, packagerOptions({ password: "secret" })));
|
|
201
|
+
check("a password is not forwarded to the archive writer",
|
|
202
|
+
prologue.includes("<title>Intro & "start" <b></title>"), true);
|
|
203
|
+
}
|
|
204
|
+
|
|
141
205
|
console.log(failed ? "\nsome checks FAILED" : "\nall checks passed");
|
|
142
206
|
Deno.exit(failed ? 1 : 0);
|
|
143
207
|
|
|
@@ -150,6 +214,39 @@ async function makePage(seed, { url, title, originalUrls }) {
|
|
|
150
214
|
return { url, title, originalUrls, getData: async () => bytes };
|
|
151
215
|
}
|
|
152
216
|
|
|
217
|
+
// a page archive holding, on purpose, nothing the packager's own writer would produce by default:
|
|
218
|
+
// a directory record, a name that needs the language encoding flag, a stored entry beside one
|
|
219
|
+
// deflated at the highest level, unix ownership, an extra field zip.js does not interpret, and
|
|
220
|
+
// dates outside the one the packager pins on its writer
|
|
221
|
+
async function makeMetadataPage() {
|
|
222
|
+
const zipWriter = new ZipWriter(new Uint8ArrayWriter(), { lastModDate: SOURCE_DATE });
|
|
223
|
+
await zipWriter.add("folder/", null, { directory: true, comment: "a folder" });
|
|
224
|
+
await zipWriter.add("styles.css", new TextReader("body{font-family:serif}"), {
|
|
225
|
+
level: 9,
|
|
226
|
+
comment: "https://example.com/café.css",
|
|
227
|
+
creationDate: SOURCE_DATE,
|
|
228
|
+
lastAccessDate: SOURCE_DATE,
|
|
229
|
+
internalFileAttributes: 1,
|
|
230
|
+
msDosCompatible: false,
|
|
231
|
+
unixMode: 0o100755,
|
|
232
|
+
uid: 501,
|
|
233
|
+
gid: 20,
|
|
234
|
+
extraField: new Map([[0x7777, new Uint8Array([1, 2, 3, 4])]])
|
|
235
|
+
});
|
|
236
|
+
await zipWriter.add("café.txt", new TextReader("un café"), { level: 0 });
|
|
237
|
+
const bytes = await zipWriter.close();
|
|
238
|
+
return { url: "https://example.com/metadata.html", title: "Metadata", getData: async () => bytes };
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
function dateOf(value) {
|
|
242
|
+
return value === undefined ? undefined : value.getTime();
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
function extraFieldOf(entry) {
|
|
246
|
+
const value = entry.extraField && entry.extraField.get(0x7777);
|
|
247
|
+
return value ? value.data.join(",") : undefined;
|
|
248
|
+
}
|
|
249
|
+
|
|
153
250
|
function packagerOptions(overrides = {}) {
|
|
154
251
|
return {
|
|
155
252
|
selfExtractingArchive: true,
|
|
@@ -170,6 +267,14 @@ function readEntry(entries, filename) {
|
|
|
170
267
|
return entries.get(filename).getData(new TextWriter());
|
|
171
268
|
}
|
|
172
269
|
|
|
270
|
+
function readRawData(entry) {
|
|
271
|
+
return entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
function equalData(dataLeft, dataRight) {
|
|
275
|
+
return dataLeft.length == dataRight.length && dataLeft.every((value, index) => value == dataRight[index]);
|
|
276
|
+
}
|
|
277
|
+
|
|
173
278
|
function check(label, actual, expected) {
|
|
174
279
|
const ok = actual === expected;
|
|
175
280
|
console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
// The router picks the page a multi-page archive opens on. Nothing drove it until this file: the
|
|
2
|
+
// function is inlined into the archive as source text, so it only ever ran inside a saved page,
|
|
3
|
+
// and the suites around it checked what the packager WROTE rather than what a reader would see.
|
|
4
|
+
// That is how the table of contents shipped unreachable. It was stored, the route existed, and
|
|
5
|
+
// no link and no landing rule pointed at it, so --crawl-save-archive-toc looked like it did
|
|
6
|
+
// nothing at all.
|
|
7
|
+
//
|
|
8
|
+
// The rule this file pins: an archive that stores a table of contents opens on it, and one that
|
|
9
|
+
// does not opens on the first page. Two cases guard the edges of that rule. A route in the hash
|
|
10
|
+
// names a page explicitly and has to win over the landing rule, or every deep link into an
|
|
11
|
+
// archive would land on the table of contents instead. A hash that is NOT a route is a plain
|
|
12
|
+
// fragment, and the only page it can mean is the first one, which is where the archive used to
|
|
13
|
+
// land before the fragment was ever read.
|
|
14
|
+
import "./dom-stub.js";
|
|
15
|
+
import { makePageData, makeOptions, runProcess } from "./common.js";
|
|
16
|
+
import { createPagesArchive } from "../../processors/compression/compression-packager.js";
|
|
17
|
+
import { router } from "../../processors/compression/compression-router.js";
|
|
18
|
+
import * as zip from "../../vendor/zip/zip.js";
|
|
19
|
+
|
|
20
|
+
const ARCHIVE_URL = "https://example.com/archive.html";
|
|
21
|
+
|
|
22
|
+
let failed = false;
|
|
23
|
+
|
|
24
|
+
const pages = [
|
|
25
|
+
await makePage(1, { url: "https://example.com/docs/intro.html", title: "Intro" }),
|
|
26
|
+
await makePage(2, { url: "https://example.com/docs/api/reference.html", title: "Reference" })
|
|
27
|
+
];
|
|
28
|
+
const withTOC = await createPagesArchive(pages, packagerOptions({ tocPage: true }));
|
|
29
|
+
const withoutTOC = await createPagesArchive(pages, packagerOptions());
|
|
30
|
+
|
|
31
|
+
{
|
|
32
|
+
const content = await open(withTOC);
|
|
33
|
+
check("an archive holding a table of contents opens on it", content.includes("<h1>Table of contents</h1>"), true);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
{
|
|
37
|
+
const content = await open(withoutTOC);
|
|
38
|
+
check("an archive holding no table of contents opens on the first page", content, "page at \"\"");
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
{
|
|
42
|
+
const content = await open(withTOC, "#sfz/pages/2/");
|
|
43
|
+
check("a route in the hash names the page to open", content, "page at \"pages/2/\"");
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// a bare fragment is what a hand-written link into the saved page looks like. The router scrolls
|
|
47
|
+
// to it after rendering, and the table of contents is not the document it belongs to
|
|
48
|
+
{
|
|
49
|
+
const content = await open(withTOC, "#introduction");
|
|
50
|
+
check("a hash that is not a route opens the first page", content, "page at \"\"");
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
console.log(failed ? "\nsome checks FAILED" : "\nall checks passed");
|
|
54
|
+
Deno.exit(failed ? 1 : 0);
|
|
55
|
+
|
|
56
|
+
// the router reads its world out of globalThis and renders through the two functions it is given,
|
|
57
|
+
// so a stub of each is enough to see the page it chose. Only what the first render touches is
|
|
58
|
+
// stubbed here; navigation, scroll restoration and link marking read more of the DOM than this
|
|
59
|
+
async function open(bytes, hash = "") {
|
|
60
|
+
let displayed;
|
|
61
|
+
installEnvironment(hash);
|
|
62
|
+
await router(new Blob([bytes]), {
|
|
63
|
+
extract: (content, { pagePath }) => ({ docContent: "page at " + JSON.stringify(pagePath) }),
|
|
64
|
+
display: (document, docContent) => displayed = docContent
|
|
65
|
+
});
|
|
66
|
+
return displayed;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function installEnvironment(hash) {
|
|
70
|
+
// the router asks for web workers, which the archive serves from its own extension URL. There
|
|
71
|
+
// is no such URL here, so the request is answered with the synchronous codec instead
|
|
72
|
+
globalThis.zip = { ...zip, configure: options => zip.configure({ ...options, useWebWorkers: false }) };
|
|
73
|
+
globalThis.document = {
|
|
74
|
+
head: { appendChild() { } },
|
|
75
|
+
styleSheets: [],
|
|
76
|
+
createElement: () => ({ setAttribute() { }, remove() { } }),
|
|
77
|
+
querySelectorAll: () => [],
|
|
78
|
+
querySelector: () => null,
|
|
79
|
+
getElementById: () => null
|
|
80
|
+
};
|
|
81
|
+
globalThis.history = {
|
|
82
|
+
state: null,
|
|
83
|
+
scrollRestoration: "auto",
|
|
84
|
+
replaceState(state) {
|
|
85
|
+
this.state = state;
|
|
86
|
+
}
|
|
87
|
+
};
|
|
88
|
+
// Deno defines location as a getter that throws without --location, so it is replaced rather
|
|
89
|
+
// than assigned
|
|
90
|
+
Object.defineProperty(globalThis, "location", {
|
|
91
|
+
value: { href: ARCHIVE_URL + hash, hash },
|
|
92
|
+
configurable: true,
|
|
93
|
+
writable: true
|
|
94
|
+
});
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
async function makePage(seed, { url, title }) {
|
|
98
|
+
const pageData = makePageData(seed, 2 * 1024);
|
|
99
|
+
pageData.title = title;
|
|
100
|
+
const { bytes } = await runProcess(pageData, makeOptions({ url }));
|
|
101
|
+
return { url, title, getData: async () => bytes };
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
function packagerOptions(overrides = {}) {
|
|
105
|
+
return {
|
|
106
|
+
selfExtractingArchive: true,
|
|
107
|
+
extractDataFromPage: true,
|
|
108
|
+
zipScript: "/* zip script stub */",
|
|
109
|
+
...overrides
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function check(label, actual, expected) {
|
|
114
|
+
const ok = actual === expected;
|
|
115
|
+
console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
|
|
116
|
+
failed ||= !ok;
|
|
117
|
+
}
|
|
@@ -33,7 +33,7 @@ function check(label, actual, expected) {
|
|
|
33
33
|
const options = makeOptions();
|
|
34
34
|
const pageData = makePageData(2, 64 * 1024);
|
|
35
35
|
pageData.resources.images.push(storedResource("photo.jpg",
|
|
36
|
-
["-->", "</
|
|
36
|
+
["-->", "</script>", "</style>", "</noframes>", "</noembed>", "</iframe>", "</xmp>", "]]>"]));
|
|
37
37
|
const result = await runProcess(pageData, options);
|
|
38
38
|
check("all closers exhaust to", result.fallbackTag, "<plaintext>");
|
|
39
39
|
check("all closers keep extraction", result.extractionDisabled, false);
|
|
@@ -47,4 +47,49 @@ function check(label, actual, expected) {
|
|
|
47
47
|
check("stored '</xmp>' alone stays on comment path", result.fallbackTag, null);
|
|
48
48
|
}
|
|
49
49
|
|
|
50
|
+
// the rung patterns are matched over the archive BYTES rather than a decoded string, so these
|
|
51
|
+
// four pin the parts of that match a byte scan is easy to get wrong: closers are matched
|
|
52
|
+
// case-insensitively, a closer only counts when a tag-name terminator follows it, and the
|
|
53
|
+
// comment closer is '-->' or '--!>' and nothing else in between
|
|
54
|
+
|
|
55
|
+
{
|
|
56
|
+
const options = makeOptions();
|
|
57
|
+
const pageData = makePageData(4, 64 * 1024);
|
|
58
|
+
pageData.resources.images.push(storedResource("photo.jpg", ["-->", "</SCRIPT>"]));
|
|
59
|
+
const result = await runProcess(pageData, options);
|
|
60
|
+
check("an upper-case closer counts, so '</SCRIPT>' escalates to", result.fallbackTag, "<style type=sfz-data>");
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
{
|
|
64
|
+
const options = makeOptions();
|
|
65
|
+
const pageData = makePageData(5, 64 * 1024);
|
|
66
|
+
pageData.resources.images.push(storedResource("photo.jpg", ["-->", "</script\tsrc"]));
|
|
67
|
+
const result = await runProcess(pageData, options);
|
|
68
|
+
check("a tab terminates a closer, so '</script\\t' escalates to", result.fallbackTag, "<style type=sfz-data>");
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
{
|
|
72
|
+
const options = makeOptions();
|
|
73
|
+
const pageData = makePageData(6, 64 * 1024);
|
|
74
|
+
pageData.resources.images.push(storedResource("photo.jpg", ["-->", "</scriptx"]));
|
|
75
|
+
const result = await runProcess(pageData, options);
|
|
76
|
+
check("'</scriptx' is not a closer, so it stops at", result.fallbackTag, "<script type=sfz-data>");
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
{
|
|
80
|
+
const options = makeOptions();
|
|
81
|
+
const pageData = makePageData(7, 64 * 1024);
|
|
82
|
+
pageData.resources.images.push(storedResource("photo.jpg", ["--!>"]));
|
|
83
|
+
const result = await runProcess(pageData, options);
|
|
84
|
+
check("'--!>' closes a comment, so it falls back to", result.fallbackTag, "<script type=sfz-data>");
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
{
|
|
88
|
+
const options = makeOptions();
|
|
89
|
+
const pageData = makePageData(8, 64 * 1024);
|
|
90
|
+
pageData.resources.images.push(storedResource("photo.jpg", ["--?>"]));
|
|
91
|
+
const result = await runProcess(pageData, options);
|
|
92
|
+
check("'--?>' does not close a comment", result.fallbackTag, null);
|
|
93
|
+
}
|
|
94
|
+
|
|
50
95
|
Deno.exit(failed ? 1 : 0);
|