single-file-core 1.5.128 → 1.5.130

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,205 @@
1
+ import "./dom.js";
2
+
3
+ // core/util.js captures globalThis.DOMParser when it loads, and deno-dom throws on "text/xml", so
4
+ // the MAFF metadata could not be parsed here at all. This substitutes a stub for that one mime type,
5
+ // installed before common.js imports core, the same ordering constraint dom.js itself documents.
6
+ //
7
+ // What the stub stands for and what it does not. The two defects fixed alongside this suite are both
8
+ // about what core does with what the parser HANDS BACK — an attribute that came back null, and which
9
+ // of two almost-identical option fields gets written out — so a stub returning null or a string
10
+ // covers them faithfully. It does NOT cover the parse: that `RDF > Description > originalurl` matches
11
+ // `<RDF:RDF><RDF:Description><MAF:originalurl>` by local name, and that getAttributeNS resolves the
12
+ // RDF prefix, are properties of a real XML DOM that only a browser suite can confirm.
13
+ const XML_DOCUMENTS = new Map();
14
+ const NativeDOMParser = globalThis.DOMParser;
15
+
16
+ class StubXMLDocument {
17
+ constructor(values) {
18
+ this.values = values;
19
+ }
20
+ // undefined means the element is absent, null means it is present with no RDF:resource attribute
21
+ querySelector(selector) {
22
+ const localName = selector.split(">").pop().trim();
23
+ const value = this.values[localName];
24
+ return value === undefined ? null : { getAttributeNS: () => value };
25
+ }
26
+ }
27
+
28
+ globalThis.DOMParser = class {
29
+ parseFromString(content, mimeType) {
30
+ if (mimeType == "text/xml") {
31
+ return new StubXMLDocument(XML_DOCUMENTS.get(content) || {});
32
+ }
33
+ return new NativeDOMParser().parseFromString(content, mimeType);
34
+ }
35
+ };
36
+
37
+ const { capture, frameData, html, WIN_ID_ATTRIBUTE_NAME } = await import("./common.js");
38
+
39
+ const PAGE_URL = "https://example.com/page.html";
40
+ const RDF_URL = "https://example.com/index.rdf";
41
+ const FRAME_URL = "https://example.com/frame-dir/frame.html";
42
+ const FRAME_RDF_URL = "https://example.com/frame-dir/index.rdf";
43
+ const ORIGINAL_URL = "https://original.example/real.html";
44
+ const ARCHIVE_TIME = "Mon, 01 Jan 2024 10:20:30 GMT";
45
+ const ARCHIVE_TIME_MS = new Date(ARCHIVE_TIME).getTime();
46
+ const COMPLETE = { originalurl: ORIGINAL_URL, archivetime: ARCHIVE_TIME };
47
+ const PAGE = html("<h1>page</h1>");
48
+ const FRAME_PAGE = html("<h1>frame</h1>");
49
+ const HOST_PAGE = html("<h1>host</h1><iframe src=\"" + FRAME_URL + "\" " + WIN_ID_ATTRIBUTE_NAME + "=\"0.1\"></iframe>");
50
+
51
+ // The canonical link would be the obvious place to read the recovered url, but deno-dom reflects
52
+ // neither the href nor the type property, so core's `element.href = ...` leaves no attribute behind.
53
+ // Two observables survive that: the SingleFile comment, which core builds from options.saveUrl, and
54
+ // the embedded options block, which is where the second defect lives.
55
+ const OPTIONS_BLOCK = /<script data-single-file-options[^>]*>([^<]*)<\/script>/;
56
+
57
+ let fixtureIndex = 0;
58
+
59
+ function rdf(values) {
60
+ const content = "<?xml version=\"1.0\"?><!-- fixture " + (fixtureIndex++) + " -->";
61
+ XML_DOCUMENTS.set(content, values);
62
+ return content;
63
+ }
64
+
65
+ class CountingResources extends Map {
66
+ constructor(entries) {
67
+ super(entries);
68
+ this.counts = new Map();
69
+ }
70
+ get(key) {
71
+ this.counts.set(key, (this.counts.get(key) || 0) + 1);
72
+ return super.get(key);
73
+ }
74
+ countOf(key) {
75
+ return this.counts.get(key) || 0;
76
+ }
77
+ }
78
+
79
+ function resources(rdfContent) {
80
+ const entries = [
81
+ [PAGE_URL, { body: PAGE }],
82
+ [FRAME_URL, { body: FRAME_PAGE }]
83
+ ];
84
+ if (rdfContent !== undefined) {
85
+ entries.push([RDF_URL, { body: rdfContent, contentType: "text/xml" }]);
86
+ }
87
+ return new CountingResources(entries);
88
+ }
89
+
90
+ function commentURL(content) {
91
+ const match = content.match(/ url: ([^\n]*)/);
92
+ return match && match[1].trim();
93
+ }
94
+
95
+ function embeddedOptions(content) {
96
+ const match = content.match(OPTIONS_BLOCK);
97
+ return match && JSON.parse(match[1]);
98
+ }
99
+
100
+ let failed = false;
101
+
102
+ // readMaffMetadata is the new name of enableMaff, which was implemented in core and set by nothing:
103
+ // no CLI flag, no config key, no UI anywhere. That is what made renaming it free, and exposing it is
104
+ // what made the two defects below reachable by a user.
105
+ {
106
+ const map = resources(rdf(COMPLETE));
107
+ const content = await capture(map, { url: PAGE_URL, content: PAGE, insertSingleFileComment: true });
108
+ check("index.rdf is not requested when the option is off", map.countOf(RDF_URL), 0);
109
+ check("the page url is saved when the option is off", commentURL(content), PAGE_URL);
110
+ }
111
+
112
+ {
113
+ const map = resources(rdf(COMPLETE));
114
+ const content = await capture(map, { url: PAGE_URL, content: PAGE, readMaffMetadata: true, insertSingleFileComment: true });
115
+ check("index.rdf is requested when the option is on", map.countOf(RDF_URL), 1);
116
+ check("the original url is recovered", commentURL(content), ORIGINAL_URL);
117
+ }
118
+
119
+ // THE CRASH. An originalurl with no RDF:resource made getAttributeNS return null, saveUrl became
120
+ // null, and the canonical link then called .match() on it and failed the whole capture with a
121
+ // TypeError. The archivetime branch beside it had always guarded its own value, so the two halves of
122
+ // one method disagreed with each other.
123
+ {
124
+ let threw = false;
125
+ let content = "";
126
+ try {
127
+ content = await capture(resources(rdf({ originalurl: null, archivetime: ARCHIVE_TIME })), {
128
+ url: PAGE_URL,
129
+ content: PAGE,
130
+ readMaffMetadata: true,
131
+ insertSingleFileComment: true
132
+ });
133
+ } catch {
134
+ threw = true;
135
+ }
136
+ check("an originalurl with no resource attribute does not fail the capture", threw, false);
137
+ check("and the page url is kept", commentURL(content), PAGE_URL);
138
+ }
139
+
140
+ {
141
+ const content = await capture(resources(rdf({ originalurl: ORIGINAL_URL, archivetime: null })), {
142
+ url: PAGE_URL,
143
+ content: PAGE,
144
+ readMaffMetadata: true,
145
+ insertSingleFileComment: true
146
+ });
147
+ check("an archivetime with no resource attribute is tolerated", commentURL(content), ORIGINAL_URL);
148
+ }
149
+
150
+ {
151
+ const content = await capture(resources(), { url: PAGE_URL, content: PAGE, readMaffMetadata: true, insertSingleFileComment: true });
152
+ check("a missing index.rdf leaves the page url in place", commentURL(content), PAGE_URL);
153
+ }
154
+
155
+ // THE INCONSISTENCY. saveFilenameTemplateData wrote saveUrl from options.url, one line above writing
156
+ // saveDate from the value MAFF had just recovered, so the embedded block paired the archive's date
157
+ // with the extracted copy's path, and a recompute in the editor resolved {url-*} against the wrong
158
+ // one. options.saveUrl and options.url are identical on every other code path, which is what let it
159
+ // sit unnoticed.
160
+ {
161
+ const content = await capture(resources(rdf(COMPLETE)), {
162
+ url: PAGE_URL,
163
+ content: PAGE,
164
+ readMaffMetadata: true,
165
+ saveFilenameTemplateData: true
166
+ });
167
+ const embedded = embeddedOptions(content);
168
+ check("the embedded options block exists", Boolean(embedded), true);
169
+ if (embedded) {
170
+ check("the embedded saveUrl is the recovered one", embedded.saveUrl, ORIGINAL_URL);
171
+ check("the embedded saveDate is the recovered one", embedded.saveDate, ARCHIVE_TIME_MS);
172
+ }
173
+ }
174
+
175
+ {
176
+ const content = await capture(resources(rdf(COMPLETE)), { url: PAGE_URL, content: PAGE, saveFilenameTemplateData: true });
177
+ const embedded = embeddedOptions(content);
178
+ check("the embedded saveUrl is the page url when the option is off", embedded && embedded.saveUrl, PAGE_URL);
179
+ }
180
+
181
+ // Only the root document reaches Processor.initialize, because Runner.run guards that call with
182
+ // `if (this.root)`. That guard is the whole reason a page with twenty frames does not make twenty
183
+ // pointless index.rdf requests, and nothing else pins it. Note initializeProcessor resets a list of
184
+ // root-only options for frames and readMaffMetadata is deliberately NOT in it: next to this guard
185
+ // such a reset is dead code, which is exactly what adding one and watching nothing change proved.
186
+ {
187
+ const map = resources(rdf(COMPLETE));
188
+ const frames = [frameData("0.1", FRAME_URL, FRAME_PAGE)];
189
+ const content = await capture(map, { url: PAGE_URL, content: HOST_PAGE, frames, readMaffMetadata: true });
190
+ check("the root asks for its index.rdf", map.countOf(RDF_URL), 1);
191
+ check("a frame does not ask for one of its own", map.countOf(FRAME_RDF_URL), 0);
192
+ check("the frame is still captured", content.includes("frame"), true);
193
+ }
194
+
195
+ if (failed) {
196
+ console.log("FAILED");
197
+ Deno.exit(1);
198
+ }
199
+ console.log("OK");
200
+
201
+ function check(label, actual, expected) {
202
+ const ok = actual === expected;
203
+ console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
204
+ failed ||= !ok;
205
+ }
@@ -0,0 +1,79 @@
1
+ import { capture, frameData, html, WIN_ID_ATTRIBUTE_NAME } from "./common.js";
2
+
3
+ const PAGE_URL = "https://example.com/big.html";
4
+ const HOST_URL = "https://example.com/host.html";
5
+ const IMAGE_URL = "https://example.com/big.png";
6
+ const PAGE_MARKER = "BIG PAGE MARKER";
7
+ const HOST_MARKER = "HOST PAGE MARKER";
8
+
9
+ // One paragraph of 2.1 MB rather than many small ones, for the reason common.js gives.
10
+ const BIG_PAGE = html("<h1>" + PAGE_MARKER + "</h1><p>" + "filler ".repeat(300000) + "</p>");
11
+ const HOST_PAGE = html("<h1>" + HOST_MARKER + "</h1><iframe src=\"" + PAGE_URL + "\" " + WIN_ID_ATTRIBUTE_NAME + "=\"0.1\"></iframe>");
12
+ const IMAGE_PAGE = html("<h1>" + HOST_MARKER + "</h1><img src=\"" + IMAGE_URL + "\">");
13
+ const BIG_IMAGE = new Uint8Array(2 * 1024 * 1024).fill(0x21);
14
+
15
+ const resources = {
16
+ [PAGE_URL]: { body: BIG_PAGE },
17
+ [HOST_URL]: { body: HOST_PAGE },
18
+ [IMAGE_URL]: { body: BIG_IMAGE, contentType: "image/png" }
19
+ };
20
+
21
+ // One megabyte, so every fixture above is over it and the default of ten is not in the way.
22
+ const CAP = { maxResourceSizeEnabled: true, maxResourceSize: 1 };
23
+
24
+ let failed = false;
25
+
26
+ // The content a browser captured is handed to core as a string and never fetched, so the cap has no
27
+ // point at which it could fire. This is what every extension save and every non-raw CLI capture does.
28
+ {
29
+ const content = await capture(resources, { url: PAGE_URL, content: BIG_PAGE, ...CAP });
30
+ check("a page supplied as content is never capped", content.includes(PAGE_MARKER), true);
31
+ }
32
+
33
+ // The regression test. loadPage fetches the document itself in raw mode, and until rootDocument was
34
+ // excluded the cap emptied it: a 2.5 MB page came out as 525 bytes with no body at all, exit code 0
35
+ // and no warning. The cap is documented to apply to "images, fonts, stylesheets, scripts, frames,
36
+ // videos and audios", never to the page.
37
+ {
38
+ const content = await capture(resources, { url: PAGE_URL, saveRawPage: true, ...CAP });
39
+ check("a raw page over the cap keeps its content", content.includes(PAGE_MARKER), true);
40
+ }
41
+
42
+ // The control for the test above: the same cap, in the same capture, still has to drop a resource.
43
+ // A fix that exempted everything would pass the raw-page check and break the option.
44
+ {
45
+ const capped = await capture(resources, { url: HOST_URL, content: IMAGE_PAGE, ...CAP });
46
+ const uncapped = await capture(resources, { url: HOST_URL, content: IMAGE_PAGE });
47
+ check("an image over the cap is left out", capped.includes("data:image/png;base64"), false);
48
+ check("the page holding it is kept", capped.includes(HOST_MARKER), true);
49
+ check("the same image is embedded with the cap off", uncapped.includes("data:image/png;base64"), true);
50
+ }
51
+
52
+ // Frame content captured by the content script arrives as data, like the top document above, so the
53
+ // cap cannot reach it either.
54
+ {
55
+ const frames = [frameData("0.1", PAGE_URL, BIG_PAGE)];
56
+ const content = await capture(resources, { url: HOST_URL, content: HOST_PAGE, frames, ...CAP });
57
+ check("a frame supplied as data is never capped", content.includes(PAGE_MARKER), true);
58
+ check("its host is kept", content.includes(HOST_MARKER), true);
59
+ }
60
+
61
+ // In raw mode there is no frame data: resolveFrameURLs pushes a frame with no content and its runner
62
+ // fetches the frame document, which is the one caller the cap is meant for. Dropping it is correct.
63
+ {
64
+ const content = await capture(resources, { url: HOST_URL, saveRawPage: true, ...CAP });
65
+ check("a raw frame over the cap is dropped", content.includes(PAGE_MARKER), false);
66
+ check("its host is kept", content.includes(HOST_MARKER), true);
67
+ }
68
+
69
+ if (failed) {
70
+ console.log("FAILED");
71
+ Deno.exit(1);
72
+ }
73
+ console.log("OK");
74
+
75
+ function check(label, actual, expected) {
76
+ const ok = actual === expected;
77
+ console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
78
+ failed ||= !ok;
79
+ }
@@ -0,0 +1,82 @@
1
+ import { capture, html } from "./common.js";
2
+
3
+ const PAGE_URL = "https://example.com/page.html";
4
+
5
+ // blockScripts is what removeEmbedScripts is wired to, and it is the option every save turns on to
6
+ // promise that the saved page holds no script. These fixtures are the ways a javascript: URL used to
7
+ // survive that promise.
8
+ const BLOCK = { blockScripts: true };
9
+
10
+ // The sanitizer used to read the resolved IDL property, element.href and element.src, and rewrite the
11
+ // attribute only when the property was a string starting with "javascript:". That missed every SVG
12
+ // link, because SVGAElement.href is an SVGAnimatedString and the guard skipped it, and it never
13
+ // looked at form submission targets at all. Each of these executed on one click in a saved page.
14
+ const VECTORS = [
15
+ ["svg link", "<svg xmlns=\"http://www.w3.org/2000/svg\"><a id=\"target\" href=\"javascript:alert(1)\"><rect/></a></svg>"],
16
+ ["form action", "<form action=\"javascript:alert(1)\"><input type=\"submit\"></form>"],
17
+ ["button formaction", "<form><button formaction=\"javascript:alert(1)\">go</button></form>"],
18
+ ["image input formaction", "<form><input type=\"image\" formaction=\"javascript:alert(1)\"></form>"],
19
+ ["object data", "<object data=\"javascript:alert(1)\"></object>"],
20
+ ["anchor href", "<a href=\"javascript:alert(1)\">go</a>"],
21
+ ["iframe src", "<iframe src=\"javascript:alert(1)\"></iframe>"]
22
+ ];
23
+
24
+ let failed = false;
25
+
26
+ for (const [label, body] of VECTORS) {
27
+ const content = await capture({}, { url: PAGE_URL, content: html(body), ...BLOCK });
28
+ check(`${label} is neutralized`, content.includes("javascript:alert"), false);
29
+ }
30
+
31
+ // The URL parser decides what a javascript: URL is, so the obfuscations it folds away have to be
32
+ // folded away here too: leading whitespace is stripped, a tab inside the scheme is removed, and the
33
+ // scheme is compared case-insensitively.
34
+ const OBFUSCATIONS = [
35
+ ["leading whitespace and mixed case", " \tJaVaScRiPt:alert(1)"],
36
+ ["tab inside the scheme", "ja&#9;vascript:alert(1)"],
37
+ ["newline before the scheme", "&#10;javascript:alert(1)"],
38
+ ["uppercase scheme", "JAVASCRIPT:alert(1)"]
39
+ ];
40
+
41
+ for (const [label, value] of OBFUSCATIONS) {
42
+ const content = await capture({}, { url: PAGE_URL, content: html(`<a href="${value}">go</a>`), ...BLOCK });
43
+ check(`${label} is neutralized`, content.includes("alert(1)"), false);
44
+ }
45
+
46
+ // The control. A space inside the scheme is not a javascript: URL, and neither is a relative path, so
47
+ // a sanitizer that rewrote either would be matching on text rather than on what the URL parser says.
48
+ // Neither href survives as written, because resolveHrefs makes both absolute afterwards; what the
49
+ // control asserts is that the sanitizer did not claim them.
50
+ {
51
+ const content = await capture({}, { url: PAGE_URL, content: html("<a href=\"java script:alert(1)\">go</a><a href=\"page.html\">go</a>"), ...BLOCK });
52
+ check("a space inside the scheme is not treated as a script URI", content.includes("javascript:void(0)"), false);
53
+ check("a relative href is not treated as a script URI", content.includes("page.html"), true);
54
+ }
55
+
56
+ // The other half of removeEmbedScripts, kept here so that a rewrite of the attribute walk cannot drop
57
+ // it silently. Only the script elements can be checked from here: the event handler attribute names
58
+ // come from enumerating the on* IDL properties of an element, and deno-dom implements none of them,
59
+ // so the set is empty in this harness and a handler fixture would pass whatever the code did. The
60
+ // browser suite in single-file-cli covers the handlers.
61
+ {
62
+ const scripts = await capture({}, { url: PAGE_URL, content: html("<script>alert(1)</script><svg xmlns=\"http://www.w3.org/2000/svg\"><script>alert(2)</script></svg>"), ...BLOCK });
63
+ check("html and svg script elements are removed", scripts.includes("alert("), false);
64
+ }
65
+
66
+ // The option still has to be an option: with blockScripts off none of this is rewritten.
67
+ {
68
+ const content = await capture({}, { url: PAGE_URL, content: html("<a href=\"javascript:alert(1)\">go</a>") });
69
+ check("nothing is rewritten with blockScripts off", content.includes("javascript:alert(1)"), true);
70
+ }
71
+
72
+ if (failed) {
73
+ console.log("FAILED");
74
+ Deno.exit(1);
75
+ }
76
+ console.log("OK");
77
+
78
+ function check(label, actual, expected) {
79
+ const ok = actual === expected;
80
+ console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
81
+ failed ||= !ok;
82
+ }
package/test/run.js ADDED
@@ -0,0 +1,109 @@
1
+ // Runs every suite under test/, so that adding one means adding a file rather than editing a chain
2
+ // of shell commands. Two failure modes are worth naming, because this exists to remove the first
3
+ // without introducing the second: a suite nobody added to a hand-written list is never run and
4
+ // nobody notices, and a runner that takes every file it finds runs a scratch file that was never a
5
+ // test — single-file-tests did exactly that inside a release gate. So the files that are NOT suites
6
+ // are named below and anything else in these directories is run, which fails loudly rather than
7
+ // quietly. Keep this list in step when a helper or a tool is added.
8
+ //
9
+ // Unlike the chain it replaces, one red suite no longer hides the nineteen behind it: everything
10
+ // runs, and the summary says what failed.
11
+ //
12
+ // deno run --allow-read --allow-run test/run.js every suite
13
+ // deno run --allow-read --allow-run test/run.js cap font suites whose path matches an argument
14
+ // deno run --allow-read --allow-run test/run.js --verbose with the output of the suites that pass
15
+
16
+ const SUITE_DIRECTORIES = ["sfz-harness", "capture"];
17
+ const NOT_SUITES = [
18
+ "sfz-harness/common.js",
19
+ "sfz-harness/dom-stub.js",
20
+ "sfz-harness/gen-e2e-page.js",
21
+ "sfz-harness/search-triggers.js",
22
+ "sfz-harness/smoke.js",
23
+ "capture/common.js",
24
+ "capture/dom.js"
25
+ ];
26
+
27
+ const verbose = Deno.args.includes("--verbose");
28
+ const filters = Deno.args.filter(argument => !argument.startsWith("--"));
29
+ const suites = await findSuites();
30
+ const selected = filters.length ? suites.filter(suite => filters.some(filter => suite.includes(filter))) : suites;
31
+
32
+ if (!selected.length) {
33
+ console.log(filters.length ? `no suite matches ${filters.join(", ")}` : "no suite found");
34
+ Deno.exit(1);
35
+ }
36
+
37
+ let checksPassed = 0, checksFailed = 0;
38
+ const failures = [];
39
+ for (const suite of selected) {
40
+ const result = await runSuite(suite);
41
+ checksPassed += result.passed;
42
+ checksFailed += result.failed;
43
+ if (result.ok) {
44
+ console.log(`PASS ${suite}${result.passed ? ` (${result.passed} checks)` : ""}`);
45
+ if (verbose) {
46
+ console.log(indent(result.output));
47
+ }
48
+ } else {
49
+ failures.push(suite);
50
+ console.log(`FAIL ${suite}${result.failed ? ` (${result.failed} of ${result.passed + result.failed} checks)` : ` (exit ${result.code})`}`);
51
+ // a suite that fails a check has already said which one; a suite that crashed has not, and
52
+ // its output is the only thing that explains the exit code
53
+ console.log(indent(result.failed ? result.failedLines : result.output));
54
+ }
55
+ }
56
+
57
+ const skipped = NOT_SUITES.length;
58
+ console.log(`\n${selected.length} suites, ${checksPassed + checksFailed} checks, ${skipped} files skipped as tools or helpers`);
59
+ if (failures.length) {
60
+ console.log(`FAILED: ${failures.join(", ")}`);
61
+ Deno.exit(1);
62
+ }
63
+ console.log("all suites passed");
64
+
65
+ async function findSuites() {
66
+ const found = [];
67
+ for (const directory of SUITE_DIRECTORIES) {
68
+ const names = [];
69
+ for await (const entry of Deno.readDir(new URL(directory + "/", import.meta.url))) {
70
+ if (entry.isFile && entry.name.endsWith(".js")) {
71
+ names.push(entry.name);
72
+ }
73
+ }
74
+ names.sort();
75
+ for (const name of names) {
76
+ const path = directory + "/" + name;
77
+ if (!NOT_SUITES.includes(path)) {
78
+ found.push(path);
79
+ }
80
+ }
81
+ }
82
+ return found;
83
+ }
84
+
85
+ async function runSuite(suite) {
86
+ const command = new Deno.Command(Deno.execPath(), {
87
+ args: ["run", "--allow-read", new URL(suite, import.meta.url).pathname],
88
+ stdout: "piped",
89
+ stderr: "piped"
90
+ });
91
+ const { code, stdout, stderr } = await command.output();
92
+ const decoder = new TextDecoder();
93
+ const output = (decoder.decode(stdout) + decoder.decode(stderr)).trimEnd();
94
+ // the space matters: a suite ends on a bare "FAILED" line, which is a verdict and not a check
95
+ const lines = output.split("\n");
96
+ const failedLines = lines.filter(line => line.startsWith("FAIL ")).join("\n");
97
+ return {
98
+ code,
99
+ ok: code === 0,
100
+ output,
101
+ failedLines,
102
+ passed: lines.filter(line => line.startsWith("PASS ")).length,
103
+ failed: lines.filter(line => line.startsWith("FAIL ")).length
104
+ };
105
+ }
106
+
107
+ function indent(text) {
108
+ return text.split("\n").map(line => " " + line).join("\n");
109
+ }
@@ -14,7 +14,14 @@ Run them with Deno, from the repository root:
14
14
  npm test
15
15
  ```
16
16
 
17
- or one at a time:
17
+ which is [`test/run.js`](../run.js), the runner for every suite under `test/`. It takes
18
+ name filters, so this directory alone is:
19
+
20
+ ```
21
+ deno run --allow-read --allow-run test/run.js sfz-harness
22
+ ```
23
+
24
+ or one at a time, which needs no runner:
18
25
 
19
26
  ```
20
27
  deno run --allow-read test/sfz-harness/format-rules.js
@@ -42,12 +49,16 @@ any check failed.
42
49
  | `filename-characters.js` | That `getValidFilename` maps a full-width lookalike one character at a time — `C++` used to be saved as `C+` — while a run of characters with no lookalike still collapses to a single replacement. |
43
50
  | `zip64.js` | That the `page.pdf` record injection accounts for the zip64 end of central directory record (§5.7): all four EOCD fields left at their sentinels, the entry counts and directory size carried in the zip64 record, the directory offset pointing at the injected record, and the archive still readable. The branch runs only past 4 GiB or 65535 entries, so nothing reached it before; the suite forces zip64 through `zipWriter.options` from inside the `writeEntries` callback, with no production lever. |
44
51
  | `byte-map.js` | That the byte offsets §8.2 of the specification prints still describe what the writer emits: the prologue order, the doctype and root tag with nothing between them, the identifier's length ahead of the region, absolute EOCD offsets, and the entry order. The specimen §8.2 documents is saved from a live URL and has never been in this repository, so none of its numbers could be checked; three of them were wrong. This builds an equivalent with no network. |
52
+ | `relocation-cost.js` | That the figures §5.2 prints for relocating the extra-data element reconstruct. Two of the three came from live captures and did not: the paragraph subtracted the terminator and the end tags from the reservation without subtracting the element, which the appended placement carries too, so it over-counted by the whole element. This pins the corrected arithmetic — the cost is the reservation margin alone, it is positive on every rung whenever an element exists, and the only way relocation saves bytes is to have no element to relocate. |
45
53
  | `charset-round-trip.js` | That the encoding tables §8.4 prints still describe the WHATWG index: which 20 of the 38 encodings carry all 256 byte values through a decode injectively, the sizes of the reverse tables they need, and the five windows-1252 positions a platform codec of the same name leaves undefined. It also re-derives the reverse table the extractor ships as a literal, which no build step checks and which corrupts one byte per occurrence when wrong. |
46
54
  | `css-fonts-minifier.js` | That `removeUnusedFonts` reads the font families it prunes on correctly: a `var()` family resolved from the values the document declares and not only from the ones the body inherits, every font kept when the value is genuinely undetermined, and a multi-word family name that does not also claim a font named after its own tail. |
55
+ | `font-face-composite.js` | That several `@font-face` rules declaring the same family with the same style descriptors are one composite face and not a stack where the last rule wins, which is what CSS Fonts 4 §5.2 and §4.5.1 say: both members are kept with their own sources, faces split by `unicode-range` are all kept, an outright duplicate rule is emitted once, and a source repeated inside one rule is listed once, at the position of its later declaration. |
47
56
 
48
57
  ## The tools
49
58
 
50
- Not tests — they print, they do not assert, and CI does not run them.
59
+ Not tests — they print, they do not assert, and CI does not run them. The runner skips
60
+ them by name, in the `NOT_SUITES` list of [`test/run.js`](../run.js). Everything else in
61
+ this directory IS run, so a new file is either a suite or a line in that list.
51
62
 
52
63
  | Script | Use |
53
64
  |---|---|
@@ -4,9 +4,11 @@
4
4
  //
5
5
  // Three of its rules are worth stating, because they look arbitrary in the code:
6
6
  //
7
- // - the first page is stored at the ROOT and the others under pages/N/. The root page is what a
8
- // reader opens, so it cannot be moved into a folder without changing every relative URL the
9
- // capture already resolved.
7
+ // - the first page is stored at the ROOT and the others under pages/N/, unless
8
+ // createRootDirectory asks for a folder for the first page too. A page's resources travel with
9
+ // it, so either layout resolves; what the root buys is a reader who unzips the archive and
10
+ // opens index.html without being told where to look, and what it costs is that the first page
11
+ // shares the root with the archive's own files.
10
12
  // - a duplicate entry becomes a SYMLINK rather than being dropped. The router resolves it from
11
13
  // the alias map in the manifest and never reads it, but a plain unzip has to produce complete
12
14
  // page folders, and only a symlink gives both.
@@ -46,6 +48,37 @@ const pages = [
46
48
  (manifest.pages[0].originalUrls || []).join(" "), "https://example.com/docs/");
47
49
  }
48
50
 
51
+ // createRootDirectory gives the first page a folder of its own. Without it the first page is
52
+ // written at the root, mixed in with the archive's own files, which is the reason the router needs
53
+ // a special case at all: belongsToPage() has to read "everything not under pages/ and not named
54
+ // sfz-*" as the first page. With every page under pages/N/ that rule is a plain prefix match.
55
+ {
56
+ const entries = await readArchive(await createPagesArchive(pages, packagerOptions({ createRootDirectory: true, tocPage: true })));
57
+ const manifest = JSON.parse(await readEntry(entries, "sfz-pages.json"));
58
+ const toc = await readEntry(entries, "sfz-toc.html");
59
+ check("the first page is stored in a folder of its own when a root directory is asked for",
60
+ entries.has("pages/1/index.html"), true);
61
+ check("and the first page is no longer at the root", entries.has("index.html"), false);
62
+ check("the manifest names the folder of the first page too",
63
+ manifest.pages.map(page => page.path).join(" "), "pages/1/ pages/2/");
64
+ check("the table of contents links to the first page in its folder",
65
+ toc.includes("href=\"pages/1/index.html\""), true);
66
+ // the archive's own files stay at the root whatever the option says: the router finds them by
67
+ // exact name, and an archive whose sfz-pages.json moved stops being read as multi-page at all
68
+ check("the archive's own files are the only thing left at the root",
69
+ [...entries.keys()].filter(filename => !filename.includes("/")).sort().join(" "),
70
+ "sfz-pages.json sfz-toc.html");
71
+ }
72
+
73
+ // deduplication writes the link target relative to the folder the repeated entry sits in. With the
74
+ // first page at the root that walk never has a common prefix to drop; with both pages in folders it
75
+ // has to climb out of one and back into the other, which nothing exercised before
76
+ {
77
+ const entries = await readArchive(await createPagesArchive(pages, packagerOptions({ createRootDirectory: true, dedupPages: true })));
78
+ check("a repeated entry points across folders at the one that was kept",
79
+ await readEntry(entries, "pages/2/styles.css"), "../1/styles.css");
80
+ }
81
+
49
82
  // the router reads these two out of the manifest, and "auto" is the absence of a choice rather
50
83
  // than a value: writing it would pin the default of the day into every archive
51
84
  {
@@ -216,8 +249,8 @@ async function makePage(seed, { url, title, originalUrls }) {
216
249
 
217
250
  // a page archive holding, on purpose, nothing the packager's own writer would produce by default:
218
251
  // a directory record, a name that needs the language encoding flag, a stored entry beside one
219
- // deflated at the highest level, unix ownership, an extra field zip.js does not interpret, and
220
- // dates outside the one the packager pins on its writer
252
+ // deflated at the highest level, unix ownership, a non-default version-made-by spec byte, an
253
+ // extra field zip.js does not interpret, and dates outside the one the packager pins on its writer
221
254
  async function makeMetadataPage() {
222
255
  const zipWriter = new ZipWriter(new Uint8ArrayWriter(), { lastModDate: SOURCE_DATE });
223
256
  await zipWriter.add("folder/", null, { directory: true, comment: "a folder" });
@@ -228,6 +261,9 @@ async function makeMetadataPage() {
228
261
  lastAccessDate: SOURCE_DATE,
229
262
  internalFileAttributes: 1,
230
263
  msDosCompatible: false,
264
+ // only the high byte is rewritten to host Unix, so the 0x32 spec byte survives and the
265
+ // copied value differs from the 0x0300 both writers land on by default
266
+ versionMadeBy: 0x0332,
231
267
  unixMode: 0o100755,
232
268
  uid: 501,
233
269
  gid: 20,
@@ -20,6 +20,7 @@ import * as zip from "../../vendor/zip/zip.js";
20
20
  const ARCHIVE_URL = "https://example.com/archive.html";
21
21
 
22
22
  let failed = false;
23
+ let openedEntries;
23
24
 
24
25
  const pages = [
25
26
  await makePage(1, { url: "https://example.com/docs/intro.html", title: "Intro" }),
@@ -50,6 +51,28 @@ const withoutTOC = await createPagesArchive(pages, packagerOptions());
50
51
  check("a hash that is not a route opens the first page", content, "page at \"\"");
51
52
  }
52
53
 
54
+ // createRootDirectory moves the first page into pages/1/, so the landing rule has to come from the
55
+ // manifest rather than from the root. belongsToPage() also leaves its special case behind: while
56
+ // the first page is at the root it can only be described as "everything not under pages/ and not
57
+ // named sfz-*", and a page in a folder is selected by prefix like any other. The entries handed to
58
+ // extract are the assertion, because a landing path alone would still read right if that selection
59
+ // silently picked up the archive's own files
60
+ {
61
+ const rooted = await createPagesArchive(pages, packagerOptions({ createRootDirectory: true }));
62
+ const content = await open(rooted);
63
+ check("an archive with a root directory opens on the first page in its folder", content, "page at \"pages/1/\"");
64
+ check("and the router hands it only the entries of that folder",
65
+ openedEntries.join(" "), "pages/1/index.html pages/1/manifest.json pages/1/styles.css");
66
+ }
67
+
68
+ {
69
+ const rooted = await createPagesArchive(pages, packagerOptions({ createRootDirectory: true, tocPage: true }));
70
+ check("an archive with a root directory still opens on its table of contents",
71
+ (await open(rooted)).includes("<h1>Table of contents</h1>"), true);
72
+ check("and a route still names a page in it",
73
+ await open(rooted, "#sfz/pages/1/"), "page at \"pages/1/\"");
74
+ }
75
+
53
76
  console.log(failed ? "\nsome checks FAILED" : "\nall checks passed");
54
77
  Deno.exit(failed ? 1 : 0);
55
78
 
@@ -60,7 +83,10 @@ async function open(bytes, hash = "") {
60
83
  let displayed;
61
84
  installEnvironment(hash);
62
85
  await router(new Blob([bytes]), {
63
- extract: (content, { pagePath }) => ({ docContent: "page at " + JSON.stringify(pagePath) }),
86
+ extract: (content, { entries, pagePath }) => {
87
+ openedEntries = entries.map(entry => entry.filename).sort();
88
+ return { docContent: "page at " + JSON.stringify(pagePath) };
89
+ },
64
90
  display: (document, docContent) => displayed = docContent
65
91
  });
66
92
  return displayed;