single-file-core 1.5.119 → 1.5.121
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/core/index.js +4 -0
- package/core/lib/processor-helper-inline.js +3 -3
- package/core/util.js +3 -1
- package/doc/assets/singlefile-archive-byte-map.svg +195 -217
- package/doc/singlefile-archive.md +439 -180
- package/eslint.config.mjs +6 -0
- package/package.json +2 -2
- package/processors/compression/compression-display.js +0 -11
- package/processors/compression/compression-extract.js +0 -4
- package/processors/compression/compression-packager.js +0 -12
- package/processors/compression/compression-router.js +0 -35
- package/processors/compression/compression.js +105 -181
- package/processors/hooks/content/content-hooks-frames-web.js +2 -1
- package/processors/lazy/content/content-lazy-loader.js +21 -18
- package/test/sfz-harness/README.md +10 -1
- package/test/sfz-harness/byte-map.js +137 -0
- package/test/sfz-harness/format-rules.js +132 -2
- package/test/sfz-harness/option-wiring.js +0 -3
- package/test/sfz-harness/zip64.js +77 -0
|
@@ -322,7 +322,7 @@
|
|
|
322
322
|
const rootBounds = getBoundingClientRectDefined ? rootBoundingRect : docBoundingRect;
|
|
323
323
|
const time = 0;
|
|
324
324
|
return { target, intersectionRatio, boundingClientRect, intersectionRect: boundingClientRect, isIntersecting, rootBounds, time };
|
|
325
|
-
}).filter(params => params.boundingClientRect.width && params.boundingClientRect.height);
|
|
325
|
+
}).filter(params => params.boundingClientRect.width && params.boundingClientRect.height > 1);
|
|
326
326
|
if (params.length) {
|
|
327
327
|
observer.callback.call(intersectionObserver, params, intersectionObserver);
|
|
328
328
|
}
|
|
@@ -349,6 +349,7 @@
|
|
|
349
349
|
delete globalThis._singleFileImage;
|
|
350
350
|
}
|
|
351
351
|
if (!keepZoomLevel) {
|
|
352
|
+
resetScreenSize();
|
|
352
353
|
dispatchResizeEvent();
|
|
353
354
|
}
|
|
354
355
|
}
|
|
@@ -33,6 +33,8 @@ const helper = {
|
|
|
33
33
|
|
|
34
34
|
const MAX_IDLE_TIMEOUT_CALLS = 10;
|
|
35
35
|
const ATTRIBUTES_MUTATION_TYPE = "attributes";
|
|
36
|
+
const CHILD_LIST_MUTATION_TYPE = "childList";
|
|
37
|
+
const STYLESHEET_TAG_NAMES = ["STYLE", "LINK"];
|
|
36
38
|
|
|
37
39
|
const browser = globalThis.browser;
|
|
38
40
|
const document = globalThis.document;
|
|
@@ -88,24 +90,25 @@ function triggerLazyLoading(options) {
|
|
|
88
90
|
let loadingImages;
|
|
89
91
|
const pendingImages = new Set();
|
|
90
92
|
const observer = new MutationObserver(async mutations => {
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
93
|
+
const attributeMutations = mutations.filter(mutation => mutation.type == ATTRIBUTES_MUTATION_TYPE);
|
|
94
|
+
const insertedStylesheets = mutations.filter(mutation => mutation.type == CHILD_LIST_MUTATION_TYPE &&
|
|
95
|
+
Array.from(mutation.addedNodes).some(node => node.tagName && STYLESHEET_TAG_NAMES.includes(node.tagName.toUpperCase()) &&
|
|
96
|
+
(!node.classList || !node.classList.contains(helper.SINGLE_FILE_UI_ELEMENT_CLASS))));
|
|
97
|
+
const updated = attributeMutations.filter(mutation => {
|
|
98
|
+
if (mutation.attributeName == "src") {
|
|
99
|
+
mutation.target.setAttribute(helper.LAZY_SRC_ATTRIBUTE_NAME, mutation.target.src);
|
|
100
|
+
mutation.target.addEventListener("load", onResourceLoad);
|
|
101
|
+
}
|
|
102
|
+
if (mutation.attributeName == "src" || mutation.attributeName == "srcset" ||
|
|
103
|
+
(mutation.target.tagName && mutation.target.tagName.toUpperCase() == "SOURCE")) {
|
|
104
|
+
return !mutation.target.classList || !mutation.target.classList.contains(helper.SINGLE_FILE_UI_ELEMENT_CLASS);
|
|
105
|
+
}
|
|
106
|
+
});
|
|
107
|
+
if (updated.length || insertedStylesheets.length) {
|
|
108
|
+
loadingImages = true;
|
|
109
|
+
await deferForceLazyLoadEnd(observer, options, cleanupAndResolve);
|
|
110
|
+
if (!pendingImages.size) {
|
|
111
|
+
await deferLazyLoadEnd(observer, options, cleanupAndResolve);
|
|
109
112
|
}
|
|
110
113
|
}
|
|
111
114
|
});
|
|
@@ -1,9 +1,13 @@
|
|
|
1
1
|
# SFZ format harness
|
|
2
2
|
|
|
3
|
-
Tests for the SingleFile archive writer in `processors/compression
|
|
3
|
+
Tests for the SingleFile archive writer in `processors/compression/`, which drive
|
|
4
4
|
`process()` directly with synthetic page data, so they need no browser and no network.
|
|
5
5
|
The format itself is specified in [`doc/singlefile-archive.md`](../../doc/singlefile-archive.md).
|
|
6
6
|
|
|
7
|
+
The directory has since taken in suites for code the archive writer does not own — the
|
|
8
|
+
CSS processors, the download filename helpers — because they need the same Deno-with-a-
|
|
9
|
+
DOM-stub setup and there was nowhere else to put them. The table says which is which.
|
|
10
|
+
|
|
7
11
|
Run them with Deno, from the repository root:
|
|
8
12
|
|
|
9
13
|
```
|
|
@@ -31,6 +35,11 @@ any check failed.
|
|
|
31
35
|
| `adopted-stylesheets-hook.js` | That the page-world hook answers the adopted-stylesheets request for a CLOSED shadow root, which its host does not expose. |
|
|
32
36
|
| `inlined-functions.js` | That a function serialized into a self-extracting archive names nothing outside itself. An import survives bundling and still reads correctly, and the archive then throws a bare `ReferenceError` and renders nothing. |
|
|
33
37
|
| `pages-archive.js` | That `createPagesArchive` packs several pages into one archive correctly: the first page at the root and the others in folders, the manifest, the symlink a deduplicated entry leaves behind, and the escaping of crawled titles in both tables of contents. |
|
|
38
|
+
| `entry-compression.js` | That an entry is deflated or stored on the content type the server sent, not on the extension alone — a module served as `text/javascript` from a `.ts` URL used to go in uncompressed — and that an unrecognized `application/octet-stream` still stays stored. |
|
|
39
|
+
| `filename-max-length.js` | That `formatFilename` counts the ellipsis as well as the extension in the budget it truncates to, so a filename at `filenameMaxLength` stays at it, and that a limit shorter than the extension does not reach `Blob.slice` with a negative start. |
|
|
40
|
+
| `filename-characters.js` | That `getValidFilename` maps a full-width lookalike one character at a time — `C++` used to be saved as `C+` — while a run of characters with no lookalike still collapses to a single replacement. |
|
|
41
|
+
| `zip64.js` | That the `page.pdf` record injection accounts for the zip64 end of central directory record (§5.7): all four EOCD fields left at their sentinels, the entry counts and directory size carried in the zip64 record, the directory offset pointing at the injected record, and the archive still readable. The branch runs only past 4 GiB or 65535 entries, so nothing reached it before; the suite forces zip64 through `zipWriter.options` from inside the `writeEntries` callback, with no production lever. |
|
|
42
|
+
| `byte-map.js` | That the byte offsets §8.2 of the specification prints still describe what the writer emits: the prologue order, the doctype and root tag with nothing between them, the identifier's length ahead of the region, absolute EOCD offsets, and the entry order. The specimen §8.2 documents is saved from a live URL and has never been in this repository, so none of its numbers could be checked; three of them were wrong. This builds an equivalent with no network. |
|
|
34
43
|
| `css-fonts-minifier.js` | That `removeUnusedFonts` reads the font families it prunes on correctly: a `var()` family resolved from the values the document declares and not only from the ones the body inherits, every font kept when the value is genuinely undetermined, and a multi-word family name that does not also claim a font named after its own tail. |
|
|
35
44
|
|
|
36
45
|
## The tools
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
// §8.2 of doc/singlefile-archive.md is a table of byte offsets. Nothing could check it: the
|
|
2
|
+
// specimen it describes is saved from a live URL through the CLI, so it needs a network and a
|
|
3
|
+
// browser and has never existed in this repository. Three of its rows were wrong for an unknown
|
|
4
|
+
// length of time — the doctype is 15 bytes and nothing separates it from the root element start
|
|
5
|
+
// tag, so the first three offsets were each one too high — and §4.2's derived figure is stale by
|
|
6
|
+
// about 20 KB. This builds an equivalent specimen from the harness, deterministically and with no
|
|
7
|
+
// network, and asserts both the offsets the document prints and the structural relations they
|
|
8
|
+
// stand for. A writer change that moves the layout fails here instead of rotting in the prose.
|
|
9
|
+
import "./dom-stub.js";
|
|
10
|
+
import { makeOptions, runProcess, freezeDate } from "./common.js";
|
|
11
|
+
|
|
12
|
+
const zipScript = await Deno.readTextFile(new URL("../../vendor/zip/zip.min.js", import.meta.url));
|
|
13
|
+
|
|
14
|
+
let failed = false;
|
|
15
|
+
|
|
16
|
+
function check(label, actual, expected) {
|
|
17
|
+
const ok = actual === expected;
|
|
18
|
+
console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
|
|
19
|
+
failed ||= !ok;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
// a two-entry archive, matching the shape §8.2 documents: index.html and manifest.json only
|
|
23
|
+
function makeSpecimenPageData() {
|
|
24
|
+
return {
|
|
25
|
+
title: "Example Domain",
|
|
26
|
+
doctype: "<!DOCTYPE html>",
|
|
27
|
+
content: "<html><body><h1>Example Domain</h1><p>" + "specimen ".repeat(64) + "</p></body></html>",
|
|
28
|
+
comment: "\n url: https://example.com/ \n saved date: Wed Aug 13 2025 \n",
|
|
29
|
+
resources: { stylesheets: [], images: [] }
|
|
30
|
+
};
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
const restoreDate = freezeDate();
|
|
34
|
+
const { bytes } = await runProcess(makeSpecimenPageData(), makeOptions({
|
|
35
|
+
zipScript,
|
|
36
|
+
url: "https://example.com/",
|
|
37
|
+
insertCanonicalLink: false
|
|
38
|
+
}));
|
|
39
|
+
restoreDate();
|
|
40
|
+
|
|
41
|
+
const text = new TextDecoder("windows-1252").decode(bytes);
|
|
42
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
43
|
+
const at = needle => text.indexOf(needle);
|
|
44
|
+
|
|
45
|
+
const doctype = at("<!DOCTYPE html>");
|
|
46
|
+
const root = at("<html data-sfz>");
|
|
47
|
+
const charset = at("<meta charset=");
|
|
48
|
+
const charsetEnd = charset + text.substring(charset).indexOf(">") + 1;
|
|
49
|
+
const comment = at("<!--\n url:");
|
|
50
|
+
const title = at("<title>");
|
|
51
|
+
const style = at("<style>");
|
|
52
|
+
const body = at("<body hidden>");
|
|
53
|
+
const script = at("<script>");
|
|
54
|
+
const wrapperStart = at("<!--sfz-data");
|
|
55
|
+
const region = at("PK\x03\x04");
|
|
56
|
+
const central = text.indexOf("PK\x01\x02", region);
|
|
57
|
+
const eocd = text.indexOf("PK\x05\x06", central);
|
|
58
|
+
const payload = at("<sfz-extra-data>");
|
|
59
|
+
const endTags = at("</body></html>");
|
|
60
|
+
|
|
61
|
+
console.log(`specimen: ${bytes.length} bytes`);
|
|
62
|
+
console.log([
|
|
63
|
+
["html-prologue begins", doctype], ["root element start tag", root],
|
|
64
|
+
["charset declaration", charset], ["implementation comment", comment],
|
|
65
|
+
["title", title], ["stylesheet", style], ["body hidden", body],
|
|
66
|
+
["bootstrap script", script], ["wrapper start tag", wrapperStart],
|
|
67
|
+
["ZIP region begins", region], ["central directory", central],
|
|
68
|
+
["EOCD", eocd], ["recovery payload", payload], ["end tags", endTags]
|
|
69
|
+
].map(([label, offset]) => ` ${String(offset).padStart(7)} ${label}`).join("\n"));
|
|
70
|
+
|
|
71
|
+
// the prologue order, which §3.1 and §6.1 both got backwards until 2026-09-01: the charset
|
|
72
|
+
// declaration precedes the comment, because the comment carries an unbounded URL
|
|
73
|
+
check("doctype opens the file", doctype, 0);
|
|
74
|
+
check("root element start tag follows the doctype with nothing between", root, doctype + "<!DOCTYPE html>".length);
|
|
75
|
+
check("charset declaration follows the root element start tag", charset, root + "<html data-sfz>".length);
|
|
76
|
+
check("the comment begins where the charset declaration ends", comment, charsetEnd);
|
|
77
|
+
check("the charset declaration is inside the prescan window", charsetEnd <= 1024, true);
|
|
78
|
+
|
|
79
|
+
// the ordering the byte map asserts, region by region
|
|
80
|
+
check("prologue order holds end to end",
|
|
81
|
+
[doctype, root, charset, comment, title, style, body, script, wrapperStart, region, central, eocd, payload, endTags]
|
|
82
|
+
.every((offset, index, all) => offset >= 0 && (index === 0 || offset > all[index - 1])), true);
|
|
83
|
+
|
|
84
|
+
// the identifier sits inside the wrapper, so the region starts 12 bytes after the tag (§1.3)
|
|
85
|
+
check("the identifier precedes the region by the length of the start tag", region - wrapperStart, "<!--sfz-data".length);
|
|
86
|
+
|
|
87
|
+
// offsets are absolute file positions, not region-relative (§5.3)
|
|
88
|
+
check("the EOCD directory offset is an absolute file position", view.getUint32(eocd + 16, true), central);
|
|
89
|
+
check("the EOCD counts both entries", view.getUint16(eocd + 10, true), 2);
|
|
90
|
+
check("the EOCD declares no archive comment", view.getUint16(eocd + 20, true), 0);
|
|
91
|
+
|
|
92
|
+
// the region ends at the wrapper close tag, and the payload describes it minus the two
|
|
93
|
+
// comment-length bytes (§4.5)
|
|
94
|
+
const wrapperEnd = text.indexOf("-->", eocd);
|
|
95
|
+
check("the wrapper closes after the EOCD", wrapperEnd > eocd, true);
|
|
96
|
+
check("the appended run fits the budget", bytes.length - wrapperEnd <= 65535, true);
|
|
97
|
+
|
|
98
|
+
// the entries the document names, in the order it names them (§4.2)
|
|
99
|
+
let records = 0;
|
|
100
|
+
for (let index = text.indexOf("PK\x01\x02"); index != -1; index = text.indexOf("PK\x01\x02", index + 1)) {
|
|
101
|
+
records++;
|
|
102
|
+
}
|
|
103
|
+
check("the central directory holds two records", records, 2);
|
|
104
|
+
check("index.html is listed first", text.indexOf("index.html", central) < text.indexOf("manifest.json", central), true);
|
|
105
|
+
|
|
106
|
+
// §4.2: page.pdf's local header lies before the ZIP region, so the prepended-data compensation
|
|
107
|
+
// a reader of the recovered region applies takes its offset negative. The magnitude is the
|
|
108
|
+
// region's start and moves with the size of the inlined library, so only the relation is checked
|
|
109
|
+
{
|
|
110
|
+
const pdf = new TextEncoder().encode("%PDF-1.4\n1 0 obj\n<< /X (specimen) >>\nendobj\ntrailer\n<<>>\n%%EOF\n");
|
|
111
|
+
const restore = freezeDate();
|
|
112
|
+
const { bytes: pdfBytes } = await runProcess(makeSpecimenPageData(), makeOptions({
|
|
113
|
+
zipScript,
|
|
114
|
+
url: "https://example.com/",
|
|
115
|
+
insertCanonicalLink: false,
|
|
116
|
+
embeddedPdf: pdf
|
|
117
|
+
}));
|
|
118
|
+
restore();
|
|
119
|
+
const pdfText = new TextDecoder("windows-1252").decode(pdfBytes);
|
|
120
|
+
const pdfView = new DataView(pdfBytes.buffer, pdfBytes.byteOffset, pdfBytes.byteLength);
|
|
121
|
+
const pdfRegion = pdfText.indexOf("PK\x03\x04", pdfText.indexOf("<!--sfz-data"));
|
|
122
|
+
let record = pdfText.indexOf("PK\x01\x02", pdfRegion);
|
|
123
|
+
let stored;
|
|
124
|
+
while (record != -1 && stored === undefined) {
|
|
125
|
+
if (pdfText.substr(record + 46, pdfView.getUint16(record + 28, true)) == "page.pdf") {
|
|
126
|
+
stored = pdfView.getUint32(record + 42, true);
|
|
127
|
+
}
|
|
128
|
+
record = pdfText.indexOf("PK\x01\x02", record + 1);
|
|
129
|
+
}
|
|
130
|
+
check("page.pdf is listed in the central directory", stored !== undefined, true);
|
|
131
|
+
check("its local header sits where the central record says", pdfText.indexOf("PK\x03\x04"), stored);
|
|
132
|
+
check("its header is inside the PDF scan window", stored < 1024, true);
|
|
133
|
+
check("its header lies before the ZIP region", stored < pdfRegion, true);
|
|
134
|
+
check("so the compensated offset is negative", stored - pdfRegion < 0, true);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
Deno.exit(failed ? 1 : 0);
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import "./dom-stub.js";
|
|
2
2
|
import { makePageData, makeOptions, runProcess, mulberry32 } from "./common.js";
|
|
3
|
-
import { ZipReader, BlobReader } from "../../vendor/zip/zip.js";
|
|
3
|
+
import { ZipReader, ZipWriter, BlobReader } from "../../vendor/zip/zip.js";
|
|
4
4
|
|
|
5
5
|
// the quote is there because the escaper the title shares with the table of contents encodes
|
|
6
6
|
// it for an attribute value, where it matters, and a title has to round-trip through that too
|
|
@@ -121,7 +121,22 @@ function imageResource(url) {
|
|
|
121
121
|
const pageData = makePageData(6, 4 * 1024);
|
|
122
122
|
pageData.doctype = "<!DOCTYPE html PUBLIC \"-//W3C//DTD XHTML 1.1//EN\" \"" + "x".repeat(2000) + ".dtd\">";
|
|
123
123
|
const { bytes } = await runProcess(pageData, options);
|
|
124
|
-
|
|
124
|
+
const text = decodeText(bytes);
|
|
125
|
+
check("oversized doctype is replaced without the pdf face", text.startsWith("<!DOCTYPE html><html data-sfz>"), true);
|
|
126
|
+
check("charset declaration stays inside the scan window", charsetDeclarationEnd(text) <= 1024, true);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
{
|
|
130
|
+
const options = makeOptions();
|
|
131
|
+
const pageData = makePageData(6, 4 * 1024);
|
|
132
|
+
pageData.doctype = "<!DOCTYPE html PUBLIC \"-//W3C//DTD XHTML 1.0 Transitional//EN\" \"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd\">";
|
|
133
|
+
const { bytes } = await runProcess(pageData, options);
|
|
134
|
+
check("ordinary doctype is kept verbatim", decodeText(bytes).startsWith(pageData.doctype), true);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function charsetDeclarationEnd(text) {
|
|
138
|
+
const index = text.indexOf("<meta charset=");
|
|
139
|
+
return index == -1 ? -1 : index + text.substring(index).indexOf(">") + 1;
|
|
125
140
|
}
|
|
126
141
|
|
|
127
142
|
function triggerResource(literals) {
|
|
@@ -163,6 +178,67 @@ function countIdentifiers(text) {
|
|
|
163
178
|
check("relocated payload needs no separator node", /<\/sfz-extra-data> +<!--sfz-dataPK/.test(text), true);
|
|
164
179
|
}
|
|
165
180
|
|
|
181
|
+
// relocating the payload shifts the central directory offsets, which moves the payload length by
|
|
182
|
+
// a few base64 quanta; a reservation margin below that shift costs a third layout in about one
|
|
183
|
+
// build out of four, so a dozen relocations must all settle in two. The image bytes never contain
|
|
184
|
+
// a hyphen, so a wrapper collision cannot add a layout of its own
|
|
185
|
+
{
|
|
186
|
+
const { appendZip } = ZipWriter.prototype;
|
|
187
|
+
let layouts = 0;
|
|
188
|
+
ZipWriter.prototype.appendZip = function (reader) {
|
|
189
|
+
layouts++;
|
|
190
|
+
return appendZip.call(this, reader);
|
|
191
|
+
};
|
|
192
|
+
let retried = 0;
|
|
193
|
+
for (let seed = 30; seed < 42; seed++) {
|
|
194
|
+
const options = makeOptions({ preventAppendedData: true });
|
|
195
|
+
const pageData = makePageData(seed, 64 * 1024);
|
|
196
|
+
const rand = mulberry32(seed);
|
|
197
|
+
for (let index = 0; index < 40; index++) {
|
|
198
|
+
const content = new Uint8Array(8 * 1024).map(() => {
|
|
199
|
+
const byte = (rand() * 255) | 0;
|
|
200
|
+
return byte < 0x2D ? byte : byte + 1;
|
|
201
|
+
});
|
|
202
|
+
pageData.resources.images.push({ name: "images/" + index + ".jpg", extension: ".jpg", content, url: "https://example.com/" + index + ".jpg" });
|
|
203
|
+
}
|
|
204
|
+
layouts = 0;
|
|
205
|
+
await runProcess(pageData, options);
|
|
206
|
+
if (layouts > 2) {
|
|
207
|
+
retried++;
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
ZipWriter.prototype.appendZip = appendZip;
|
|
211
|
+
check("the reservation margin absorbs the offset shift", retried, 0);
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// a payload too large for the appended budget is relocated, and relocation is final: the writer
|
|
215
|
+
// then emits no appended run at all, so the file ends at the EOCD like a plain zip. A stored
|
|
216
|
+
// resource made of random newline bytes gives a payload past the budget from a few hundred KB
|
|
217
|
+
{
|
|
218
|
+
const { appendZip } = ZipWriter.prototype;
|
|
219
|
+
let layouts = 0;
|
|
220
|
+
ZipWriter.prototype.appendZip = function (reader) {
|
|
221
|
+
layouts++;
|
|
222
|
+
return appendZip.call(this, reader);
|
|
223
|
+
};
|
|
224
|
+
const options = makeOptions({ disableCompression: true });
|
|
225
|
+
const pageData = makePageData(42, 4 * 1024);
|
|
226
|
+
const rand = mulberry32(42);
|
|
227
|
+
const newlines = ["\n", "\r", "\r\n"];
|
|
228
|
+
let content = "";
|
|
229
|
+
while (content.length < 512 * 1024) {
|
|
230
|
+
content += newlines[(rand() * 3) | 0];
|
|
231
|
+
}
|
|
232
|
+
pageData.resources.stylesheets.push({ name: "newlines.txt", extension: ".txt", content, url: "https://example.com/newlines.txt" });
|
|
233
|
+
const { bytes } = await runProcess(pageData, options);
|
|
234
|
+
ZipWriter.prototype.appendZip = appendZip;
|
|
235
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
236
|
+
check("an oversized payload is relocated", options.extraDataSize > 0, true);
|
|
237
|
+
check("relocation needs a single extra layout", layouts, 2);
|
|
238
|
+
check("a relocated archive ends at the EOCD", view.getUint32(bytes.length - 22, true), 0x06054b50);
|
|
239
|
+
check("a relocated archive declares no comment", view.getUint16(bytes.length - 2, true), 0);
|
|
240
|
+
}
|
|
241
|
+
|
|
166
242
|
{
|
|
167
243
|
const options = makeOptions({ embeddedPdf: PDF });
|
|
168
244
|
const pageData = makePageData(15, 4 * 1024);
|
|
@@ -297,6 +373,24 @@ const ALL_FACE_RUNGS = "<!--sfz-data<script<style<noframes<noembed<iframe<xmp<![
|
|
|
297
373
|
check("no page.pdf entry is left behind", entries.some(entry => entry.filename.endsWith("page.pdf")), false);
|
|
298
374
|
}
|
|
299
375
|
|
|
376
|
+
// page.pdf is the only record the writer builds by hand, so it is the only place the
|
|
377
|
+
// language encoding flag can go missing: without it a reader decodes that one name
|
|
378
|
+
// through CP437 while reading every other name in the same archive as UTF-8
|
|
379
|
+
{
|
|
380
|
+
const options = makeOptions({ embeddedPdf: PDF });
|
|
381
|
+
const pageData = makePageData(23, 4 * 1024);
|
|
382
|
+
pageData.resources.images.push(imageResource("https://example.com/image.png"));
|
|
383
|
+
const { bytes } = await runProcess(pageData, options);
|
|
384
|
+
const zipReader = new ZipReader(new BlobReader(new Blob([bytes])));
|
|
385
|
+
const entries = await zipReader.getEntries();
|
|
386
|
+
await zipReader.close();
|
|
387
|
+
check("the pdf entry is listed", entries.some(entry => entry.filename == "page.pdf"), true);
|
|
388
|
+
check("every central record declares utf-8 names", entries.every(entry => entry.filenameUTF8), true);
|
|
389
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
390
|
+
const localFlags = entries.map(entry => view.getUint16(entry.offset + 6, true));
|
|
391
|
+
check("every local header declares utf-8 names", localFlags.every(flags => Boolean(flags & 0x0800)), true);
|
|
392
|
+
}
|
|
393
|
+
|
|
300
394
|
{
|
|
301
395
|
const embeddedImage = new Uint8Array(8 + 25 + 512 + 12);
|
|
302
396
|
embeddedImage.set([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]);
|
|
@@ -414,6 +508,42 @@ function readAppendedData(bytes) {
|
|
|
414
508
|
check("no trailing bytes when appended data is prevented", trailing, 0);
|
|
415
509
|
}
|
|
416
510
|
|
|
511
|
+
// a comment cannot escape its own delimiters, so text reaching it from the captured page —
|
|
512
|
+
// the infobar template resolves {page-title} and its kind against the document — would close
|
|
513
|
+
// it early and turn the rest into markup. The page the archive restores carries no scripts by
|
|
514
|
+
// default, and this was the one way to put one back
|
|
515
|
+
{
|
|
516
|
+
for (const [label, comment] of [
|
|
517
|
+
["-->", "\n info: --><script>INJECTED</script><!-- \n"],
|
|
518
|
+
["--!>", "\n info: --!><script>INJECTED</script><!-- \n"],
|
|
519
|
+
["--->", "\n info: ---><script>INJECTED</script><!-- \n"]
|
|
520
|
+
]) {
|
|
521
|
+
const pageData = makePageData(12, 1024);
|
|
522
|
+
pageData.comment = comment;
|
|
523
|
+
const { bytes } = await runProcess(pageData, makeOptions());
|
|
524
|
+
const text = decodeText(bytes);
|
|
525
|
+
const start = text.indexOf("<!--\n info:");
|
|
526
|
+
check(`the comment survives ${label} in its content`, start != -1, true);
|
|
527
|
+
// "--!>" closes a comment too, so the end is the first of either form, not the first "-->"
|
|
528
|
+
const end = start + text.substring(start).search(/--!?>/);
|
|
529
|
+
check(`${label} in the comment does not close it early`,
|
|
530
|
+
text.indexOf("INJECTED") < end, true);
|
|
531
|
+
}
|
|
532
|
+
const pageData = makePageData(12, 1024);
|
|
533
|
+
pageData.comment = "\n info: a-->b \n";
|
|
534
|
+
const { bytes } = await runProcess(pageData, makeOptions());
|
|
535
|
+
check("the characters either side of the run are kept",
|
|
536
|
+
decodeText(bytes).includes("info: a-- >b"), true);
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
// neither directive falls back to default-src, so each has to be named to have any effect
|
|
540
|
+
{
|
|
541
|
+
const { bytes } = await runProcess(makePageData(13, 1024), makeOptions({ insertMetaCSP: true }));
|
|
542
|
+
const policy = decodeText(bytes).match(/content-security-policy content="([^"]*)"/)[1];
|
|
543
|
+
check("the policy forbids form submission", policy.includes("form-action 'none'"), true);
|
|
544
|
+
check("the policy forbids a base element", policy.includes("base-uri 'none'"), true);
|
|
545
|
+
}
|
|
546
|
+
|
|
417
547
|
if (failed) {
|
|
418
548
|
console.log("FAILED");
|
|
419
549
|
Deno.exit(1);
|
|
@@ -11,13 +11,10 @@ import { PROCESS_OPTION_NAMES } from "../../processors/compression/compression.j
|
|
|
11
11
|
// one thing or the other before the suite goes green again
|
|
12
12
|
const INTERNAL_OPTION_NAMES = [
|
|
13
13
|
// state the module sets on itself between build passes
|
|
14
|
-
"extraData",
|
|
15
14
|
"extraDataSize",
|
|
16
|
-
"extraDataSizeDropped",
|
|
17
15
|
"extractDataFromPageTags",
|
|
18
16
|
"preventEmbeddedPdfEntry",
|
|
19
17
|
// supplied by compression-packager.js, which calls createArchive directly
|
|
20
|
-
"embeddedScreenshotImage",
|
|
21
18
|
"multiPageArchive",
|
|
22
19
|
// read for the root directory name, but no caller has ever passed it
|
|
23
20
|
"tabId",
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
// §5.7 states that the page.pdf central-record injection applies its accounting to the zip64
|
|
2
|
+
// end of central directory record when one is present, and claimed the combination "has been
|
|
3
|
+
// verified on a forced-zip64 build". No such build existed here: the writer has no zip64 lever
|
|
4
|
+
// and §8.3 described the specimen only in prose, so nothing checked the branch. It runs only past
|
|
5
|
+
// 4 GiB or 65535 entries, which a saved page never reaches, so a defect in it would be invisible.
|
|
6
|
+
//
|
|
7
|
+
// zipWriter.options is a live reference to the object the ZipWriter was constructed with, so the
|
|
8
|
+
// writeEntries callback can force zip64 on without any production change.
|
|
9
|
+
import "./dom-stub.js";
|
|
10
|
+
import { createArchive } from "../../processors/compression/compression.js";
|
|
11
|
+
import { TextReader, ZipReader, BlobReader } from "../../vendor/zip/zip.js";
|
|
12
|
+
import { FIXED_DATE, blobBytes } from "./common.js";
|
|
13
|
+
|
|
14
|
+
const ZIP64_EOCD_SIGNATURE = 0x06064b50;
|
|
15
|
+
const ZIP64_LOCATOR_SIGNATURE = 0x07064b50;
|
|
16
|
+
const EOCD_SIGNATURE = 0x06054b50;
|
|
17
|
+
|
|
18
|
+
let failed = false;
|
|
19
|
+
|
|
20
|
+
function check(label, actual, expected) {
|
|
21
|
+
const ok = actual === expected;
|
|
22
|
+
console.log(`${ok ? "PASS" : "FAIL"} ${label}: ${actual}${ok ? "" : " (expected " + expected + ")"}`);
|
|
23
|
+
failed ||= !ok;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
const PDF = new TextEncoder().encode("%PDF-1.4\n1 0 obj\n<< /X (zip64) >>\nendobj\ntrailer\n<<>>\n%%EOF\n");
|
|
27
|
+
const pageData = { doctype: "<!DOCTYPE html>", content: "<html><body>zip64</body></html>", title: "zip64" };
|
|
28
|
+
const archiveOptions = { url: "https://example.com/", selfExtractingArchive: true, embeddedPdf: PDF };
|
|
29
|
+
|
|
30
|
+
const blob = await createArchive(pageData, archiveOptions, "/* zip script stub */", async zipWriter => {
|
|
31
|
+
zipWriter.options.zip64 = true;
|
|
32
|
+
await zipWriter.add("index.html", new TextReader("<html><body>zip64</body></html>"));
|
|
33
|
+
await zipWriter.add("manifest.json", new TextReader("{}"));
|
|
34
|
+
}, FIXED_DATE);
|
|
35
|
+
const bytes = await blobBytes(blob);
|
|
36
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
37
|
+
|
|
38
|
+
// locate the three records from the end, the way a reader does
|
|
39
|
+
let eocd = -1;
|
|
40
|
+
for (let index = bytes.length - 22; index >= 0 && eocd == -1; index--) {
|
|
41
|
+
if (view.getUint32(index, true) == EOCD_SIGNATURE) {
|
|
42
|
+
eocd = index;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
check("the archive ends with an EOCD record", eocd != -1, true);
|
|
46
|
+
const locator = eocd - 20;
|
|
47
|
+
check("a zip64 locator precedes the EOCD", view.getUint32(locator, true), ZIP64_LOCATOR_SIGNATURE);
|
|
48
|
+
const zip64 = Number(view.getBigUint64(locator + 8, true));
|
|
49
|
+
check("the locator points at the zip64 record", view.getUint32(zip64, true), ZIP64_EOCD_SIGNATURE);
|
|
50
|
+
|
|
51
|
+
// §5.7: the EOCD's saturated fields stay at their sentinels, and the writer saturates all of
|
|
52
|
+
// them once zip64 is emitted rather than only the one that overflowed
|
|
53
|
+
check("the EOCD entry count is a sentinel", view.getUint16(eocd + 8, true), 0xFFFF);
|
|
54
|
+
check("the EOCD total count is a sentinel", view.getUint16(eocd + 10, true), 0xFFFF);
|
|
55
|
+
check("the EOCD directory size is a sentinel", view.getUint32(eocd + 12, true), 0xFFFFFFFF);
|
|
56
|
+
check("the EOCD directory offset is a sentinel", view.getUint32(eocd + 16, true), 0xFFFFFFFF);
|
|
57
|
+
|
|
58
|
+
// the injection's accounting lands in the zip64 record: three entries, not the writer's two
|
|
59
|
+
check("the zip64 record counts the injected record on this disk", Number(view.getBigUint64(zip64 + 24, true)), 3);
|
|
60
|
+
check("the zip64 record counts it in the total", Number(view.getBigUint64(zip64 + 32, true)), 3);
|
|
61
|
+
|
|
62
|
+
// the directory begins at the injected record, which sits ahead of the writer's own directory
|
|
63
|
+
const directoryOffset = Number(view.getBigUint64(zip64 + 48, true));
|
|
64
|
+
const directorySize = Number(view.getBigUint64(zip64 + 40, true));
|
|
65
|
+
check("the zip64 directory offset points at the injected record",
|
|
66
|
+
view.getUint32(directoryOffset, true), 0x02014b50);
|
|
67
|
+
check("the first record in the directory is page.pdf",
|
|
68
|
+
new TextDecoder().decode(bytes.subarray(directoryOffset + 46, directoryOffset + 54)), "page.pdf");
|
|
69
|
+
check("the directory size covers every record", directoryOffset + directorySize, zip64);
|
|
70
|
+
|
|
71
|
+
// and the result is still an archive every reader can read
|
|
72
|
+
const reader = new ZipReader(new BlobReader(new Blob([bytes])));
|
|
73
|
+
const entries = await reader.getEntries();
|
|
74
|
+
await reader.close();
|
|
75
|
+
check("a reader lists all three entries", entries.map(entry => entry.filename).join(","), "page.pdf,index.html,manifest.json");
|
|
76
|
+
|
|
77
|
+
Deno.exit(failed ? 1 : 0);
|