single-file-core 1.5.119 → 1.5.121

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -27,6 +27,7 @@ import {
27
27
  BlobReader,
28
28
  TextReader,
29
29
  ZipWriter,
30
+ Uint8ArrayReader,
30
31
  Uint8ArrayWriter
31
32
  } from "./../../vendor/zip/zip.js";
32
33
  import {
@@ -41,29 +42,12 @@ import {
41
42
 
42
43
  const { Blob, fetch, TextEncoder, TextDecoder, DOMParser } = globalThis;
43
44
 
44
- // windows-1252 never decodes bytes >= 0x80 into the ASCII range, the scanned patterns are all ASCII
45
45
  const TEXT_DECODER = new TextDecoder("windows-1252");
46
46
 
47
- // the extension is only a guess when it comes from the URL, and a wrong one costs the whole
48
- // gain: a script served as text/javascript from a ".ts" URL was stored uncompressed at 2260
49
- // bytes where the same bytes named ".js" deflate to 70. A textual content type is authoritative
50
- // when the server sent one, the extension list decides everything else
51
47
  const COMPRESSIBLE_CONTENT_TYPES = ["application/javascript", "application/x-javascript", "application/ecmascript", "application/json", "application/ld+json", "application/manifest+json", "application/xml", "application/xhtml+xml", "application/rss+xml", "application/atom+xml", "image/svg+xml"];
52
48
  const TEXT_CONTENT_TYPE_PREFIX = "text/";
53
49
  const NO_COMPRESSION_EXTENSIONS = [".jpg", ".jpeg", ".png", ".apng", ".gif", ".webp", ".avif", ".heif", ".heic", ".jxl", ".pdf", ".woff", ".woff2", ".mp4", ".webm", ".avi", ".mpeg", ".mov", ".ts", ".ogv", ".mp3", ".ogg", ".oga", ".weba", ".m4a", ".aac", ".opus", ".flac"];
54
50
  const SCRIPT_PATH = "/lib/single-file-zip.min.js";
55
- // <noscript> is excluded: it is the only tag whose content is raw text when scripting is
56
- // enabled and markup when it is not, so the archive bytes would be parsed on a page opened
57
- // without scripting
58
- // the script and style rungs come first because text extractors drop their content the way
59
- // they drop a comment: macOS Spotlight indexes the content of every other rung, and textutil
60
- // also reads the last two. Neither is applied nor executed, the type is neither CSS nor
61
- // JavaScript. <plaintext> stays last, it is the only rung that cannot be closed
62
- // the CDATA section sits second to last, and it is the one rung a payload is unlikely to hold
63
- // the terminator of: every other rung ends on a sequence that real documents carry, which is
64
- // also why each level of self-nesting burns one. It is low in the ladder only because text
65
- // extractors read its content; nothing about the parse is weaker. A CDATA section is only a
66
- // CDATA section in foreign content, hence the <svg> element around it
67
51
  const EXTRA_DATA_TAGS = [
68
52
  ["<script type=sfz-data>", "</script>"],
69
53
  ["<style type=sfz-data>", "</style>"],
@@ -78,8 +62,6 @@ const EMBEDDED_DATA_TAGS = [
78
62
  ["<!--", "-->"],
79
63
  ...EXTRA_DATA_TAGS,
80
64
  ];
81
- // the identifier the extractor addresses the zip data with; the faces hidden by the same
82
- // wrapper ladder must never carry it, they are located by byte structure instead
83
65
  const DATA_IDENTIFIER = "sfz-data";
84
66
  const EXTRA_DATA_REGEXPS = [
85
67
  [/<script/i, /<\/script[\t\n\f\r />]/i],
@@ -88,17 +70,9 @@ const EXTRA_DATA_REGEXPS = [
88
70
  [/<noembed/i, /<\/noembed[\t\n\f\r />]/i],
89
71
  [/<iframe/i, /<\/iframe[\t\n\f\r />]/i],
90
72
  [/<xmp/i, /<\/xmp[\t\n\f\r />]/i],
91
- // a CDATA section ends on "]]>" and nothing else, so the terminator is the whole test; the
92
- // start pattern is the same conservatism the raw text rungs get, since a nested "<![CDATA["
93
- // is text like any other. Trailing brackets are safe: a payload ending "]]" against the
94
- // writer's "]]>" gives "]]]]>", and the tokenizer emits the payload's own two before closing
95
73
  [/<!\[CDATA\[/i, /\]\]>/],
96
74
  [/<plaintext/i, /<\/plaintext[\t\n\f\r />]/i]
97
75
  ];
98
- // a comment must also not end with "<!-", the last of the restrictions HTML puts on comment
99
- // text. The remaining one, that it must not start with ">" or "->", is not a matter of what
100
- // the payload contains: the zip data starts with the identifier and the PDF with a signature,
101
- // while the image data starts with a checksum, tested where that checksum is computed
102
76
  const EMBEDDED_DATA_REGEXPS = [
103
77
  [/<!--/i, /--!?>|<!-$/i],
104
78
  ...EXTRA_DATA_REGEXPS,
@@ -117,7 +91,11 @@ const PNG_IHDR_LENGTH = 25;
117
91
  const COMMENT_LENGTH_FIELD_LENGTH = 2;
118
92
  const MAX_APPENDED_DATA_LENGTH = 65535;
119
93
  const PDF_ENTRY_FILENAME = "page.pdf";
120
- const PDF_HEADER_MAX_OFFSET = 1024;
94
+ const PRESCAN_WINDOW_LENGTH = 1024;
95
+ const PNG_TEXT_CHUNK_HEADER_LENGTH = 12;
96
+ const PNG_ZIP_CHUNK_TYPE_KEYWORD = new Uint8Array([0x74, 0x45, 0x58, 0x74, 0x5a, 0x49, 0x50, 0]);
97
+ const MAX_HIDDEN_PNG_CHUNK_LENGTH = 0x2D000000;
98
+ const WRAPPER_PATTERN_WINDOW_LENGTH = 12;
121
99
  const MINIMAL_DOCTYPE = "<!DOCTYPE html>";
122
100
  const UNHIDDEN_FACE_WARNING_MESSAGE = "SingleFile: the page data contains every HTML tag that could hide an embedded file, the archive was written without its";
123
101
  const EMBEDDED_IMAGE_LABEL = "PNG image";
@@ -127,14 +105,10 @@ const CENTRAL_FILE_HEADER_SIGNATURE = 0x02014b50;
127
105
  const END_OF_CENTRAL_DIR_SIGNATURE = 0x06054b50;
128
106
  const ZIP64_END_OF_CENTRAL_DIR_SIGNATURE = 0x06064b50;
129
107
  const ZIP64_END_OF_CENTRAL_DIR_LOCATOR_SIGNATURE = 0x07064b50;
108
+ const LANGUAGE_ENCODING_FLAG = 0x0800;
130
109
 
131
110
  const browser = globalThis.browser;
132
111
 
133
- // the options process() accepts from its caller. single-file.js builds its argument from this
134
- // list instead of a literal, because an option added here and forgotten there is undefined at
135
- // every call site and the feature silently does nothing: that is how declareAppendedData and
136
- // includeBOM both shipped inert. Options the module sets on itself between passes, and options
137
- // the packager supplies, are deliberately absent
138
112
  const PROCESS_OPTION_NAMES = [
139
113
  "createRootDirectory",
140
114
  "declareAppendedData",
@@ -163,12 +137,6 @@ export {
163
137
 
164
138
  async function process(pageData, options, lastModDate = new Date()) {
165
139
  let script;
166
- // The worker is configured before anything else, and outside the extension it is turned off
167
- // rather than left alone. Given no address, zip.js resolves its default one against the page
168
- // being saved, so the browser asks the CAPTURED SITE for a file that site has never heard of:
169
- // three 404s in the user's own server logs for every archive, and then a fallback to the main
170
- // thread anyway, which is where the work was always going to happen. Choosing the fallback
171
- // costs nothing that was ever gained and asks the site for nothing.
172
140
  const extensionContext = Boolean(browser && browser.runtime && browser.runtime.getURL);
173
141
  if (extensionContext) {
174
142
  configure({ workerURI: "/lib/single-file-z-worker.js" });
@@ -188,15 +156,21 @@ async function process(pageData, options, lastModDate = new Date()) {
188
156
  }
189
157
 
190
158
  async function createArchive(pageData, options, script, writeEntries, lastModDate = new Date()) {
159
+ const zipWriterOptions = { bufferedWrite: true, keepOrder: true, lastModDate, useCompressionStream: true };
160
+ const entriesWriter = new ZipWriter(new Uint8ArrayWriter(), zipWriterOptions);
161
+ await writeEntries(entriesWriter);
162
+ const entriesData = await entriesWriter.close();
163
+ return buildArchive(pageData, options, script, entriesData, zipWriterOptions);
164
+ }
165
+
166
+ async function buildArchive(pageData, options, script, entriesData, zipWriterOptions) {
167
+ const { lastModDate } = zipWriterOptions;
191
168
  const zipDataWriter = new Uint8ArrayWriter();
192
169
  zipDataWriter.init();
193
170
  let extraDataOffset, extraData, embeddedImageDataOffset, endTag, pdfEntry;
194
171
  if (options.embeddedImage) {
195
172
  options.embeddedImage = new Uint8Array(options.embeddedImage);
196
173
  }
197
- // the whole chunk is built before the first byte of the image is written, because building it
198
- // is what settles the rung, and the search can end with no rung at all: the image is then left
199
- // out altogether rather than written unwrapped
200
174
  let imageChunk;
201
175
  if (options.embeddedImage && options.selfExtractingArchive) {
202
176
  imageChunk = getImageHTMLChunk(pageData, options, lastModDate);
@@ -211,8 +185,7 @@ async function createArchive(pageData, options, script, writeEntries, lastModDat
211
185
  endTag = imageChunk.endTag;
212
186
  if (imageChunk.startHTMLData.pdfEntry) {
213
187
  pdfEntry = imageChunk.startHTMLData.pdfEntry;
214
- // the htmlArray starts after the 4-byte length and the 8-byte type of the tEXt chunk
215
- pdfEntry.offset += zipDataWriter.offset + 12;
188
+ pdfEntry.offset += zipDataWriter.offset + PNG_TEXT_CHUNK_HEADER_LENGTH;
216
189
  }
217
190
  await writeData(zipDataWriter.writable, imageChunk.htmlData);
218
191
  await writeData(zipDataWriter.writable, imageChunk.htmlDataCRC);
@@ -224,7 +197,7 @@ async function createArchive(pageData, options, script, writeEntries, lastModDat
224
197
  await writeData(zipDataWriter.writable, embeddedImageData);
225
198
  await writeData(zipDataWriter.writable, new Uint8Array(4));
226
199
  embeddedImageDataOffset = zipDataWriter.offset;
227
- await writeData(zipDataWriter.writable, new Uint8Array([0x74, 0x45, 0x58, 0x74, 0x5a, 0x49, 0x50, 0]));
200
+ await writeData(zipDataWriter.writable, PNG_ZIP_CHUNK_TYPE_KEYWORD);
228
201
  if (options.selfExtractingArchive) {
229
202
  await writeData(zipDataWriter.writable, new TextEncoder().encode(endTag));
230
203
  }
@@ -236,53 +209,35 @@ async function createArchive(pageData, options, script, writeEntries, lastModDat
236
209
  } else if (!options.embeddedImage && options.embeddedPdf) {
237
210
  await writeData(zipDataWriter.writable, new Uint8Array(options.embeddedPdf));
238
211
  }
239
- // a WritableWriter object is passed instead of the writer so that the ZipWriter
240
- // never takes ownership of the stream: preventClose is only honored when the
241
- // caller owns the writable, and the HTML suffix still gets written after it;
242
- // its size property tells the ZipWriter the offset of the data written so far
243
212
  const startOffset = zipDataWriter.offset;
244
- const zipWriter = new ZipWriter({ writable: zipDataWriter.writable, size: startOffset }, { bufferedWrite: true, keepOrder: true, lastModDate, useCompressionStream: true });
245
- await writeEntries(zipWriter);
213
+ const zipWriter = new ZipWriter({ writable: zipDataWriter.writable, size: startOffset }, zipWriterOptions);
214
+ await zipWriter.appendZip(new Uint8ArrayReader(entriesData));
246
215
  if (pdfEntry) {
247
- // the record is written where the central directory will start so that the PDF is listed
248
- // first; the ZipWriter is unaware of these bytes, so the central directory offset it
249
- // stores in the end of central directory record points here
250
216
  new DataView(pdfEntry.centralRecord.buffer).setUint32(42, pdfEntry.offset, true);
251
217
  await writeData(zipDataWriter.writable, pdfEntry.centralRecord);
252
218
  }
253
219
  await zipWriter.close(undefined, { preventClose: true });
254
220
  if (pdfEntry && !patchEndOfCentralDirectory(zipDataWriter, pdfEntry.centralRecord.length)) {
255
- // the record cannot be declared in the end of central directory record: rebuild the
256
- // archive without it rather than leave a record the directory does not count
257
221
  options.preventEmbeddedPdfEntry = true;
258
- return createArchive(pageData, options, script, writeEntries, lastModDate);
222
+ return buildArchive(pageData, options, script, entriesData, zipWriterOptions);
259
223
  }
260
224
  const data = zipDataWriter.getData();
261
- // the last two bytes of the archive are the comment length field of the end of central
262
- // directory record: they are left out of the data the extraction payload describes, so that
263
- // declaring the appended data as the archive comment cannot invalidate a payload computed
264
- // before that length is known
265
225
  const zipDataEnd = data.length - COMMENT_LENGTH_FIELD_LENGTH;
266
226
  if (options.selfExtractingArchive) {
267
227
  const lfCodes = [];
268
228
  let crc32 = -1;
269
- // the wrapper must not be closed by the zip data itself, whether or not the page
270
- // carries the data: a premature closer parses the rest of the archive as markup
271
229
  if (!options.extractDataFromPageTags || options.extractDataFromPageTags[0] != "<plaintext>") {
272
230
  const textContent = TEXT_DECODER.decode(data.subarray(startOffset));
273
231
  if (options.extractDataFromPageTags) {
274
- // the rung is matched on its start tag, not on the identity of the array holding it:
275
- // the option is set from EXTRA_DATA_TAGS internally, but a caller passing an equal
276
- // pair of its own would otherwise index the regexps with -1
277
232
  const tagIndex = getExtraDataTagIndex(options.extractDataFromPageTags);
278
233
  const regExpsTag = EXTRA_DATA_REGEXPS[tagIndex];
279
234
  if (textContent.match(regExpsTag[0]) || textContent.match(regExpsTag[1])) {
280
- return findExtraDataTags(textContent, pageData, options, script, writeEntries, lastModDate, tagIndex + 1);
235
+ return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, tagIndex + 1);
281
236
  }
282
237
  } else {
283
238
  const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[0];
284
239
  if (textContent.match(startRegExp) || textContent.match(endRegExp)) {
285
- return findExtraDataTags(textContent, pageData, options, script, writeEntries, lastModDate);
240
+ return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions);
286
241
  }
287
242
  }
288
243
  }
@@ -314,35 +269,23 @@ async function createArchive(pageData, options, script, writeEntries, lastModDat
314
269
  }
315
270
  const endTags = options.preventAppendedData || options.embeddedImage ? "" : "</body></html>";
316
271
  if (options.extractDataFromPage) {
317
- // payload layout: [crc32, zip data length, LF codes count, 2-bit codes (0=LF, 1=CR, 2=CRLF) packed LSB-first]
318
- const payload = new Uint32Array(3 + Math.ceil(lfCodes.length / 16));
319
- payload[0] = crc32;
320
- payload[1] = zipDataEnd - startOffset;
321
- payload[2] = lfCodes.length;
322
- lfCodes.forEach((lfCode, indexLFCode) => payload[3 + (indexLFCode >> 4)] |= lfCode << ((indexLFCode & 15) * 2));
323
- extraData = "<sfz-extra-data>" + base64Encode(deflateRaw(new Uint8Array(payload.buffer))) + "</sfz-extra-data>";
324
- // the bytes appended after the EOCD record (wrapper end tag, extra data, end tags
325
- // and, with an embedded image, the tEXt CRC and IEND chunk) must fit the 65535-byte
326
- // window readers scan backward to find the EOCD record
272
+ const words = new Uint32Array(3 + Math.ceil(lfCodes.length / 16));
273
+ words[0] = crc32;
274
+ words[1] = zipDataEnd - startOffset;
275
+ words[2] = lfCodes.length;
276
+ lfCodes.forEach((lfCode, indexLFCode) => words[3 + (indexLFCode >> 4)] |= lfCode << ((indexLFCode & 15) * 2));
277
+ const payload = new Uint8Array(words.length * 4);
278
+ const payloadView = new DataView(payload.buffer);
279
+ words.forEach((word, indexWord) => payloadView.setUint32(indexWord * 4, word, true));
280
+ extraData = "<sfz-extra-data>" + base64Encode(deflateRaw(payload)) + "</sfz-extra-data>";
327
281
  if (options.preventAppendedData || extraData.length > MAX_APPENDED_DATA_LENGTH - pageContent.length - endTags.length - (options.embeddedImage ? PNG_IEND_LENGTH + PNG_CHUNK_CRC_LENGTH : 0)) {
328
282
  if (!options.extraDataSize) {
283
+ options.preventAppendedData = true;
329
284
  options.extraDataSize = getReservationSize(extraData.length);
330
- return createArchive(pageData, options, script, writeEntries, lastModDate);
285
+ return buildArchive(pageData, options, script, entriesData, zipWriterOptions);
331
286
  }
332
287
  } else {
333
- if (options.extraDataSize) {
334
- // dropping the reservation moves the archive back, which changes the payload
335
- // that made it necessary: a payload sitting on the boundary would be too large
336
- // appended and small enough relocated, forever. The reservation is dropped at
337
- // most once, so the build cannot oscillate between the two placements
338
- if (!options.extraDataSizeDropped) {
339
- options.extraDataSizeDropped = true;
340
- options.extraDataSize = undefined;
341
- return createArchive(pageData, options, script, writeEntries, lastModDate);
342
- }
343
- } else {
344
- pageContent += extraData;
345
- }
288
+ pageContent += extraData;
346
289
  }
347
290
  }
348
291
  pageContent += endTags;
@@ -354,22 +297,23 @@ async function createArchive(pageData, options, script, writeEntries, lastModDat
354
297
  if (options.extraDataSize >= extraData.length) {
355
298
  pageContent.set(new TextEncoder().encode(extraData), startOffset - extraDataOffset);
356
299
  } else {
357
- options.extraData = extraData;
358
300
  options.extraDataSize = getReservationSize(extraData.length);
359
- return createArchive(pageData, options, script, writeEntries, lastModDate);
301
+ return buildArchive(pageData, options, script, entriesData, zipWriterOptions);
360
302
  }
361
303
  }
362
304
  if (options.declareAppendedData) {
363
- // readers that reject undeclared bytes after the end of central directory record, notably
364
- // java.util.zip, accept the file when the same bytes are declared as the archive comment
365
305
  const appendedDataLength = pageContent.length - data.length +
366
306
  (options.embeddedImage ? PNG_CHUNK_CRC_LENGTH + PNG_IEND_LENGTH : 0);
367
- if (appendedDataLength && appendedDataLength <= MAX_APPENDED_DATA_LENGTH) {
307
+ if (appendedDataLength && appendedDataLength <= MAX_APPENDED_DATA_LENGTH && isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options)) {
368
308
  new DataView(pageContent.buffer, pageContent.byteOffset).setUint16(zipDataEnd, appendedDataLength, true);
369
309
  }
370
310
  }
371
311
  if (options.embeddedImage) {
372
- pageContent.set(getLength(zipDataWriter.offset - embeddedImageDataOffset - 4), embeddedImageDataOffset - 4);
312
+ const chunkLength = zipDataWriter.offset - embeddedImageDataOffset - 4;
313
+ if (options.selfExtractingArchive && chunkLength >= MAX_HIDDEN_PNG_CHUNK_LENGTH) {
314
+ throw new Error("SingleFile: the embedded PNG chunk is too large to be hidden from the HTML parser");
315
+ }
316
+ pageContent.set(getLength(chunkLength), embeddedImageDataOffset - 4);
373
317
  return new Blob([
374
318
  pageContent,
375
319
  getCRC32(pageContent, embeddedImageDataOffset),
@@ -380,6 +324,18 @@ async function createArchive(pageData, options, script, writeEntries, lastModDat
380
324
  }
381
325
  }
382
326
 
327
+ function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options) {
328
+ if (options.extractDataFromPageTags && options.extractDataFromPageTags[0] == "<plaintext>") {
329
+ return true;
330
+ }
331
+ const tail = pageContent.slice(zipDataEnd - WRAPPER_PATTERN_WINDOW_LENGTH, zipDataEnd + COMMENT_LENGTH_FIELD_LENGTH);
332
+ new DataView(tail.buffer).setUint16(WRAPPER_PATTERN_WINDOW_LENGTH, appendedDataLength, true);
333
+ const tagIndex = options.extractDataFromPageTags ? getExtraDataTagIndex(options.extractDataFromPageTags) + 1 : 0;
334
+ const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[tagIndex];
335
+ const tailText = TEXT_DECODER.decode(tail);
336
+ return !tailText.match(startRegExp) && !tailText.match(endRegExp);
337
+ }
338
+
383
339
  function getCRC32(data, indexData = 0) {
384
340
  const crcArray = new Uint8Array(4);
385
341
  setUint32(crcArray, getCRC32Value(data, indexData));
@@ -403,6 +359,7 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
403
359
  const localHeaderView = new DataView(localHeader.buffer);
404
360
  localHeaderView.setUint32(0, LOCAL_FILE_HEADER_SIGNATURE, true);
405
361
  localHeaderView.setUint16(4, 20, true);
362
+ localHeaderView.setUint16(6, LANGUAGE_ENCODING_FLAG, true);
406
363
  localHeaderView.setUint16(10, dosTime, true);
407
364
  localHeaderView.setUint16(12, dosDate, true);
408
365
  localHeaderView.setUint32(14, crc32, true);
@@ -415,6 +372,7 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
415
372
  centralRecordView.setUint32(0, CENTRAL_FILE_HEADER_SIGNATURE, true);
416
373
  centralRecordView.setUint16(4, 0x0300, true);
417
374
  centralRecordView.setUint16(6, 20, true);
375
+ centralRecordView.setUint16(8, LANGUAGE_ENCODING_FLAG, true);
418
376
  centralRecordView.setUint16(12, dosTime, true);
419
377
  centralRecordView.setUint16(14, dosDate, true);
420
378
  centralRecordView.setUint32(16, crc32, true);
@@ -438,7 +396,6 @@ function patchEndOfCentralDirectory(zipDataWriter, centralRecordLength) {
438
396
  const offsetLocator = offsetEOCD - 20;
439
397
  let offsetZip64EOCD;
440
398
  if (offsetLocator >= 0 && view.getUint32(offsetLocator, true) == ZIP64_END_OF_CENTRAL_DIR_LOCATOR_SIGNATURE) {
441
- // the offset stored in the locator does not account for the PDF record either
442
399
  offsetZip64EOCD = Number(view.getBigUint64(offsetLocator + 8, true)) + centralRecordLength;
443
400
  if (view.getUint32(offsetZip64EOCD, true) != ZIP64_END_OF_CENTRAL_DIR_SIGNATURE) {
444
401
  return false;
@@ -464,16 +421,12 @@ function patchEndOfCentralDirectory(zipDataWriter, centralRecordLength) {
464
421
  return true;
465
422
  }
466
423
 
467
- // the functions inlined in the archive lose their newlines, so a line comment would swallow
468
- // the rest of the script: whole-line comments are removed before the newlines are
469
424
  function inlineFunction(bootstrapFunction) {
470
425
  return bootstrapFunction.toString().replace(/^[ \t]*\/\/.*$/gm, "").replace(/\n|\t/g, "");
471
426
  }
472
427
 
473
- // the reservation must be strictly larger than the payload it was computed from, otherwise
474
- // the retry loop can be handed the same size again and oscillate instead of converging
475
428
  function getReservationSize(length) {
476
- return Math.max(length + 1, Math.floor(length * 1.001));
429
+ return Math.ceil(length * 1.01) + 32;
477
430
  }
478
431
 
479
432
  function getLength(length) {
@@ -513,14 +466,11 @@ async function prependHTMLData(pageData, zipDataWriter, script, options, lastMod
513
466
  if (pageData.tocContent) {
514
467
  pageContent += pageData.tocContent;
515
468
  }
516
- // the text body repeats the page content outside the archive, where no password reaches it
517
469
  if (options.insertTextBody && !options.password) {
518
470
  const doc = (new DOMParser()).parseFromString(pageData.content, "text/html");
519
471
  doc.body.querySelectorAll("style, script, noscript").forEach(element => element.remove());
520
472
  let textBody = "";
521
473
  if (options.extractDataFromPage) {
522
- // the text body is read as raw bytes by text tools, so the title goes in unencoded;
523
- // the < and > escaping below covers it
524
474
  textBody += (pageData.title || "") + "\n\n";
525
475
  }
526
476
  textBody += doc.body.innerText;
@@ -534,16 +484,12 @@ async function prependHTMLData(pageData, zipDataWriter, script, options, lastMod
534
484
  textBody = textBody.replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/\n +/g, "\n").replace(/\n\n\n+/g, "\n\n").trim();
535
485
  pageContent += "\n<main hidden>\n" + textBody + "\n</main>\n";
536
486
  }
537
- const displayOptions = {
538
- insertEmbeddedImage: Boolean(options.embeddedImage),
539
- insertEmbeddedScreenshotImage: Boolean(options.embeddedScreenshotImage)
540
- };
541
487
  const bootstrapBody = options.multiPageArchive ?
542
488
  "(" + inlineFunction(router) + ")(content,{extract:" +
543
489
  inlineFunction(extract) + ",display:" +
544
490
  inlineFunction(display) + "})" :
545
491
  "(" + inlineFunction(extract) + ")(content,{prompt}).then(({docContent}) => " +
546
- inlineFunction(display) + "(document,docContent," + JSON.stringify(displayOptions) + "))";
492
+ inlineFunction(display) + "(document,docContent))";
547
493
  script = "<script>" +
548
494
  script +
549
495
  "document.currentScript.remove();" +
@@ -571,14 +517,10 @@ async function prependHTMLData(pageData, zipDataWriter, script, options, lastMod
571
517
  return { extraDataOffset, pdfEntry };
572
518
  }
573
519
 
574
- // the extractor finds the zip data by identifier instead of by its position in the tree: an
575
- // element carries it as an attribute, a comment as the first characters of its data
576
520
  function getDataStartTag([startTag]) {
577
521
  if (startTag == "<!--") {
578
522
  return startTag + DATA_IDENTIFIER;
579
523
  }
580
- // the attribute belongs to the element that opens the wrapper, which is not always the whole
581
- // start tag: the CDATA rung opens with an <svg> and then a markup declaration that takes none
582
524
  const tagEnd = startTag.indexOf(">");
583
525
  return startTag.slice(0, tagEnd) + " id=" + DATA_IDENTIFIER + startTag.slice(tagEnd);
584
526
  }
@@ -591,14 +533,13 @@ function getStartHTMLArray(pageData, options, lastModDate, startTag = "") {
591
533
  const doctype = options.embeddedImage ? "" : pageData.doctype;
592
534
  const charset = options.extractDataFromPage ? "windows-1252" : "utf-8";
593
535
  const documentStart = "<html data-sfz><meta charset=" + charset + ">";
594
- const html = bom + doctype + documentStart;
595
- // the comment carries the page URL, whose length is unbounded: it is emitted after the
596
- // declaration of the character encoding, and after the embedded PDF when there is one, so
597
- // that neither the encoding declaration nor the PDF header leaves the first 1024 bytes.
598
- // it is left out of a password-protected archive, like the title below: the same URL is
599
- // in manifest.json, which is encrypted, so emitting it here would publish what the
600
- // password is meant to cover
601
- const comment = pageData.comment && !options.embeddedImage && !options.password ? "<!--" + pageData.comment + "-->" : "";
536
+ const startOffset = options.embeddedImage ?
537
+ PNG_SIGNATURE_LENGTH + PNG_IHDR_LENGTH + PNG_TEXT_CHUNK_HEADER_LENGTH : 0;
538
+ let html = bom + doctype + documentStart;
539
+ if (startOffset + new TextEncoder().encode(html).length > PRESCAN_WINDOW_LENGTH) {
540
+ html = bom + MINIMAL_DOCTYPE + documentStart;
541
+ }
542
+ const comment = pageData.comment && !options.embeddedImage && !options.password ? "<!--" + escapeCommentData(pageData.comment) + "-->" : "";
602
543
  const htmlHeadData = getHTMLHeadData(pageData, options);
603
544
  let htmlArray, pdfEntry;
604
545
  if (options.embeddedPdf) {
@@ -613,10 +554,7 @@ function getStartHTMLArray(pageData, options, lastModDate, startTag = "") {
613
554
  } else {
614
555
  const [pdfStartTag, pdfEndTag] = EMBEDDED_DATA_TAGS[pdfTagIndex];
615
556
  let htmlArray1 = new TextEncoder().encode(html + pdfStartTag);
616
- if (htmlArray1.length + localHeader.length > PDF_HEADER_MAX_OFFSET) {
617
- // PDF readers only scan the start of the file for the %PDF- header, and the page
618
- // doctype is copied verbatim: it is the one part of the prefix with no bound, so a
619
- // long one is replaced rather than pushing the header out of the scan window
557
+ if (startOffset + htmlArray1.length + localHeader.length > PRESCAN_WINDOW_LENGTH) {
620
558
  htmlArray1 = new TextEncoder().encode(bom + MINIMAL_DOCTYPE + documentStart + pdfStartTag);
621
559
  }
622
560
  const htmlArray2 = new TextEncoder().encode(pdfEndTag + comment + htmlHeadData + startTag);
@@ -638,23 +576,19 @@ function getStartHTMLArray(pageData, options, lastModDate, startTag = "") {
638
576
 
639
577
  function getHTMLHeadData(pageData, options) {
640
578
  let pageContent = "";
641
- // the title is left out of a password-protected archive: manifest.json carries it too and
642
- // is encrypted, so emitting it here would publish what the password is meant to cover
643
579
  const title = options.password ? "" : escapeHTML(pageData.title || "");
644
580
  pageContent += "<title>" + title + "</title>";
645
- // the canonical link publishes the URL the archive was saved from, for the same reason
646
- // the title above is left out of a password-protected archive
647
581
  if (options.insertCanonicalLink && !options.password) {
648
- pageContent += "<link rel=canonical href=\"" + options.url + "\">";
582
+ pageContent += "<link rel=canonical href=\"" + escapeHTML(options.url) + "\">";
649
583
  }
650
584
  if (options.insertMetaNoIndex) {
651
585
  pageContent += "<meta name=robots content=noindex>";
652
586
  }
653
587
  if (pageData.viewport) {
654
- pageContent += "<meta name=viewport content=" + JSON.stringify(pageData.viewport) + ">";
588
+ pageContent += "<meta name=viewport content=\"" + escapeHTML(pageData.viewport) + "\">";
655
589
  }
656
590
  if (options.insertMetaCSP) {
657
- const cspContent = "default-src 'none';connect-src 'self' data: blob:;font-src 'self' data: blob:;img-src 'self' data: blob:;style-src 'self' 'unsafe-inline' data: blob:;frame-src 'self' data: blob:;media-src 'self' data: blob:;script-src 'self' 'unsafe-inline' data: blob:;object-src 'self' data: blob:";
591
+ const cspContent = "default-src 'none';connect-src 'self' data: blob:;font-src 'self' data: blob:;img-src 'self' data: blob:;style-src 'self' 'unsafe-inline' data: blob:;frame-src 'self' data: blob:;media-src 'self' data: blob:;script-src 'self' 'unsafe-inline' data: blob:;object-src 'self' data: blob:;form-action 'none';base-uri 'none'";
658
592
  pageContent += `<meta http-equiv=content-security-policy content=${JSON.stringify(cspContent)}>`;
659
593
  }
660
594
  pageContent += "<style>@keyframes display-wait-message{0%{opacity:0}100%{opacity:1}}body{color:transparent}div{color:initial}body>:not(#sfz-wait-message,#sfz-error-message){display:none}</style>";
@@ -662,9 +596,17 @@ function getHTMLHeadData(pageData, options) {
662
596
  return pageContent;
663
597
  }
664
598
 
665
- // every piece of text in the prelude goes through this, wherever it is assembled: the prelude
666
- // declares a single-byte charset, and numeric character references are ASCII bytes, so they
667
- // survive it and any parser decodes them back to the original text
599
+ function escapeCommentData(value) {
600
+ let data = value.replace(/--(!?)>/g, "--$1 >");
601
+ if (data.startsWith(">") || data.startsWith("->")) {
602
+ data = " " + data;
603
+ }
604
+ if (data.endsWith("<!-")) {
605
+ data += " ";
606
+ }
607
+ return data;
608
+ }
609
+
668
610
  function escapeHTML(value) {
669
611
  return Array.from(value).map(character => {
670
612
  const codePoint = character.codePointAt(0);
@@ -681,55 +623,38 @@ function getExtraDataTagIndex(extractDataFromPageTags) {
681
623
  return tagIndex;
682
624
  }
683
625
 
684
- function findExtraDataTags(textContent, pageData, options, script, writeEntries, lastModDate, indexExtractDataFromPageTags = 0) {
626
+ function findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags = 0) {
685
627
  const regExpsTag = EXTRA_DATA_REGEXPS[indexExtractDataFromPageTags];
686
628
  const plaintextTag = EXTRA_DATA_TAGS[indexExtractDataFromPageTags][0] == "<plaintext>";
687
629
  const matchTag = !plaintextTag && (textContent.match(regExpsTag[0]) || textContent.match(regExpsTag[1]));
688
630
  if (matchTag) {
689
- if (indexExtractDataFromPageTags < EXTRA_DATA_TAGS.length - 1) {
690
- return findExtraDataTags(textContent, pageData, options, script, writeEntries, lastModDate, indexExtractDataFromPageTags + 1);
691
- } else {
692
- options.extractDataFromPage = false;
693
- return createArchive(pageData, options, script, writeEntries, lastModDate);
694
- }
631
+ return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags + 1);
695
632
  } else {
696
633
  options.extractDataFromPageTags = EXTRA_DATA_TAGS[indexExtractDataFromPageTags];
697
634
  if (options.extractDataFromPageTags[0] == "<plaintext>") {
698
- // <plaintext> cannot be closed, the file must end with the zip data
699
635
  options.preventAppendedData = true;
700
636
  }
701
- return createArchive(pageData, options, script, writeEntries, lastModDate);
637
+ return buildArchive(pageData, options, script, entriesData, zipWriterOptions);
702
638
  }
703
639
  }
704
640
 
705
- // a rung is rejected on its end pattern, which terminates the wrapper, and on its start
706
- // pattern: script data has escape states the raw text rungs do not have, where "<!--"
707
- // followed by "<script" leaves "</script>" unable to close the element at all
708
641
  function findEmbeddedDataTagIndex(text, fromIndex = 0) {
709
642
  const tagIndex = EMBEDDED_DATA_REGEXPS.slice(fromIndex, -1).findIndex(([startRegExp, endRegExp]) => !text.match(startRegExp) && !text.match(endRegExp));
710
643
  return tagIndex == -1 ? -1 : tagIndex + fromIndex;
711
644
  }
712
645
 
713
- // a face exists only while a rung can hide it. When the payload names every rung, the older
714
- // fallback emitted it unwrapped, on the grounds that the page still rendered: but the payload's
715
- // markup then joins the document, and a payload that is itself an archive contributes an
716
- // sfz-data node ahead of this file's own. A reader takes that one and extracts it, checksum and
717
- // all, with nothing to say the archive it returned is not the archive the file was built around.
718
- // Dropping the face costs a picture; keeping it costs the archive
719
- // what follows the start tag is the checksum of the chunk carrying it, four bytes only known once
720
- // the tag is chosen: a comment they open with ">" or "->" is closed by the parser there and then,
721
- // leaving the image data to be read as markup. Stepping past the comment rung means searching from
722
- // the next one, not taking it — a rung qualifies on the payload, and the payload had no say in
723
- // which rung the checksum sent the writer to
724
646
  function getImageHTMLChunk(pageData, options, lastModDate) {
725
- const embeddedImageText = TEXT_DECODER.decode(getEmbeddedImageData(options.embeddedImage));
647
+ const embeddedImageText = TEXT_DECODER.decode(getEmbeddedImageData(options.embeddedImage)) +
648
+ TEXT_DECODER.decode(new Uint8Array(4)) + TEXT_DECODER.decode(PNG_ZIP_CHUNK_TYPE_KEYWORD);
726
649
  let tagIndex = findEmbeddedDataTagIndex(embeddedImageText);
727
650
  while (tagIndex != -1) {
728
651
  const [startTag, endTag] = EMBEDDED_DATA_TAGS[tagIndex];
729
652
  const startHTMLData = getStartHTMLArray(pageData, options, lastModDate, startTag);
730
653
  const htmlData = new Uint8Array([...getLength(startHTMLData.htmlArray.length + 4), ...[0x74, 0x45, 0x58, 0x74, 0x50, 0x4e, 0x47, 0], ...startHTMLData.htmlArray]);
731
654
  const htmlDataCRC = getCRC32(htmlData, 4);
732
- if (tagIndex == 0 && (htmlDataCRC[0] == 0x3e || (htmlDataCRC[0] == 0x2d && htmlDataCRC[1] == 0x3e))) {
655
+ const wrappedText = TEXT_DECODER.decode(htmlDataCRC) + embeddedImageText;
656
+ if ((tagIndex == 0 && (htmlDataCRC[0] == 0x3e || (htmlDataCRC[0] == 0x2d && htmlDataCRC[1] == 0x3e))) ||
657
+ findEmbeddedDataTagIndex(wrappedText, tagIndex) != tagIndex) {
733
658
  tagIndex = findEmbeddedDataTagIndex(embeddedImageText, tagIndex + 1);
734
659
  } else {
735
660
  return { endTag, startHTMLData, htmlData, htmlDataCRC };
@@ -778,6 +703,7 @@ async function addPageResources(zipWriter, pageData, options, prefixName, url) {
778
703
  Promise.all(Object.keys(pageData.resources).map(async resourceType =>
779
704
  Promise.all(pageData.resources[resourceType].map(data => {
780
705
  if (resourceType == "frames") {
706
+ data.archiveTime = pageData.archiveTime;
781
707
  return addPageResources(zipWriter, data, options, prefixName + data.name, data.url);
782
708
  } else {
783
709
  return addFile(zipWriter, prefixName, data, options.disableCompression);
@@ -791,8 +717,6 @@ async function addFile(zipWriter, prefixName, data, disableCompression) {
791
717
  const dataReader = typeof data.content == "string" ? new TextReader(data.content) : new BlobReader(new Blob([new Uint8Array(data.content)]));
792
718
  const options = { password: data.password, bufferedWrite: true };
793
719
  if (!data.password) {
794
- // entry comments are stored in the central directory, which is never encrypted: with a
795
- // password the resource URLs would be readable while the same map in manifest.json is not
796
720
  options.comment = data.url && data.url.startsWith("data:") ? "data:" : data.url;
797
721
  }
798
722
  if (disableCompression || (!isCompressibleContentType(data.contentType) && NO_COMPRESSION_EXTENSIONS.includes(data.extension))) {
@@ -807,7 +731,6 @@ function isCompressibleContentType(contentType) {
807
731
 
808
732
  async function getContent() {
809
733
  const BASE64_TABLE = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";
810
- // the function is inlined in the archive as source, it cannot close over the module scope
811
734
  const DATA_IDENTIFIER = "sfz-data";
812
735
  const { Blob, XMLHttpRequest, NodeFilter, document, zip, location } = globalThis;
813
736
  const characterMap = new Map([
@@ -849,8 +772,6 @@ async function getContent() {
849
772
  const xhr = new XMLHttpRequest();
850
773
  xhr.responseType = "blob";
851
774
  xhr.open("GET", "");
852
- // a failure of the full download is recoverable when the page carries the data:
853
- // the wait message left the document intact, so the page text can still be read
854
775
  xhr.onerror = () => extractDataFromDocument();
855
776
  xhr.send();
856
777
  xhr.onreadystatechange = () => {
@@ -869,14 +790,19 @@ async function getContent() {
869
790
  getPageData();
870
791
  }
871
792
  } else {
872
- // an HTTP error status fires no error event; fall back like a network failure
873
793
  xhr.abort();
874
794
  extractDataFromDocument();
875
795
  }
876
796
  }
877
797
  };
878
798
  if (aborted) {
879
- xhr.onload = () => resolve(xhr.response);
799
+ xhr.onload = () => {
800
+ if (xhr.status === 200) {
801
+ resolve(xhr.response);
802
+ } else {
803
+ extractDataFromDocument();
804
+ }
805
+ };
880
806
  }
881
807
  }
882
808
  });
@@ -913,13 +839,12 @@ async function getContent() {
913
839
  const zipDataElement = document.querySelector("sfz-extra-data");
914
840
  if (zipDataElement) {
915
841
  const inflatedPayload = zip.inflateRaw(base64Decode(zipDataElement.textContent));
916
- const payload = new Uint32Array(inflatedPayload.buffer, inflatedPayload.byteOffset, inflatedPayload.length >> 2);
917
- // the zip data is identified, not located: its node can be moved before this runs
842
+ const payload = new DataView(inflatedPayload.buffer, inflatedPayload.byteOffset, inflatedPayload.length & -4);
918
843
  const dataElement = document.getElementById(DATA_IDENTIFIER);
919
844
  if (dataElement) {
920
845
  return decodeZipData(dataElement, payload, 0);
921
846
  }
922
- const walker = document.createTreeWalker(document.body, NodeFilter.SHOW_COMMENT);
847
+ const walker = document.createTreeWalker(document, NodeFilter.SHOW_COMMENT);
923
848
  while (walker.nextNode()) {
924
849
  if (walker.currentNode.data.startsWith(DATA_IDENTIFIER)) {
925
850
  return decodeZipData(walker.currentNode, payload, DATA_IDENTIFIER.length);
@@ -930,12 +855,9 @@ async function getContent() {
930
855
  }
931
856
 
932
857
  function decodeZipData(dataNode, payload, startIndex) {
933
- const expectedCRC32 = payload[0];
934
- const zipDataLength = payload[1];
935
- const lfCodesLength = payload[2];
936
- // the two extra bytes are the comment length field of the end of central directory
937
- // record, which the payload does not describe: left at zero they declare no comment,
938
- // which is what the recovered data holds
858
+ const expectedCRC32 = payload.getUint32(0, true);
859
+ const zipDataLength = payload.getUint32(4, true);
860
+ const lfCodesLength = payload.getUint32(8, true);
939
861
  const zipData = new Uint8Array(zipDataLength + 2);
940
862
  const { textContent } = dataNode;
941
863
  let offset = 0;
@@ -944,9 +866,11 @@ async function getContent() {
944
866
  for (let index = startIndex; index < textContent.length && offset < zipDataLength; index++) {
945
867
  const charCode = textContent.charCodeAt(index);
946
868
  if (charCode == 10) {
947
- const lfCode = (payload[3 + (indexLFCode >> 4)] >>> ((indexLFCode & 15) * 2)) & 3;
869
+ const lfCode = (payload.getUint32(12 + (indexLFCode >> 4) * 4, true) >>> ((indexLFCode & 15) * 2)) & 3;
948
870
  indexLFCode++;
949
- if (lfCode == 0) {
871
+ if (lfCode == 3) {
872
+ throw new Error("Unsupported newline code in the extracted zip data");
873
+ } else if (lfCode == 0) {
950
874
  writeByte(10);
951
875
  } else {
952
876
  writeByte(13);