single-file-core 1.5.126 → 1.5.128

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -449,12 +449,17 @@ class ProcessorHelperCommon {
449
449
  await this.processFontFaceRules(ruleData.block.children, sheetIndex, fontsDetails.layers.get("layer-" + sheetIndex + "-" + layerIndex + "-" + layerText), fonts, fontTests, stats);
450
450
  layerIndex++;
451
451
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
452
- const key = this.getFontKey(ruleData);
453
- const fontInfo = fontsDetails.fonts.get(key);
454
- if (fontInfo && fontsDetails.lastRules.get(key) == ruleData) {
455
- const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
456
- if (!keptRule) {
452
+ const fontInfo = fontsDetails.fonts.get(ruleData);
453
+ if (fontInfo) {
454
+ const ruleKey = this.getFontKey(ruleData) + " " + this.getPropertyValue(ruleData, "src");
455
+ if (fontsDetails.emittedFonts.has(ruleKey)) {
457
456
  removedRules.push(cssRule);
457
+ } else {
458
+ fontsDetails.emittedFonts.add(ruleKey);
459
+ const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
460
+ if (!keptRule) {
461
+ removedRules.push(cssRule);
462
+ }
458
463
  }
459
464
  } else {
460
465
  removedRules.push(cssRule);
@@ -486,28 +491,19 @@ class ProcessorHelperCommon {
486
491
  layerIndex++;
487
492
  this.getFontsDetails(doc, ruleData.block.children, sheetIndex, fontsDetails);
488
493
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face" && ruleData.block && ruleData.block.children) {
489
- const fontKey = this.getFontKey(ruleData);
490
- let fontInfo = mediaFontsDetails.fonts.get(fontKey);
491
- if (!fontInfo) {
492
- fontInfo = [];
493
- mediaFontsDetails.fonts.set(fontKey, fontInfo);
494
- }
495
- mediaFontsDetails.lastRules.set(fontKey, ruleData);
494
+ const fontInfo = [];
495
+ mediaFontsDetails.fonts.set(ruleData, fontInfo);
496
496
  const src = this.getPropertyValue(ruleData, "src");
497
497
  if (src) {
498
498
  const fontSources = src.match(REGEXP_URL_FUNCTION);
499
499
  if (fontSources) {
500
- const ruleSources = [];
501
- fontSources.forEach(source => {
500
+ fontSources.forEach(fontSource => {
501
+ const source = fontSource.match(REGEXP_FONT_SRC)[1];
502
502
  if (fontInfo.includes(source)) {
503
503
  fontInfo.splice(fontInfo.indexOf(source), 1);
504
504
  }
505
- if (ruleSources.includes(source)) {
506
- ruleSources.splice(ruleSources.indexOf(source), 1);
507
- }
508
- ruleSources.unshift(source);
505
+ fontInfo.unshift(source);
509
506
  });
510
- ruleSources.forEach(source => fontInfo.push(source));
511
507
  }
512
508
  }
513
509
  }
@@ -520,7 +516,7 @@ class ProcessorHelperCommon {
520
516
  medias: new Map(),
521
517
  supports: new Map(),
522
518
  layers: new Map(),
523
- lastRules: new Map()
519
+ emittedFonts: new Set()
524
520
  };
525
521
  }
526
522
 
package/core/util.js CHANGED
@@ -63,6 +63,8 @@ const CONTENT_TYPE_EXTENSIONS = {
63
63
  "font/collection": ".ttc"
64
64
  };
65
65
  const CONTENT_TYPE_OCTET_STREAM = "application/octet-stream";
66
+ const TRANSPORT_STREAM_SYNC_BYTE = 71;
67
+ const TRANSPORT_STREAM_PACKET_SIZE = 188;
66
68
  const CONTENT_TYPES_HTML = ["text/html", "application/xhtml+xml"];
67
69
  const EXPECTED_TYPES_MEDIA = ["font", "image", "video", "audio"];
68
70
 
@@ -295,11 +297,11 @@ function getInstance(utilOptions) {
295
297
  } catch (error) {
296
298
  // ignored
297
299
  }
298
- if (!contentType || (contentType == CONTENT_TYPE_OCTET_STREAM && options.asBinary)) {
299
- contentType = guessMIMEType(options.expectedType, buffer);
300
- if (!contentType) {
301
- contentType = options.contentType ? options.contentType : options.asBinary ? CONTENT_TYPE_OCTET_STREAM : "";
302
- }
300
+ const guessedContentType = guessMIMEType(options.expectedType, buffer);
301
+ if (guessedContentType) {
302
+ contentType = guessedContentType;
303
+ } else if (!contentType || (contentType == CONTENT_TYPE_OCTET_STREAM && options.asBinary)) {
304
+ contentType = options.contentType ? options.contentType : options.asBinary ? CONTENT_TYPE_OCTET_STREAM : "";
303
305
  }
304
306
  if (!charset && options.charset) {
305
307
  charset = options.charset;
@@ -403,6 +405,24 @@ function guessMIMEType(expectedType, buffer) {
403
405
  if (compareBytes([255, 255, 255], [255, 216, 255])) {
404
406
  return "image/jpeg";
405
407
  }
408
+ if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 97, 118, 105, 102]) ||
409
+ compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 97, 118, 105, 115])) {
410
+ return "image/avif";
411
+ }
412
+ if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 105, 99]) ||
413
+ compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 105, 120]) ||
414
+ compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 118, 99]) ||
415
+ compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 104, 101, 118, 120])) {
416
+ return "image/heic";
417
+ }
418
+ if (compareBytes([255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 12, 74, 88, 76, 32, 13, 10, 135, 10]) ||
419
+ compareBytes([255, 255], [255, 10])) {
420
+ return "image/jxl";
421
+ }
422
+ if (compareBytes([255, 255, 255, 255], [73, 73, 42, 0]) ||
423
+ compareBytes([255, 255, 255, 255], [77, 77, 0, 42])) {
424
+ return "image/tiff";
425
+ }
406
426
  }
407
427
  if (expectedType == "font") {
408
428
  if (compareBytes([0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 255, 255],
@@ -444,7 +464,7 @@ function guessMIMEType(expectedType, buffer) {
444
464
  if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 51, 103])) {
445
465
  return "video/3gpp";
446
466
  }
447
- if (compareBytes([255], [71])) {
467
+ if (isTransportStream()) {
448
468
  return "video/mp2t";
449
469
  }
450
470
  }
@@ -475,6 +495,14 @@ function guessMIMEType(expectedType, buffer) {
475
495
  }
476
496
  }
477
497
 
498
+ function isTransportStream() {
499
+ const value = new Uint8Array(buffer);
500
+ return value.length > TRANSPORT_STREAM_PACKET_SIZE * 2 &&
501
+ value[0] == TRANSPORT_STREAM_SYNC_BYTE &&
502
+ value[TRANSPORT_STREAM_PACKET_SIZE] == TRANSPORT_STREAM_SYNC_BYTE &&
503
+ value[TRANSPORT_STREAM_PACKET_SIZE * 2] == TRANSPORT_STREAM_SYNC_BYTE;
504
+ }
505
+
478
506
  function compareBytes(mask, pattern) {
479
507
  let patternMatch = true, index = 0;
480
508
  if (buffer.byteLength >= pattern.length) {
@@ -357,7 +357,7 @@ face adds, then the regions the PNG face adds.
357
357
  | `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
358
358
  | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the appended-data budget (§5.2) or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
359
359
  | `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
360
- | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag set as on every other entry — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
360
+ | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag left clear as on every other ASCII name — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
361
361
  | `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
362
362
  | `pdf-central-record` | ZIP | PDF face with HTML | The central-directory record for `page.pdf`, injected *before* the writer's own central directory. The start of the central directory is the one place a record can be added without moving any offset the writer already committed, and it makes `page.pdf` the first entry ZIP tools list (§6). |
363
363
  | `png-signature · IHDR` | PNG | PNG face | The 8-byte PNG signature and the `IHDR` chunk declaring the source image's dimensions — the first 33 bytes of the file. |
@@ -1291,18 +1291,26 @@ How a name is encoded is ZIP's own business, not this format's: bit 11 of the ge
1291
1291
  purpose bit flag selects UTF-8, and its absence selects the legacy code page. This
1292
1292
  document adds two requirements to that and specifies nothing else about it.
1293
1293
 
1294
- **A writer MUST set bit 11 on every entry**, not only on the entries whose names need
1295
- it. The two encodings agree over printable ASCII, so setting it unconditionally costs
1296
- nothing, and it means no name in the archive is decoded through the legacy path at all.
1294
+ **A writer MUST set bit 11 whenever a name or a comment needs it**, and the rule for
1295
+ when it does is ZIP's, not this format's: an encoded name or comment holding a byte
1296
+ outside printable ASCII needs it, one holding only printable ASCII does not, since the
1297
+ two encodings agree there. Control characters count as needing it, the legacy code page
1298
+ mapping them to graphic characters rather than to themselves. Setting it on names that
1299
+ do not need it is allowed and used to be required here; it was dropped because readers
1300
+ disagree about the flag more than they disagree about ASCII, so the safest name is the
1301
+ one that does not exercise the question. A writer MUST NOT set it on a name it then
1302
+ encodes in the legacy code page, which is the one combination that is simply wrong.
1297
1303
 
1298
1304
  **A reader MUST honor the flag** rather than assume one encoding, and MUST expect to
1299
- meet a clear one: the hand-built `page.pdf` records (§3.1, §6) are the only ones the
1300
- reference writer does not produce through its ZIP writer, and an archive may carry them
1301
- with no flag set at all. That single entry is then decoded as legacy while every other
1302
- name in the same file is UTF-8.
1303
- Its name is ASCII, where the two encodings agree, so a correct reader sees `page.pdf`
1304
- either way — but a reader that hardcodes UTF-8 on the strength of the other entries has
1305
- not covered it.
1305
+ meet a clear one — which, in an archive from the reference writer, is most of them:
1306
+ that writer percent-encodes every name it produces, so every name is printable ASCII
1307
+ and carries no flag, while an entry comment holding the original URL of a resource can
1308
+ carry one when that URL is not ASCII. The hand-built `page.pdf` records (§3.1, §6) are
1309
+ the only ones the reference writer does not produce through its ZIP writer, and they
1310
+ follow the same rule: `page.pdf` is ASCII, so they carry no flag either, and no entry
1311
+ in the archive is decoded differently from the rest. Archives written before this rule
1312
+ was relaxed carry the flag on every entry instead. Both decode identically, which is
1313
+ the point, but a reader that hardcodes either answer meets the other one eventually.
1306
1314
 
1307
1315
  A name is not a path. §7.3's rule that entry names are untrusted applies to the decoded
1308
1316
  name, and decoding is the step before that check, not a substitute for it.
@@ -1399,7 +1407,7 @@ pages can stop at the first row; the files it produces are accepted by every rea
1399
1407
  wrapper start tag chosen for the PDF payload (§5.1), the hand-built `page.pdf`
1400
1408
  local file header, the PDF document, the wrapper end tag, and record the local
1401
1409
  header's absolute position; then resume the prologue. The reference writer's
1402
- header declares version 2.0, the language encoding flag alone, method STORE, the
1410
+ header declares version 2.0, no general purpose bit flag, method STORE, the
1403
1411
  build's modification date in DOS form, the precomputed CRC-32, the document's
1404
1412
  length as both sizes, and no extra field; its central record adds a Unix
1405
1413
  "made by" version and external attributes of a regular file, mode 0644.
@@ -1892,6 +1900,7 @@ predicts.
1892
1900
  | August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
1893
1901
  | August 2026 | Core 1.5.119: the inlined ZIP library is built ASCII-only, and §2.1 now requires it of any bootstrap. Its CP437 table had been emitted as literal characters, which the page re-decoded as windows-1252, growing the table from 256 entries to 508 and shifting every lookup by 60 — so the one entry read without the UTF-8 flag, `page.pdf`, came back mangled and no archive with a PDF face extracted in any engine (§5.8) |
1894
1902
  | August 2026 | Core 1.5.120: the hand-built `page.pdf` records set the language encoding flag, like every entry the ZIP writer produces (§5.8). Its name is ASCII, so no decoded name changes; what changes is that no entry in an archive is read through CP437 any more, closing the path the 1.5.119 defect surfaced on |
1903
+ | September 2026 | §5.8 no longer requires bit 11 on every entry, deferring to ZIP's own rule: the flag is set when a name or a comment holds a byte outside printable ASCII, and left clear otherwise, because readers disagree about the flag more than they disagree about ASCII. The reference writer's names are all percent-encoded, so in practice none of them carries it now, and the hand-built `page.pdf` records follow the writer instead of overriding it — reversing the 1.5.120 row below, whose reason was that `page.pdf` would otherwise be the only entry read through the legacy path. It no longer is: every name in the archive takes the same path again, the other one |
1895
1904
  | September 2026 | Core 1.5.126: the appended-data budget becomes the `maxAppendedDataLength` writer option and its default drops from 65535 to 16361 bytes, so the EOCD record stays inside libarchive's scan and `bsdtar` opens archives it used to reject (§5.2, §8.1). The 65535-byte comment ceiling is now a separate limit, stated in §4.2: a budget raised past it produces a run that cannot be declared |
1896
1905
 
1897
1906
  This document was itself revised in August 2026, against core 1.5.108, after several
package/package.json CHANGED
@@ -1,26 +1,26 @@
1
1
  {
2
- "name": "single-file-core",
3
- "version": "1.5.126",
4
- "description": "SingleFile Core",
5
- "author": "Gildas Lormeau",
6
- "license": "AGPL-3.0-or-later",
7
- "scripts": {
8
- "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
9
- "bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
10
- "bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
11
- "bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
12
- "bump-commit": "git commit -m \"bump up version\" package.json package-lock.json"
13
- },
14
- "repository": {
15
- "type": "git",
16
- "url": "git+https://github.com/gildas-lormeau/single-file-core.git"
17
- },
18
- "bugs": {
19
- "url": "https://github.com/gildas-lormeau/single-file-core/issues"
20
- },
21
- "homepage": "https://github.com/gildas-lormeau/single-file-core#readme",
22
- "devDependencies": {
23
- "@eslint/js": "^9.39.5",
24
- "eslint": "^10.9.1"
25
- }
2
+ "name": "single-file-core",
3
+ "version": "1.5.128",
4
+ "description": "SingleFile Core",
5
+ "author": "Gildas Lormeau",
6
+ "license": "AGPL-3.0-or-later",
7
+ "scripts": {
8
+ "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/font-face-composite.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/pages-router.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/content-type-sniffing.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
9
+ "bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
10
+ "bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
11
+ "bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
12
+ "bump-commit": "git commit -m \"bump up version\" package.json package-lock.json"
13
+ },
14
+ "repository": {
15
+ "type": "git",
16
+ "url": "git+https://github.com/gildas-lormeau/single-file-core.git"
17
+ },
18
+ "bugs": {
19
+ "url": "https://github.com/gildas-lormeau/single-file-core/issues"
20
+ },
21
+ "homepage": "https://github.com/gildas-lormeau/single-file-core#readme",
22
+ "devDependencies": {
23
+ "@eslint/js": "^9.39.5",
24
+ "eslint": "^10.9.1"
25
+ }
26
26
  }
@@ -32,7 +32,8 @@ import {
32
32
  } from "./../../vendor/zip/zip.js";
33
33
  import {
34
34
  createArchive,
35
- escapeHTML
35
+ escapeHTML,
36
+ PROCESS_OPTION_NAMES
36
37
  } from "./compression.js";
37
38
 
38
39
  const browser = globalThis.browser;
@@ -46,6 +47,13 @@ const TOC_STYLE = "body{font-family:system-ui,sans-serif;margin:2em auto;max-wid
46
47
  "summary{cursor:pointer;font-weight:bold;margin:.5em 0}" +
47
48
  "details{padding-left:1em}ul{margin:.25em 0;padding-left:1.5em}" +
48
49
  "@media(prefers-color-scheme:dark){body{background-color:#111;color:#eee}a{color:#8ab4f8}a:visited{color:#c58af9}}";
50
+ const ARCHIVE_EXCLUDED_OPTION_NAMES = [
51
+ "createRootDirectory",
52
+ "disableCompression",
53
+ "insertTextBody",
54
+ "password",
55
+ "url"
56
+ ];
49
57
  const COMMENT_HEADER = "Page saved with SingleFile";
50
58
  const SYMLINK_UNIX_MODE = 0o120777;
51
59
 
@@ -80,18 +88,11 @@ async function createPagesArchive(pages, options) {
80
88
  };
81
89
  const archiveOptions = {
82
90
  url: pages[0].url,
83
- multiPageArchive: true,
84
- selfExtractingArchive: options.selfExtractingArchive,
85
- extractDataFromPage: options.extractDataFromPage,
86
- preventAppendedData: options.preventAppendedData,
87
- declareAppendedData: options.declareAppendedData,
88
- embeddedPdf: options.embeddedPdf,
89
- embeddedImage: options.embeddedImage,
90
- includeBOM: options.includeBOM,
91
- insertMetaCSP: options.insertMetaCSP,
92
- insertCanonicalLink: options.insertCanonicalLink,
93
- insertMetaNoIndex: options.insertMetaNoIndex
91
+ multiPageArchive: true
94
92
  };
93
+ PROCESS_OPTION_NAMES
94
+ .filter(name => !ARCHIVE_EXCLUDED_OPTION_NAMES.includes(name) && !(name in archiveOptions))
95
+ .forEach(name => archiveOptions[name] = options[name]);
95
96
  const writtenEntries = options.dedupPages ? new Map() : undefined;
96
97
  const aliases = {};
97
98
  const blob = await createArchive(pageData, archiveOptions, options.zipScript, async zipWriter => {
@@ -100,16 +101,12 @@ async function createPagesArchive(pages, options) {
100
101
  const zipReader = new ZipReader(new Uint8ArrayReader(await pages[pageIndex].getData()));
101
102
  for (const entry of await zipReader.getEntries()) {
102
103
  const filename = pagePath + entry.filename;
103
- const rawData = await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
104
+ const rawData = entry.directory ? undefined : await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
104
105
  const canonicalFilename = writtenEntries && findDuplicate(writtenEntries, filename, entry, rawData);
105
106
  if (canonicalFilename === undefined) {
106
- await zipWriter.add(filename, new Uint8ArrayReader(rawData), {
107
+ await zipWriter.add(filename, entry.directory ? null : new Uint8ArrayReader(rawData), {
107
108
  passThrough: true,
108
- compressionMethod: entry.compressionMethod,
109
- uncompressedSize: entry.uncompressedSize,
110
- crc32: entry.crc32,
111
- comment: entry.comment,
112
- lastModDate: entry.lastModDate
109
+ entry
113
110
  });
114
111
  } else {
115
112
  aliases[filename] = canonicalFilename;
@@ -202,7 +202,7 @@ async function router(content, { extract, display }) {
202
202
  function parseRoute() {
203
203
  const hash = location.hash;
204
204
  const routed = !hash || hash.startsWith(ROUTE_PREFIX);
205
- let path = pages[0].path;
205
+ let path = routed && tocEntry ? TOC_ROUTE : pages[0].path;
206
206
  let fragment;
207
207
  if (routed && hash) {
208
208
  ({ path, fragment } = parseRouteHash(hash));
@@ -40,9 +40,7 @@ import {
40
40
  router
41
41
  } from "./compression-router.js";
42
42
 
43
- const { Blob, fetch, TextEncoder, TextDecoder, DOMParser } = globalThis;
44
-
45
- const TEXT_DECODER = new TextDecoder("windows-1252");
43
+ const { Blob, fetch, TextEncoder, DOMParser } = globalThis;
46
44
 
47
45
  const COMPRESSIBLE_CONTENT_TYPES = ["application/javascript", "application/x-javascript", "application/ecmascript", "application/json", "application/ld+json", "application/manifest+json", "application/xml", "application/xhtml+xml", "application/rss+xml", "application/atom+xml", "image/svg+xml"];
48
46
  const TEXT_CONTENT_TYPE_PREFIX = "text/";
@@ -63,19 +61,20 @@ const EMBEDDED_DATA_TAGS = [
63
61
  ...EXTRA_DATA_TAGS,
64
62
  ];
65
63
  const DATA_IDENTIFIER = "sfz-data";
66
- const EXTRA_DATA_REGEXPS = [
67
- [/<script/i, /<\/script[\t\n\f\r />]/i],
68
- [/<style/i, /<\/style[\t\n\f\r />]/i],
69
- [/<noframes/i, /<\/noframes[\t\n\f\r />]/i],
70
- [/<noembed/i, /<\/noembed[\t\n\f\r />]/i],
71
- [/<iframe/i, /<\/iframe[\t\n\f\r />]/i],
72
- [/<xmp/i, /<\/xmp[\t\n\f\r />]/i],
73
- [/<!\[CDATA\[/i, /\]\]>/],
74
- [/<plaintext/i, /<\/plaintext[\t\n\f\r />]/i]
64
+ const TAG_NAME_TERMINATORS = "\t\n\f\r />";
65
+ const EXTRA_DATA_PATTERNS = [
66
+ [["<script"], ["</script", TAG_NAME_TERMINATORS]],
67
+ [["<style"], ["</style", TAG_NAME_TERMINATORS]],
68
+ [["<noframes"], ["</noframes", TAG_NAME_TERMINATORS]],
69
+ [["<noembed"], ["</noembed", TAG_NAME_TERMINATORS]],
70
+ [["<iframe"], ["</iframe", TAG_NAME_TERMINATORS]],
71
+ [["<xmp"], ["</xmp", TAG_NAME_TERMINATORS]],
72
+ [["<![CDATA["], ["]]>"]],
73
+ [["<plaintext"], ["</plaintext", TAG_NAME_TERMINATORS]]
75
74
  ];
76
- const EMBEDDED_DATA_REGEXPS = [
77
- [/<!--/i, /--!?>|<!-$/i],
78
- ...EXTRA_DATA_REGEXPS,
75
+ const EMBEDDED_DATA_PATTERNS = [
76
+ [["<!--"], ["-->"], ["--!>"], ["<!-", undefined, true]],
77
+ ...EXTRA_DATA_PATTERNS,
79
78
  ];
80
79
  const CRC32_TABLE = new Uint32Array(256).map((_, indexTable) => {
81
80
  let crc = indexTable;
@@ -106,7 +105,6 @@ const CENTRAL_FILE_HEADER_SIGNATURE = 0x02014b50;
106
105
  const END_OF_CENTRAL_DIR_SIGNATURE = 0x06054b50;
107
106
  const ZIP64_END_OF_CENTRAL_DIR_SIGNATURE = 0x06064b50;
108
107
  const ZIP64_END_OF_CENTRAL_DIR_LOCATOR_SIGNATURE = 0x07064b50;
109
- const LANGUAGE_ENCODING_FLAG = 0x0800;
110
108
 
111
109
  const browser = globalThis.browser;
112
110
 
@@ -229,18 +227,14 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
229
227
  const lfCodes = [];
230
228
  let crc32 = -1;
231
229
  if (!options.extractDataFromPageTags || options.extractDataFromPageTags[0] != "<plaintext>") {
232
- const textContent = TEXT_DECODER.decode(data.subarray(startOffset));
230
+ const zipData = data.subarray(startOffset);
233
231
  if (options.extractDataFromPageTags) {
234
232
  const tagIndex = getExtraDataTagIndex(options.extractDataFromPageTags);
235
- const regExpsTag = EXTRA_DATA_REGEXPS[tagIndex];
236
- if (textContent.match(regExpsTag[0]) || textContent.match(regExpsTag[1])) {
237
- return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, tagIndex + 1);
238
- }
239
- } else {
240
- const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[0];
241
- if (textContent.match(startRegExp) || textContent.match(endRegExp)) {
242
- return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions);
233
+ if (containsDataPattern(zipData, EXTRA_DATA_PATTERNS[tagIndex])) {
234
+ return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, tagIndex + 1);
243
235
  }
236
+ } else if (containsDataPattern(zipData, EMBEDDED_DATA_PATTERNS[0])) {
237
+ return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions);
244
238
  }
245
239
  }
246
240
  if (options.extractDataFromPage) {
@@ -337,9 +331,7 @@ function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, opt
337
331
  const tail = pageContent.slice(zipDataEnd - WRAPPER_PATTERN_WINDOW_LENGTH, zipDataEnd + COMMENT_LENGTH_FIELD_LENGTH);
338
332
  new DataView(tail.buffer).setUint16(WRAPPER_PATTERN_WINDOW_LENGTH, appendedDataLength, true);
339
333
  const tagIndex = options.extractDataFromPageTags ? getExtraDataTagIndex(options.extractDataFromPageTags) + 1 : 0;
340
- const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[tagIndex];
341
- const tailText = TEXT_DECODER.decode(tail);
342
- return !tailText.match(startRegExp) && !tailText.match(endRegExp);
334
+ return !containsDataPattern(tail, EMBEDDED_DATA_PATTERNS[tagIndex]);
343
335
  }
344
336
 
345
337
  function getCRC32(data, indexData = 0) {
@@ -365,7 +357,6 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
365
357
  const localHeaderView = new DataView(localHeader.buffer);
366
358
  localHeaderView.setUint32(0, LOCAL_FILE_HEADER_SIGNATURE, true);
367
359
  localHeaderView.setUint16(4, 20, true);
368
- localHeaderView.setUint16(6, LANGUAGE_ENCODING_FLAG, true);
369
360
  localHeaderView.setUint16(10, dosTime, true);
370
361
  localHeaderView.setUint16(12, dosDate, true);
371
362
  localHeaderView.setUint32(14, crc32, true);
@@ -378,7 +369,6 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
378
369
  centralRecordView.setUint32(0, CENTRAL_FILE_HEADER_SIGNATURE, true);
379
370
  centralRecordView.setUint16(4, 0x0300, true);
380
371
  centralRecordView.setUint16(6, 20, true);
381
- centralRecordView.setUint16(8, LANGUAGE_ENCODING_FLAG, true);
382
372
  centralRecordView.setUint16(12, dosTime, true);
383
373
  centralRecordView.setUint16(14, dosDate, true);
384
374
  centralRecordView.setUint32(16, crc32, true);
@@ -552,8 +542,7 @@ function getStartHTMLArray(pageData, options, lastModDate, startTag = "") {
552
542
  const embeddedPdf = new Uint8Array(options.embeddedPdf);
553
543
  pdfEntry = options.preventEmbeddedPdfEntry ? undefined : getPDFEntry(embeddedPdf, lastModDate);
554
544
  const localHeader = pdfEntry ? pdfEntry.localHeader : new Uint8Array(0);
555
- const embeddedPdfText = TEXT_DECODER.decode(localHeader) + TEXT_DECODER.decode(embeddedPdf);
556
- const pdfTagIndex = findEmbeddedDataTagIndex(embeddedPdfText);
545
+ const pdfTagIndex = findEmbeddedDataTagIndex(concatArrays(localHeader, embeddedPdf));
557
546
  if (pdfTagIndex == -1) {
558
547
  dropUnhiddenFace(options, "embeddedPdf", EMBEDDED_PDF_LABEL);
559
548
  pdfEntry = undefined;
@@ -629,12 +618,11 @@ function getExtraDataTagIndex(extractDataFromPageTags) {
629
618
  return tagIndex;
630
619
  }
631
620
 
632
- function findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags = 0) {
633
- const regExpsTag = EXTRA_DATA_REGEXPS[indexExtractDataFromPageTags];
621
+ function findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags = 0) {
634
622
  const plaintextTag = EXTRA_DATA_TAGS[indexExtractDataFromPageTags][0] == "<plaintext>";
635
- const matchTag = !plaintextTag && (textContent.match(regExpsTag[0]) || textContent.match(regExpsTag[1]));
623
+ const matchTag = !plaintextTag && containsDataPattern(zipData, EXTRA_DATA_PATTERNS[indexExtractDataFromPageTags]);
636
624
  if (matchTag) {
637
- return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags + 1);
625
+ return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags + 1);
638
626
  } else {
639
627
  options.extractDataFromPageTags = EXTRA_DATA_TAGS[indexExtractDataFromPageTags];
640
628
  if (options.extractDataFromPageTags[0] == "<plaintext>") {
@@ -644,24 +632,93 @@ function findExtraDataTags(textContent, pageData, options, script, entriesData,
644
632
  }
645
633
  }
646
634
 
647
- function findEmbeddedDataTagIndex(text, fromIndex = 0) {
648
- const tagIndex = EMBEDDED_DATA_REGEXPS.slice(fromIndex, -1).findIndex(([startRegExp, endRegExp]) => !text.match(startRegExp) && !text.match(endRegExp));
635
+ function findEmbeddedDataTagIndex(data, fromIndex = 0) {
636
+ const tagIndex = EMBEDDED_DATA_PATTERNS.slice(fromIndex, -1).findIndex(patterns => !containsDataPattern(data, patterns));
649
637
  return tagIndex == -1 ? -1 : tagIndex + fromIndex;
650
638
  }
651
639
 
640
+ function containsDataPattern(data, patterns) {
641
+ const patternsByCharCode = new Map();
642
+ for (const pattern of patterns) {
643
+ const [text, , atEnd] = pattern;
644
+ if (atEnd) {
645
+ if (matchesDataPatternAt(data, pattern, data.length - text.length)) {
646
+ return true;
647
+ }
648
+ } else {
649
+ const charCode = text.charCodeAt(0);
650
+ indexPatternByCharCode(patternsByCharCode, charCode, pattern);
651
+ const alternateCharCode = getAlternateCharCode(charCode);
652
+ if (alternateCharCode != -1) {
653
+ indexPatternByCharCode(patternsByCharCode, alternateCharCode, pattern);
654
+ }
655
+ }
656
+ }
657
+ for (const [charCode, candidates] of patternsByCharCode) {
658
+ for (let index = data.indexOf(charCode); index != -1; index = data.indexOf(charCode, index + 1)) {
659
+ for (let indexCandidate = 0; indexCandidate < candidates.length; indexCandidate++) {
660
+ if (matchesDataPatternAt(data, candidates[indexCandidate], index)) {
661
+ return true;
662
+ }
663
+ }
664
+ }
665
+ }
666
+ return false;
667
+ }
668
+
669
+ function indexPatternByCharCode(patternsByCharCode, charCode, pattern) {
670
+ if (!patternsByCharCode.has(charCode)) {
671
+ patternsByCharCode.set(charCode, []);
672
+ }
673
+ patternsByCharCode.get(charCode).push(pattern);
674
+ }
675
+
676
+ function matchesDataPatternAt(data, [text, terminators], index) {
677
+ const textLength = text.length;
678
+ if (index < 0 || index + textLength > data.length) {
679
+ return false;
680
+ }
681
+ for (let indexText = 0; indexText < textLength; indexText++) {
682
+ const charCode = text.charCodeAt(indexText);
683
+ const code = data[index + indexText];
684
+ if (code != charCode && code != getAlternateCharCode(charCode)) {
685
+ return false;
686
+ }
687
+ }
688
+ return terminators === undefined || isTerminatorCode(terminators, data[index + textLength]);
689
+ }
690
+
691
+ function isTerminatorCode(terminators, code) {
692
+ return code !== undefined && terminators.includes(String.fromCharCode(code));
693
+ }
694
+
695
+ function getAlternateCharCode(charCode) {
696
+ const lowerCharCode = charCode | 0x20;
697
+ return lowerCharCode >= 0x61 && lowerCharCode <= 0x7a ? charCode ^ 0x20 : -1;
698
+ }
699
+
700
+ function concatArrays(...arrays) {
701
+ const result = new Uint8Array(arrays.reduce((length, array) => length + array.length, 0));
702
+ let offset = 0;
703
+ arrays.forEach(array => {
704
+ result.set(array, offset);
705
+ offset += array.length;
706
+ });
707
+ return result;
708
+ }
709
+
652
710
  function getImageHTMLChunk(pageData, options, lastModDate) {
653
- const embeddedImageText = TEXT_DECODER.decode(getEmbeddedImageData(options.embeddedImage)) +
654
- TEXT_DECODER.decode(new Uint8Array(4)) + TEXT_DECODER.decode(PNG_ZIP_CHUNK_TYPE_KEYWORD);
655
- let tagIndex = findEmbeddedDataTagIndex(embeddedImageText);
711
+ const embeddedImageData = concatArrays(getEmbeddedImageData(options.embeddedImage), new Uint8Array(4), PNG_ZIP_CHUNK_TYPE_KEYWORD);
712
+ let tagIndex = findEmbeddedDataTagIndex(embeddedImageData);
656
713
  while (tagIndex != -1) {
657
714
  const [startTag, endTag] = EMBEDDED_DATA_TAGS[tagIndex];
658
715
  const startHTMLData = getStartHTMLArray(pageData, options, lastModDate, startTag);
659
716
  const htmlData = new Uint8Array([...getLength(startHTMLData.htmlArray.length + 4), ...[0x74, 0x45, 0x58, 0x74, 0x50, 0x4e, 0x47, 0], ...startHTMLData.htmlArray]);
660
717
  const htmlDataCRC = getCRC32(htmlData, 4);
661
- const wrappedText = TEXT_DECODER.decode(htmlDataCRC) + embeddedImageText;
718
+ const wrappedData = concatArrays(htmlDataCRC, embeddedImageData);
662
719
  if ((tagIndex == 0 && (htmlDataCRC[0] == 0x3e || (htmlDataCRC[0] == 0x2d && htmlDataCRC[1] == 0x3e))) ||
663
- findEmbeddedDataTagIndex(wrappedText, tagIndex) != tagIndex) {
664
- tagIndex = findEmbeddedDataTagIndex(embeddedImageText, tagIndex + 1);
720
+ findEmbeddedDataTagIndex(wrappedData, tagIndex) != tagIndex) {
721
+ tagIndex = findEmbeddedDataTagIndex(embeddedImageData, tagIndex + 1);
665
722
  } else {
666
723
  return { endTag, startHTMLData, htmlData, htmlDataCRC };
667
724
  }
@@ -34,7 +34,9 @@ any check failed.
34
34
  | `css-property-filter.js` | That the declaration filter keeps a property css-tree's dictionary does not know (`stop-color`, `flood-opacity`, anything newer than the pinned build) and still drops a genuinely invalid value. |
35
35
  | `adopted-stylesheets-hook.js` | That the page-world hook answers the adopted-stylesheets request for a CLOSED shadow root, which its host does not expose. |
36
36
  | `inlined-functions.js` | That a function serialized into a self-extracting archive names nothing outside itself. An import survives bundling and still reads correctly, and the archive then throws a bare `ReferenceError` and renders nothing. |
37
- | `pages-archive.js` | That `createPagesArchive` packs several pages into one archive correctly: the first page at the root and the others in folders, the manifest, the symlink a deduplicated entry leaves behind, and the escaping of crawled titles in both tables of contents. |
37
+ | `pages-archive.js` | That `createPagesArchive` packs several pages into one archive correctly: the first page at the root and the others in folders, the manifest, the symlink a deduplicated entry leaves behind, and the escaping of crawled titles in both tables of contents. Also that the options reaching the archive writer are derived from `PROCESS_OPTION_NAMES` rather than hand-listed — a hand copy dropped `maxAppendedDataLength` for months — and that `password` stays out of that derivation, since forwarding it would make the writer withhold the prologue's title as if the archive were encrypted while the table of contents and every entry comment still rode in that same cleartext prologue. |
38
+ | `pages-router.js` | That the router opens a multi-page archive on the page a reader expects: the table of contents when the archive stores one, the first page when it does not, and the page a route in the hash names whatever else is stored. The router is inlined into the archive as source text and only ever ran inside a saved page, so nothing drove it before; the table of contents shipped stored, routable and unreachable. |
39
+ | `content-type-sniffing.js` | That a magic-byte match beats the `Content-Type` header, and that the header survives when nothing matches. science.org serves its woff2 files as `text/plain`, which used to be trusted, so the fonts were embedded as `data:text/plain` and the SFZ writer deflated a file that is already Brotli-compressed. Two cases guard the rules themselves rather than the outcome: a video whose first byte is `G` must not be relabelled `video/mp2t`, and the generic `mif1` HEIF brand must identify nothing, because an AVIF can carry it and calling it HEIC would name a format no browser decodes. |
38
40
  | `entry-compression.js` | That an entry is deflated or stored on the content type the server sent, not on the extension alone — a module served as `text/javascript` from a `.ts` URL used to go in uncompressed — and that an unrecognized `application/octet-stream` still stays stored. |
39
41
  | `filename-max-length.js` | That `formatFilename` counts the ellipsis as well as the extension in the budget it truncates to, so a filename at `filenameMaxLength` stays at it, and that a limit shorter than the extension does not reach `Blob.slice` with a negative start. |
40
42
  | `filename-characters.js` | That `getValidFilename` maps a full-width lookalike one character at a time — `C++` used to be saved as `C+` — while a run of characters with no lookalike still collapses to a single replacement. |