single-file-core 1.5.124 → 1.5.126

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/core/helper.js CHANGED
@@ -293,6 +293,9 @@ function preProcessDoc(doc, win, options) {
293
293
  }
294
294
 
295
295
  function markInvalidNesting(doc) {
296
+ if (!doc.body) {
297
+ return;
298
+ }
296
299
  addTrackIds(doc.body);
297
300
  const verificationDoc = parseDocContent(serialize(doc));
298
301
  const markedMap = buildTrackIdMap(doc.body);
package/core/index.js CHANGED
@@ -1361,6 +1361,11 @@ class Processor {
1361
1361
  }
1362
1362
  }
1363
1363
  }));
1364
+ frameElements.forEach(frameElement => {
1365
+ if (!frameElement.getAttribute("src") && !frameElement.getAttribute("srcdoc") && !frameElement.getAttribute("data")) {
1366
+ frameElement.removeAttribute("sandbox");
1367
+ }
1368
+ });
1364
1369
 
1365
1370
  async function initializeProcessor(frameData, frameElement, frameWindowId, batchRequest, options) {
1366
1371
  options.insertSingleFileComment = false;
@@ -38,9 +38,6 @@ function getDoctypeString(doc) {
38
38
  } else if (docType.systemId) {
39
39
  docTypeString += " SYSTEM \"" + docType.systemId + "\"";
40
40
  }
41
- if (docType.internalSubset) {
42
- docTypeString += " [" + docType.internalSubset + "]";
43
- }
44
41
  docTypeString += ">";
45
42
  }
46
43
  return docTypeString;
@@ -243,9 +243,16 @@ class ProcessorHelperCommon {
243
243
  return serializeSrcset([Object.assign({}, srcsetValue, { url: resourceURL })]);
244
244
  }
245
245
  }));
246
- resourceElement.setAttribute("srcset", srcsetValues.filter(srcsetValue => srcsetValue).join(", "));
246
+ const newSrcset = srcsetValues.filter(srcsetValue => srcsetValue).join(", ");
247
+ if (newSrcset) {
248
+ resourceElement.setAttribute("srcset", newSrcset);
249
+ } else {
250
+ resourceElement.removeAttribute("srcset");
251
+ resourceElement.removeAttribute("sizes");
252
+ }
247
253
  } else {
248
- resourceElement.setAttribute("srcset", "");
254
+ resourceElement.removeAttribute("srcset");
255
+ resourceElement.removeAttribute("sizes");
249
256
  }
250
257
  }));
251
258
  }
@@ -267,6 +274,7 @@ class ProcessorHelperCommon {
267
274
  element.style.setProperty("background-size", style && style["background-size"] ? style["background-size"] : "100% 100%", "important");
268
275
  element.style.setProperty("background-origin", "content-box", "important");
269
276
  element.style.setProperty("background-repeat", "no-repeat", "important");
277
+ element.style.setProperty("background-attachment", "scroll", "important");
270
278
  }
271
279
 
272
280
  async getStylesheetContent(resourceURL, options) {
@@ -405,7 +413,7 @@ class ProcessorHelperCommon {
405
413
  }
406
414
  sheetIndex++;
407
415
  });
408
- processFontDetails(fontsDetails);
416
+ processFontDetails(fontsDetails, fonts);
409
417
  await Promise.all([...stylesheets].map(async ([, stylesheetInfo], sheetIndex) => {
410
418
  if (stylesheetInfo.stylesheet) {
411
419
  const cssRules = stylesheetInfo.stylesheet.children;
@@ -443,13 +451,10 @@ class ProcessorHelperCommon {
443
451
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
444
452
  const key = this.getFontKey(ruleData);
445
453
  const fontInfo = fontsDetails.fonts.get(key);
446
- if (fontInfo) {
447
- const ruleKey = key + " " + this.getPropertyValue(ruleData, "src");
448
- if (fontsDetails.emittedFonts.has(ruleKey)) {
454
+ if (fontInfo && fontsDetails.lastRules.get(key) == ruleData) {
455
+ const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
456
+ if (!keptRule) {
449
457
  removedRules.push(cssRule);
450
- } else {
451
- fontsDetails.emittedFonts.add(ruleKey);
452
- await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
453
458
  }
454
459
  } else {
455
460
  removedRules.push(cssRule);
@@ -487,16 +492,22 @@ class ProcessorHelperCommon {
487
492
  fontInfo = [];
488
493
  mediaFontsDetails.fonts.set(fontKey, fontInfo);
489
494
  }
495
+ mediaFontsDetails.lastRules.set(fontKey, ruleData);
490
496
  const src = this.getPropertyValue(ruleData, "src");
491
497
  if (src) {
492
498
  const fontSources = src.match(REGEXP_URL_FUNCTION);
493
499
  if (fontSources) {
500
+ const ruleSources = [];
494
501
  fontSources.forEach(source => {
495
502
  if (fontInfo.includes(source)) {
496
503
  fontInfo.splice(fontInfo.indexOf(source), 1);
497
504
  }
498
- fontInfo.unshift(source);
505
+ if (ruleSources.includes(source)) {
506
+ ruleSources.splice(ruleSources.indexOf(source), 1);
507
+ }
508
+ ruleSources.unshift(source);
499
509
  });
510
+ ruleSources.forEach(source => fontInfo.push(source));
500
511
  }
501
512
  }
502
513
  }
@@ -509,7 +520,7 @@ class ProcessorHelperCommon {
509
520
  medias: new Map(),
510
521
  supports: new Map(),
511
522
  layers: new Map(),
512
- emittedFonts: new Set()
523
+ lastRules: new Map()
513
524
  };
514
525
  }
515
526
 
@@ -648,6 +648,9 @@ function getProcessorHelperClass(utilInstance) {
648
648
  }
649
649
  }
650
650
  stats.fonts.discarded -= fontInfo.length;
651
+ if (!fontInfo.length) {
652
+ return false;
653
+ }
651
654
  fontInfo.reverse();
652
655
  try {
653
656
  srcDeclaration.data.value = cssTree.parse(fontInfo.map(fontSource => fontSource.src).join(","), { context: "value", parseCustomProperty: true });
@@ -656,6 +659,7 @@ function getProcessorHelperClass(utilInstance) {
656
659
  // ignored
657
660
  }
658
661
  }
662
+ return true;
659
663
  }
660
664
  };
661
665
  }
@@ -574,6 +574,9 @@ function getProcessorHelperClass(utilInstance) {
574
574
  removedNodes.forEach(node => ruleData.block.children.remove(node));
575
575
  const srcDeclaration = ruleData.block.children.filter(node => node.property == "src").tail;
576
576
  if (srcDeclaration) {
577
+ if (!fontInfo.length) {
578
+ return false;
579
+ }
577
580
  fontInfo.reverse();
578
581
  try {
579
582
  srcDeclaration.data.value = cssTree.parse(fontInfo.map(fontSource => fontSource.src).join(","), { context: "value", parseCustomProperty: true });
@@ -582,6 +585,7 @@ function getProcessorHelperClass(utilInstance) {
582
585
  // ignored
583
586
  }
584
587
  }
588
+ return true;
585
589
  }
586
590
  };
587
591
  }
package/core/util.js CHANGED
@@ -69,11 +69,6 @@ const EXPECTED_TYPES_MEDIA = ["font", "image", "video", "audio"];
69
69
  const URL = globalThis.URL;
70
70
  const DOMParser = globalThis.DOMParser;
71
71
  const Blob = globalThis.Blob;
72
- const fetch = (url, options) => {
73
- options.cache = "force-cache";
74
- options.referrerPolicy = "strict-origin-when-cross-origin";
75
- return globalThis.fetch(url, options);
76
- };
77
72
  const TextDecoder = globalThis.TextDecoder;
78
73
  const URLSearchParams = globalThis.URLSearchParams;
79
74
 
@@ -83,8 +78,8 @@ export {
83
78
 
84
79
  function getInstance(utilOptions) {
85
80
  utilOptions = utilOptions || {};
86
- utilOptions.fetch = utilOptions.fetch || fetch;
87
- utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch || fetch;
81
+ utilOptions.fetch = utilOptions.fetch || ((url, options) => globalThis.fetch(url, { ...options, cache: "force-cache", referrerPolicy: "strict-origin-when-cross-origin" }));
82
+ utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch;
88
83
  return {
89
84
  getDoctypeString,
90
85
  getFilenameExtension(resourceURL, replacedCharacters, replacementCharacter, replacementCharacters) {
@@ -240,7 +235,7 @@ function getInstance(utilOptions) {
240
235
  startTime = Date.now();
241
236
  log(" // STARTED download url =", resourceURL, "asBinary =", options.asBinary);
242
237
  }
243
- if (options.blockMixedContent && /^https:/i.test(options.baseURI) && !/^https:/i.test(resourceURL)) {
238
+ if (options.blockMixedContent && /^https:/i.test(options.baseURI) && !/^https:/i.test(resourceURL) && !/^blob:https:/i.test(resourceURL)) {
244
239
  return getFetchResponse(resourceURL, options);
245
240
  }
246
241
  if (options.networkTimeout) {
@@ -264,7 +259,7 @@ function getInstance(utilOptions) {
264
259
  // eslint-disable-next-line no-unused-vars
265
260
  } catch (error) {
266
261
  response = await Promise.race([
267
- fetchResource(resourceURL, { headers: { accept } }),
262
+ fetchResource(resourceURL, { referrer: options.resourceReferrer, headers: { accept } }),
268
263
  networkTimeoutPromise
269
264
  ]);
270
265
  }
@@ -390,13 +385,13 @@ function guessMIMEType(expectedType, buffer) {
390
385
  if (compareBytes([255, 255, 255, 255], [0, 0, 2, 0])) {
391
386
  return "image/x-icon";
392
387
  }
393
- if (compareBytes([255, 255], [78, 77])) {
388
+ if (compareBytes([255, 255], [66, 77])) {
394
389
  return "image/bmp";
395
390
  }
396
391
  if (compareBytes([255, 255, 255, 255, 255, 255], [71, 73, 70, 56, 57, 97])) {
397
392
  return "image/gif";
398
393
  }
399
- if (compareBytes([255, 255, 255, 255, 255, 255], [71, 73, 70, 56, 59, 97])) {
394
+ if (compareBytes([255, 255, 255, 255, 255, 255], [71, 73, 70, 56, 55, 97])) {
400
395
  return "image/gif";
401
396
  }
402
397
  if (compareBytes([255, 255, 255, 255, 0, 0, 0, 0, 255, 255, 255, 255, 255, 255], [82, 73, 70, 70, 0, 0, 0, 0, 87, 69, 66, 80, 86, 80])) {
@@ -434,7 +429,7 @@ function guessMIMEType(expectedType, buffer) {
434
429
  if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 105, 115, 111, 109])) {
435
430
  return "video/mp4";
436
431
  }
437
- if (compareBytes([255, 255, 255, 255, 0, 0, 0, 0, 255, 255, 255, 255], [82, 73, 70, 70, 0, 0, 0, 0, 87, 65, 86, 69])) {
432
+ if (compareBytes([255, 255, 255, 255, 0, 0, 0, 0, 255, 255, 255, 255], [82, 73, 70, 70, 0, 0, 0, 0, 65, 86, 73, 32])) {
438
433
  return "video/x-msvideo";
439
434
  }
440
435
  if (compareBytes([255, 255, 255, 255], [0, 0, 1, 179]) || compareBytes([255, 255, 255, 255], [0, 0, 1, 186])) {
@@ -443,18 +438,18 @@ function guessMIMEType(expectedType, buffer) {
443
438
  if (compareBytes([255, 255, 255, 255], [79, 103, 103, 83])) {
444
439
  return "video/ogg";
445
440
  }
446
- if (compareBytes([255], [71])) {
447
- return "video/mp2t";
448
- }
449
441
  if (compareBytes([255, 255, 255, 255], [26, 69, 223, 163])) {
450
442
  return "video/webm";
451
443
  }
452
444
  if (compareBytes([0, 0, 0, 0, 255, 255, 255, 255, 255, 255], [0, 0, 0, 0, 102, 116, 121, 112, 51, 103])) {
453
445
  return "video/3gpp";
454
446
  }
447
+ if (compareBytes([255], [71])) {
448
+ return "video/mp2t";
449
+ }
455
450
  }
456
451
  if (expectedType == "audio") {
457
- if (compareBytes([255, 255], [255, 249]) || compareBytes([255, 255], [255, 254])) {
452
+ if (compareBytes([255, 255], [255, 241]) || compareBytes([255, 255], [255, 249]) || compareBytes([255, 255], [255, 254])) {
458
453
  return "audio/aac";
459
454
  }
460
455
  if (compareBytes([255, 255, 255, 255], [77, 84, 104, 100])) {
@@ -136,7 +136,7 @@ Three consequences shape everything below:
136
136
  | **region** | A byte range with a single producer, named in §3. Regions are the units the rest of this document reasons about; a region can appear in several pieces — `html-prologue` resumes after the embedded PDF document in the PDF variants, and after the `tEXt "ZIP"` chunk header in the PNG ones, so with all four faces it comes in three. |
137
137
  | **universal mode** | The variant whose HTML face can extract the archive from the *parsed page text*, the text and comment nodes the HTML parser produced, and therefore needs no access to its own raw bytes. Named "universal" because it works from any location, including the `file:` protocol. |
138
138
  | **wrapper tag** | The HTML construct that hides a binary region from the HTML parser, `<!--`…`-->` by default (§5.1). |
139
- | **appended data** | Bytes after the ZIP End Of Central Directory record. Readers tolerate them within the window their EOCD scan already covers: 65557 bytes from the end of the file (the 22-byte record plus the 65535-byte maximum comment length); "the 64 KB window" refers to this. It may be left undeclared or declared as the archive comment; both forms are valid ZIP and readers MUST accept both (§4.2). The recovery payload can be computed before that choice is made because it stops two bytes short of the record, excluding its comment-length field (see *recovered range* below). |
139
+ | **appended data** | Bytes after the ZIP End Of Central Directory record. A reader tolerates them as far back as its EOCD scan reaches, and how far that is varies by an order of magnitude: 65557 bytes from the end of the file for Python `zipfile` (the 22-byte record plus the 65535-byte maximum comment length), but 16383 for libarchive and 32768 for perl `Archive::Zip` (§8.1). No reader's window is guaranteed, so a writer keeps its own narrower budget (§5.2). The format's one hard limit is the 65535-byte comment field, and it binds only a run the writer declares (§4.2). It may be left undeclared or declared as the archive comment; both forms are valid ZIP and readers MUST accept both (§4.2). The recovery payload can be computed before that choice is made because it stops two bytes short of the record, excluding its comment-length field (see *recovered range* below). |
140
140
  | **ZIP region** | The contiguous byte range holding the archive proper: from the first local file header the ZIP writer emitted through the last byte of the End Of Central Directory record. It spans the `zip-entries`, `pdf-central-record` (when present) and `central-directory · eocd` blocks of §3, and in the HTML variants it is the content of the last wrapper, exactly so on the element rungs and preceded by the `sfz-data` identifier on the comment rung, which the extractor steps over. It does **not** include `pdf-local-header` or the PDF document, which sit earlier in the file. |
141
141
  | **archive** | The *logical* ZIP file: the set of entries the central directory describes, wherever their bytes lie. This is distinct from the ZIP region above, which is a contiguous byte range. Every entry but one has its bytes inside the region; `page.pdf` is the deliberate exception, an entry of the archive whose local header and data sit before the region (§4.2). "Archive" in this document always means the logical file, "ZIP region" always the byte range, and the two differ only in the PDF-with-HTML variants. |
142
142
  | **recovered range** | What the universal extractor reproduces (§4.5): the ZIP region minus its last two bytes, the comment-length field of the End Of Central Directory record. That field is the one part of the record whose value depends on what follows the region, so leaving it out is what lets a writer decide the appended-data form after the recovery payload is final (§4.2). The extractor supplies the two bytes itself, as zeroes — the recovered range carries no comment. |
@@ -174,8 +174,9 @@ every acquisition path including the ones that read raw bytes and could return i
174
174
  HTML face, so the option does not apply. The *Specimen* column names the measured
175
175
  reference files this document cites; §8 records how to regenerate them.
176
176
 
177
- Other writer options shape the file without adding a face: `preventAppendedData` and
178
- `declareAppendedData` (§4.2, §5.2), `includeBOM` (§3.1), `insertTextBody` (§4.6),
177
+ Other writer options shape the file without adding a face: `preventAppendedData`,
178
+ `declareAppendedData` and `maxAppendedDataLength` (§4.2, §5.2), `includeBOM` (§3.1),
179
+ `insertTextBody` (§4.6),
179
180
  `password` (§5.6), `createRootDirectory` (§7.1), and the head-element switches
180
181
  `insertCanonicalLink`, `insertMetaNoIndex` and `insertMetaCSP` (§3.1).
181
182
 
@@ -328,7 +329,8 @@ the same way: only a universal file carries an `<sfz-extra-data>` element.
328
329
 
329
330
  Unless a row states otherwise, the layouts below are measured from specimen files
330
331
  saved from `example.com` (the generation commands are in §8). The relocated row covers
331
- two cases with one layout, `preventAppendedData` and a payload over 64 KB: the first is
332
+ two cases with one layout, `preventAppendedData` and a payload over the appended-data
333
+ budget (§5.2): the first is
332
334
  measured on the relocated specimen, the second derived from the writer rules, because
333
335
  such a payload requires an archive too large for a readable specimen. The figure below shows
334
336
  the regions and their order; the glossary of §3.1 is the normative list, and it states
@@ -353,7 +355,7 @@ face adds, then the regions the PNG face adds.
353
355
  | `<!--` / `-->` | HTML | HTML face | The wrapper tag pair hiding a binary region from the HTML parser — comment tags by default, another pair when the hidden bytes defeat them — which `-->` is only the commonest way to do, the full test being `<!--`, `--!>`, a trailing `<!-` and, for the PNG payload, a leading `>` or `->` (§5.1). Drawn at each opening and closing position. The close tag is absent whenever the recovery payload is relocated (§5.2): under `preventAppendedData`, when the payload outgrows the appended-data budget, or on the `<plaintext>` wrapper which cannot close. No markup then follows the archive and the wrapper runs to end-of-file. That does not mean the file ends at the EOCD — the PNG face's tail still follows, inside the wrapper, where it parses as text (§5.1). |
354
356
  | `zip-entries` | ZIP | always | The archive's local file headers and entry data, written by the ZIP writer. The central directory of an archive written by the reference writer lists `index.html` (the page) first, then `manifest.json` (a JSON description of the archive: original URL, title, save time, resource-to-URL map — informative; the page displays without it), then the resources; the *physical* order of the local headers inside the region is not guaranteed to match, and readers MUST NOT rely on either order — entries are addressed by name (§7.1). |
355
357
  | `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
356
- | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the 64 KB appended-data window or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
358
+ | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the appended-data budget (§5.2) or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
357
359
  | `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
358
360
  | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag set as on every other entry — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
359
361
  | `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
@@ -366,7 +368,7 @@ face adds, then the regions the PNG face adds.
366
368
  | `crc · IEND` | PNG | PNG face | The `tEXt "ZIP"` chunk's CRC, computed once the archive bytes are final (§6), followed by the empty `IEND` chunk — the last bytes of the file (PNG requires `IEND` to end the stream, which is why the PNG variants drop the end tags). |
367
369
 
368
370
  The reader-by-reader interpretation of these regions is §4; the mechanics that keep
369
- them from colliding (wrapper-tag selection, checksums, offsets, the 64 KB budget) are
371
+ them from colliding (wrapper-tag selection, checksums, offsets, the appended-data budget) are
370
372
  §5.
371
373
 
372
374
  ## 4. Reader lenses
@@ -447,10 +449,12 @@ path the plain variant's error message describes.
447
449
  ### 4.2 The ZIP reader
448
450
 
449
451
  The ZIP face is read from the end. A reader locates the End Of Central Directory
450
- record by scanning backward from end-of-file; the format guarantees it lies within
451
- the window every reader must already scan to support archive comments (§1.3,
452
- *appended data*), because everything after it — wrapper close tag, extra-data, end
453
- tags, PNG tail — fits the appended-data budget (§5.2). Accepting *undeclared* bytes
452
+ record by scanning backward from end-of-file, and how far back it scans is the one
453
+ reader property the format cannot assume (§1.3, *appended data*). Everything the
454
+ writer emits after the record — wrapper close tag, extra-data, end tags, PNG tail —
455
+ fits the appended-data budget of §5.2, and the reference writer sizes that budget to
456
+ the narrowest scan measured in §8.1, so the record stays reachable for every reader
457
+ listed there. Accepting *undeclared* bytes
454
458
  in that window is itself a customary tolerance (§1.1): the ZIP specification
455
459
  documents the comment, not trailing junk. From the EOCD
456
460
  the reader jumps to the central directory and reads only what it references;
@@ -510,6 +514,14 @@ readers accept at all: `java.util.zip`, and therefore Android and most JVM tooli
510
514
  rejects an archive with undeclared trailing bytes outright (§8.1). A writer SHOULD
511
515
  offer both and default to raw.
512
516
 
517
+ The declared form carries a ceiling the raw form does not. The comment length is a
518
+ 16-bit field, so a run longer than 65535 bytes cannot be declared at all. A writer
519
+ whose appended-data budget (§5.2) is raised past that ceiling MUST leave such a run
520
+ undeclared rather than write its length back modulo 65536, and readers that accept
521
+ only the declared form then reject the file with no diagnostic. The budget and the
522
+ ceiling are two separate limits, and a writer that exposes the first as an option
523
+ SHOULD say so where it documents it.
524
+
513
525
  Neither form constrains the other faces, and universal mode supports both, because
514
526
  the recovery payload describes the recovered range rather than the whole region: the
515
527
  comment-length field is excluded (§1.3), so its value can be decided after the
@@ -735,6 +747,23 @@ zip64 locator states an absolute offset in the original file, so it needs the sh
735
747
  this formula produces and cannot be used to find it. A reader that instead uses the
736
748
  sentinels arithmetically gets a shift in the billions, with no diagnostic.
737
749
 
750
+ Which leaves the record itself to be located without the offset that normally points at
751
+ it. Scan backward from the locator for the `PK\x06\x06` signature and confirm each
752
+ candidate against the record's own size field, the 8 bytes at `p + 4`, which by
753
+ definition excludes the leading 12:
754
+
755
+ ```
756
+ p + 12 + size == locatorPosition
757
+ ```
758
+
759
+ A well-formed archive puts the record immediately before the locator, where it is 56
760
+ bytes long if it carries no extensible data sector, so `locatorPosition - 56` is worth
761
+ testing before scanning at all. The confirmation matters on the archives that miss:
762
+ past the record the scan walks back through the central directory, whose file names and
763
+ extra fields are arbitrary bytes, and past that through entry data, and a four-byte
764
+ signature turns up in bytes nothing constrains. The test settles each candidate against
765
+ the record's own field, so it needs no offset it does not already have.
766
+
738
767
  ### 4.6 Text tools
739
768
 
740
769
  The optional text body (`insertTextBody`) addresses one more consumer: software that
@@ -972,36 +1001,54 @@ limit of what the CDATA rung buys here — its value is that real payloads rarel
972
1001
  above states a MUST rather than a quality-of-implementation preference — exhaustion is
973
1002
  reachable by construction, not only by a payload built to provoke it.
974
1003
 
975
- The check MUST be made against the payload's final bytes, for every payload the
976
- writer hides and in every variant that hides one. Every rejection restarts the build
977
- (§6): the wrapper choice changes the bytes preceding the archive, so the archive must
978
- be rewritten at its new position.
979
-
980
- Two fields are patched after that check. The EOCD comment-length field sits at the
981
- end of the ZIP region and is patched under the declared form (§6.1, step 11); the
982
- writer tests the bytes around it again with the final value in place and keeps the raw
983
- form when that value would complete a pattern, since the raw form is always valid. The
984
- `tEXt "ZIP"` length field sits inside the pixel-data wrapper, with the fixed `tEXt`
985
- type and `ZIP` keyword after it, and is written last (step 12). The header is tested
986
- with the rest of the payload, the length as zeros, which cannot join a pattern; the
987
- real length is big-endian, so a pattern byte in it would have to be the most
988
- significant byte of the chunk's size, and the smallest byte any pattern contains, `-`
989
- at 0x2D, puts that size at 0x2D000000 bytes, about 755 MB. The writer refuses to
990
- build a self-extracting PNG variant whose chunk reaches that size rather than
1004
+ The selection test above — both patterns, on every rung the writer considers — MUST
1005
+ be applied to the payload's bytes in their final form, for every payload the writer
1006
+ hides and in every variant that hides one. *Final* is the whole of the requirement:
1007
+ bytes the writer has yet to settle have not been tested. Every rejection restarts the build (§6):
1008
+ the wrapper choice changes the bytes preceding the archive, so the archive must be
1009
+ rewritten at its new position.
1010
+
1011
+ Two fields are patched after that test, and each needs one of its own. The EOCD
1012
+ comment-length field sits at the end of the ZIP region and is patched under the
1013
+ declared form (§6.1, step 11); the writer tests the bytes around it again with the
1014
+ final value in place and keeps the raw form when that value would complete a pattern,
1015
+ since the raw form is always valid. The `tEXt "ZIP"` length field sits inside the pixel-data wrapper,
1016
+ with the fixed `tEXt` type and `ZIP` keyword after it, and is written last (step 12).
1017
+ The header is tested with the rest of the payload, the length as zeros, which cannot
1018
+ join a pattern; the real length is big-endian, so a pattern byte in it would have to be
1019
+ the most significant byte of the chunk's size, and the smallest byte any pattern
1020
+ contains, `-` at 0x2D, puts that size at 0x2D000000 bytes, about 755 MB. The writer
1021
+ refuses to build a self-extracting PNG variant whose chunk reaches that size rather than
991
1022
  re-check the field.
992
1023
 
993
1024
  ### 5.2 The appended-data budget
994
1025
 
995
- Everything the writer emits after the EOCD record MUST fit in 65535 bytes — the
996
- maximum length a ZIP archive comment may declare, and therefore the distance beyond
997
- the record that every reader's backward scan already covers (§1.3). The appended run is:
1026
+ The run the writer emits after the EOCD record has two limits, and only one of them
1027
+ comes from the format. A run *declared* as the archive comment MUST fit in 65535
1028
+ bytes, the largest value a comment-length field can hold (§4.2). A run left *raw* has
1029
+ no format limit at all: the bytes are outside the archive, and nothing in ZIP bounds
1030
+ them. What bounds both in practice is the reader. Locating the EOCD record means
1031
+ scanning backward from end-of-file, and the searches measured in §8.1 stop at 16383
1032
+ bytes for libarchive, 32768 for perl `Archive::Zip` and 65557 for Python `zipfile`, so
1033
+ a run sized to the comment ceiling is already invisible to the narrowest of them. A
1034
+ writer therefore keeps a *budget*, sized to the readers it means to satisfy rather
1035
+ than to the format. The appended run is:
998
1036
 
999
1037
  ```
1000
1038
  wrapper close tag + extra-data element + end tags + (PNG face: 4-byte chunk CRC + 12-byte IEND)
1001
1039
  ```
1002
1040
 
1003
- and the writer compares its total against 65535 before committing to it. The EOCD
1004
- record's own 22 bytes sit inside the window too, giving the 65557-byte figure of §1.3.
1041
+ and the writer compares its total against that budget before committing to it. The
1042
+ EOCD record's own 22 bytes sit inside a reader's window as well, which is what turns a
1043
+ 65535-byte run into the 65557 bytes of §1.3 and the reference writer's budget into
1044
+ libarchive's 16383.
1045
+
1046
+ The reference writer exposes the budget as `maxAppendedDataLength` and defaults it to
1047
+ 16361 bytes: libarchive's window less the 22 bytes of the record, which is the largest
1048
+ run behind which every reader of §8.1 still finds the record. A writer MAY choose
1049
+ another value. Raising it above 65535 leaves the run undeclarable: it is emitted, and
1050
+ it is still valid ZIP, but no comment length can cover it, and §4.2 says what that
1051
+ costs.
1005
1052
 
1006
1053
  Only the extra-data element can outgrow the budget: it carries one 2-bit code per
1007
1054
  newline sequence in the recovered range — the ZIP region without its comment-length
@@ -1009,9 +1056,9 @@ field (§4.5), CR LF counting once, for two bytes (§5.5) — so it
1009
1056
  grows with the archive. Newline bytes
1010
1057
  occur at their natural density in compressed and STOREd binary data — about two in
1011
1058
  every 256 bytes — and the codes are compressed and base64-encoded, which measures at
1012
- one byte of element per 650 bytes of archive at scale (§8). The budget is therefore
1013
- exhausted at an archive of roughly 40 MB, so the relocated placement is rare in
1014
- practice. That ratio is the large-archive limit and must not be used to size a
1059
+ one byte of element per 650 bytes of archive at scale (§8). The default budget is
1060
+ therefore exhausted at an archive of roughly 10 MB, and the 65535-byte ceiling at
1061
+ roughly 40 MB, so the relocated placement is uncommon in practice. That ratio is the large-archive limit and must not be used to size a
1015
1062
  particular file: deflate's overhead is a fixed cost spread over a growing payload, so
1016
1063
  small archives are far less efficient. Measured on exact byte counts, a 6099-byte region
1017
1064
  needs 69 bytes of element — a ratio of 88 — and a 74057-byte region needs 189, a ratio
@@ -1032,6 +1079,14 @@ trailing bytes open it (§8.1). The parser closes the open
1032
1079
  comment or element at end of file, and `</body></html>` are implied, so the page
1033
1080
  renders the same.
1034
1081
 
1082
+ Relocation is not a move at constant size, and it can end either way. Two effects pull
1083
+ against each other: the room the writer sets aside, which in the reference writer is
1084
+ `Math.ceil(length * 1.01) + 32` bytes — a percentage of the payload plus a constant, so
1085
+ the constant dominates a small payload and the percentage a large one — and the 17 bytes
1086
+ of wrapper terminator and end tags it stops emitting. Measured on three files the net ran
1087
+ from 9 bytes saved to 190 bytes spent, so a writer sizing a file should quote that range
1088
+ rather than a single figure.
1089
+
1035
1090
  ### 5.3 Offset bookkeeping
1036
1091
 
1037
1092
  Three coordinate systems coexist in one file, and the format's job is to keep each
@@ -1043,6 +1098,15 @@ self-consistent:
1043
1098
  file positions (§4.2), so a reader of the *whole file* never needs prepended-data
1044
1099
  compensation — the repair by which a reader recomputes offsets that disagree with
1045
1100
  the file size. A reader of the recovered ZIP region alone does need it (§4.5).
1101
+
1102
+ The alternative, offsets relative to the start of the region, is not a compatibility
1103
+ problem in itself: a reader that compensates arrives at the same entries, and 7-Zip
1104
+ opens such a file when told the type. What absolute offsets buy is the step before
1105
+ that. The file is a valid archive read as it stands, so it survives format
1106
+ auto-detection — 7-Zip reports a base of 0 and a physical size covering the whole
1107
+ file — and the compensation is confined to the one path that cannot avoid it,
1108
+ universal-mode recovery. Nothing in the format depends on the choice; a writer using
1109
+ the other form produces files this document's readers still open.
1046
1110
  - **PDF offsets are header-relative.** The document's own cross-reference offsets are
1047
1111
  interpreted from the `%PDF-` header, so embedding it needs no rewriting; the writer
1048
1112
  only MUST keep the header inside the scan window (§4.3).
@@ -1439,7 +1503,7 @@ terminate because each of them advances a monotone quantity:
1439
1503
  There is no converse of the second: a pass that reserved room never discards it,
1440
1504
  even when the relocated payload would have fit the appended window. Relocation moves
1441
1505
  the archive, which changes the offsets, which changes the payload that made the
1442
- relocation necessary, so a payload lying on the 65535-byte boundary can be too large
1506
+ relocation necessary, so a payload lying on the budget boundary can be too large
1443
1507
  appended and small enough relocated, and a writer that dropped the reservation could
1444
1508
  rebuild the two placements forever. Relocation is therefore final (§5.2), and the file
1445
1509
  keeps at most the reservation's own margin of dead padding.
@@ -1594,7 +1658,7 @@ only if it affects the bytes the page is built from:
1594
1658
  | An entry's CRC-32 or AES authentication code does not match | **SHOULD** fail for that entry, and MUST NOT present a page rebuilt from it as intact |
1595
1659
  | `page.pdf` was reconstructed from the parsed page and its CRC-32 does not match | **MUST** discard the reconstruction (§4.5). The bytes are a guess about newlines the recovery payload does not describe, and the checksum is the only thing that tests it — unlike the row above, there is no read to have gone wrong, only an inference |
1596
1660
  | Bytes outside the archive proper — before the first local file header, after the EOCD record, or between an entry's data and the next header | **MUST** tolerate: they are the other faces (§7.1). The gap in the middle is not hypothetical: with the PDF face the bootstrap lies between `page.pdf`'s data and the ZIP region |
1597
- | The appended run exceeds the 65535-byte budget (§5.2) | Not a reader's problem: if the EOCD record was found, the archive is readable. Readers MAY warn |
1661
+ | The appended run exceeds the 65535-byte ceiling (§5.2) | Not a reader's problem: if the EOCD record was found, the archive is readable. Readers MAY warn |
1598
1662
  | A `tEXt` chunk CRC does not match, or a chunk holds bytes PNG does not permit (§4.4) | Irrelevant to extraction; a reader of the archive MAY ignore both |
1599
1663
  | `page.pdf` is present but its data does not begin with `%PDF-` | Not an error. The entry is data like any other |
1600
1664
  | `index.html` is present without `manifest.json` | **MUST** still extract (§7.1) |
@@ -1634,11 +1698,20 @@ and is class C.
1634
1698
  | Info-ZIP `unzip`, `zipinfo` | ✔ | ✔ | ✔ | Lists and extracts every variant. AES entries are skipped — `need PK compat. v5.1 (can do v4.5)` — a limitation of the tool, not of the file; `page.pdf` still extracts because it is never encrypted |
1635
1699
  | Python `zipfile` | ✔ | ✔ | ✔ | Lists and extracts every variant |
1636
1700
  | 7-Zip (`7zz`) | ✔ | ✔ | ✔ | Lists and extracts every variant, AES included |
1637
- | libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant |
1701
+ | libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant. Its EOCD scan is the narrowest measured, so a class-C file whose appended run pushes the record past 16383 bytes from the end is rejected with `Unrecognized archive format`; §5.2's default budget is sized to this window, and the ✔ holds for files that respect it |
1638
1702
  | libarchive `bsdtar`, piped input | ✔ | ✘ | ✘ | `Unrecognized archive format` — the forward-only case of §1.2, measured |
1639
1703
  | Java `java.util.zip` (`jar tf`) | ✔ | ✔ | ✘ | `zip END header not found` whenever bytes follow the EOCD undeclared. Declaring them as the archive comment makes the same file open, measured on every class-C variant (§4.2) |
1640
1704
  | macOS `ditto -x -k` | ✔ | ✘ | ✘ | `Couldn't read PKZip signature` — requires a local file header at offset 0, so prepended data alone defeats it |
1641
1705
 
1706
+ The backward scans behind the class-C column differ by an order of magnitude, and they
1707
+ are what §5.2's budget is sized against. Measured by padding a working archive until
1708
+ the record fell out of reach, the largest distance from end-of-file at which each
1709
+ reader still finds the EOCD record is: libarchive 16383, perl `Archive::Zip` 32768,
1710
+ Python `zipfile` 65557, zip.js 65536, Info-ZIP `unzip` 68000, and 7-Zip beyond 1 MiB,
1711
+ which scans the whole file. libarchive binds, and its window less the 22-byte record
1712
+ is the 16361-byte default budget of §5.2. macOS `ditto` is not on this axis at all: it
1713
+ requires a local file header at offset 0 whatever the tail looks like.
1714
+
1642
1715
  The cost of the declared form was measured on the same tools: it is a display cost, not
1643
1716
  a compatibility one. An archive whose trailing bytes are declared as the comment has
1644
1717
  them printed back on ordinary listings; `unzip -l` reproduces the whole run — in
@@ -1713,7 +1786,7 @@ specimen without a network.
1713
1786
  | 123006 | `<sfz-extra-data>` … `</sfz-extra-data>` | recovery payload, appended placement (§5.2); 24 base64 characters for this archive |
1714
1787
  | 123063 | `</body></html>` | end tags; end of file at 123077 |
1715
1788
 
1716
- The appended run is 74 bytes, well inside the 65535-byte budget (§5.2). The ZIP region
1789
+ The appended run is 74 bytes, well inside the 16361-byte default budget (§5.2). The ZIP region
1717
1790
  is the 998 bytes from 122005 to 123003; the universal extractor reproduces the first 996
1718
1791
  of them and supplies the last two itself (§1.3).
1719
1792
 
@@ -1747,7 +1820,7 @@ These specimens are deliberately small, and a reader tested only against them is
1747
1820
  undertested: they are all flat archives of two or three entries. None
1748
1821
  exercises a root directory, `frames/<n>/` nesting, a second `index.html`, a `data:`-URL
1749
1822
  entry comment, the optional text body (§4.6), a UTF-8 BOM, zip64
1750
- (§5.7), a payload past the 64 KB budget, or a relocated reservation with padding left
1823
+ (§5.7), a payload past the appended-data budget, or a relocated reservation with padding left
1751
1824
  in it. Two omissions matter more than the rest, because they are the parts of §5.1 a
1752
1825
  writer is most likely to get wrong: no specimen defeats a rung by its **start**
1753
1826
  pattern, and none defeats one with an **upper-case** pattern. A writer that tested only
@@ -1766,7 +1839,10 @@ of §5.7.
1766
1839
 
1767
1840
  ### 8.4 The charset round trip, measured
1768
1841
 
1769
- Two claims of §2.1 were verified.
1842
+ Two claims of §2.1 were verified. The first is re-derived on every run by
1843
+ `test/sfz-harness/charset-round-trip.js`, which reads the tables below out of the
1844
+ runtime's own decoders rather than trusting this section, and checks the reverse table
1845
+ the extractor ships against the one the rule of §5.5 produces.
1770
1846
 
1771
1847
  **Which encodings qualify.** Decoding all 256 byte values through each encoding
1772
1848
  defined by the WHATWG standard shows 20 that are injective and never produce U+FFFD:
@@ -1816,6 +1892,7 @@ predicts.
1816
1892
  | August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
1817
1893
  | August 2026 | Core 1.5.119: the inlined ZIP library is built ASCII-only, and §2.1 now requires it of any bootstrap. Its CP437 table had been emitted as literal characters, which the page re-decoded as windows-1252, growing the table from 256 entries to 508 and shifting every lookup by 60 — so the one entry read without the UTF-8 flag, `page.pdf`, came back mangled and no archive with a PDF face extracted in any engine (§5.8) |
1818
1894
  | August 2026 | Core 1.5.120: the hand-built `page.pdf` records set the language encoding flag, like every entry the ZIP writer produces (§5.8). Its name is ASCII, so no decoded name changes; what changes is that no entry in an archive is read through CP437 any more, closing the path the 1.5.119 defect surfaced on |
1895
+ | September 2026 | Core 1.5.126: the appended-data budget becomes the `maxAppendedDataLength` writer option and its default drops from 65535 to 16361 bytes, so the EOCD record stays inside libarchive's scan and `bsdtar` opens archives it used to reject (§5.2, §8.1). The 65535-byte comment ceiling is now a separate limit, stated in §4.2: a budget raised past it produces a run that cannot be declared |
1819
1896
 
1820
1897
  This document was itself revised in August 2026, against core 1.5.108, after several
1821
1898
  independent reviews. One of them was a reader built from this specification alone, with
@@ -83,15 +83,15 @@ function getSourceSrcData(sources) {
83
83
  function setSrc(srcData, imgElement, pictureElement) {
84
84
  if (srcData.src) {
85
85
  imgElement.setAttribute("src", srcData.src);
86
- imgElement.setAttribute("srcset", "");
87
- imgElement.setAttribute("sizes", "");
86
+ imgElement.removeAttribute("srcset");
87
+ imgElement.removeAttribute("sizes");
88
88
  } else {
89
89
  imgElement.setAttribute("src", EMPTY_RESOURCE);
90
90
  if (srcData.srcset) {
91
91
  imgElement.setAttribute("srcset", srcData.srcset);
92
92
  } else {
93
- imgElement.setAttribute("srcset", "");
94
- imgElement.setAttribute("sizes", "");
93
+ imgElement.removeAttribute("srcset");
94
+ imgElement.removeAttribute("sizes");
95
95
  }
96
96
  }
97
97
  if (pictureElement) {
package/package.json CHANGED
@@ -1,11 +1,11 @@
1
1
  {
2
2
  "name": "single-file-core",
3
- "version": "1.5.124",
3
+ "version": "1.5.126",
4
4
  "description": "SingleFile Core",
5
5
  "author": "Gildas Lormeau",
6
6
  "license": "AGPL-3.0-or-later",
7
7
  "scripts": {
8
- "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js",
8
+ "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
9
9
  "bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
10
10
  "bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
11
11
  "bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
@@ -89,7 +89,8 @@ const PNG_CHUNK_CRC_LENGTH = 4;
89
89
  const PNG_SIGNATURE_LENGTH = 8;
90
90
  const PNG_IHDR_LENGTH = 25;
91
91
  const COMMENT_LENGTH_FIELD_LENGTH = 2;
92
- const MAX_APPENDED_DATA_LENGTH = 65535;
92
+ const MAX_ZIP_COMMENT_LENGTH = 65535;
93
+ const DEFAULT_MAX_APPENDED_DATA_LENGTH = 16361;
93
94
  const PDF_ENTRY_FILENAME = "page.pdf";
94
95
  const PRESCAN_WINDOW_LENGTH = 1024;
95
96
  const PNG_TEXT_CHUNK_HEADER_LENGTH = 12;
@@ -121,6 +122,7 @@ const PROCESS_OPTION_NAMES = [
121
122
  "insertMetaCSP",
122
123
  "insertMetaNoIndex",
123
124
  "insertTextBody",
125
+ "maxAppendedDataLength",
124
126
  "password",
125
127
  "preventAppendedData",
126
128
  "selfExtractingArchive",
@@ -278,7 +280,7 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
278
280
  const payloadView = new DataView(payload.buffer);
279
281
  words.forEach((word, indexWord) => payloadView.setUint32(indexWord * 4, word, true));
280
282
  extraData = "<sfz-extra-data>" + base64Encode(deflateRaw(payload)) + "</sfz-extra-data>";
281
- if (options.preventAppendedData || extraData.length > MAX_APPENDED_DATA_LENGTH - pageContent.length - endTags.length - (options.embeddedImage ? PNG_IEND_LENGTH + PNG_CHUNK_CRC_LENGTH : 0)) {
283
+ if (options.preventAppendedData || extraData.length > getMaxAppendedDataLength(options) - pageContent.length - endTags.length - (options.embeddedImage ? PNG_IEND_LENGTH + PNG_CHUNK_CRC_LENGTH : 0)) {
282
284
  if (!options.extraDataSize) {
283
285
  options.preventAppendedData = true;
284
286
  options.extraDataSize = getReservationSize(extraData.length);
@@ -304,7 +306,7 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
304
306
  if (options.declareAppendedData) {
305
307
  const appendedDataLength = pageContent.length - data.length +
306
308
  (options.embeddedImage ? PNG_CHUNK_CRC_LENGTH + PNG_IEND_LENGTH : 0);
307
- if (appendedDataLength && appendedDataLength <= MAX_APPENDED_DATA_LENGTH && isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options)) {
309
+ if (appendedDataLength && appendedDataLength <= MAX_ZIP_COMMENT_LENGTH && isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options)) {
308
310
  new DataView(pageContent.buffer, pageContent.byteOffset).setUint16(zipDataEnd, appendedDataLength, true);
309
311
  }
310
312
  }
@@ -324,6 +326,10 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
324
326
  }
325
327
  }
326
328
 
329
+ function getMaxAppendedDataLength(options) {
330
+ return options.maxAppendedDataLength === undefined ? DEFAULT_MAX_APPENDED_DATA_LENGTH : options.maxAppendedDataLength;
331
+ }
332
+
327
333
  function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options) {
328
334
  if (options.extractDataFromPageTags && options.extractDataFromPageTags[0] == "<plaintext>") {
329
335
  return true;
@@ -747,11 +753,7 @@ async function getContent() {
747
753
  });
748
754
  return new Promise((resolve, reject) => {
749
755
  let aborted = false;
750
- if (location.protocol == "file:") {
751
- extractDataFromDocument();
752
- } else {
753
- getPageData();
754
- }
756
+ getPageData();
755
757
 
756
758
  async function extractDataFromDocument() {
757
759
  try {
@@ -177,7 +177,7 @@ function initRequestSync(message) {
177
177
  if (!TOP_WINDOW) {
178
178
  windowId = globalThis.frameId = message.windowId;
179
179
  }
180
- processFrames(document, message.options, windowId, sessionId);
180
+ processFrames(document, message.options, windowId, sessionId, false);
181
181
  if (!TOP_WINDOW) {
182
182
  sendInitResponse({ frames: [getFrameData(document, globalThis, windowId, message.options, message.scrolling)], sessionId, requestedFrameId: document.documentElement.dataset.requestedFrameId && windowId });
183
183
  delete document.documentElement.dataset.requestedFrameId;
@@ -190,7 +190,7 @@ async function initRequestAsync(message) {
190
190
  if (!TOP_WINDOW) {
191
191
  windowId = globalThis.frameId = message.windowId;
192
192
  }
193
- processFrames(document, message.options, windowId, sessionId);
193
+ processFrames(document, message.options, windowId, sessionId, message.waitForFrames !== false);
194
194
  if (!TOP_WINDOW) {
195
195
  sendInitResponse({ frames: [getFrameData(document, globalThis, windowId, message.options, message.scrolling)], sessionId, requestedFrameId: document.documentElement.dataset.requestedFrameId && windowId });
196
196
  delete document.documentElement.dataset.requestedFrameId;
@@ -254,15 +254,15 @@ function initResponse(message) {
254
254
  }
255
255
  }
256
256
  }
257
- function processFrames(doc, options, parentWindowId, sessionId) {
257
+ function processFrames(doc, options, parentWindowId, sessionId, waitForFrames) {
258
258
  const frameElements = getFrames(doc);
259
- processFramesAsync(doc, frameElements, options, parentWindowId, sessionId);
259
+ processFramesAsync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames);
260
260
  if (frameElements.length) {
261
- processFramesSync(doc, frameElements, options, parentWindowId, sessionId);
261
+ processFramesSync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames);
262
262
  }
263
263
  }
264
264
 
265
- function processFramesAsync(doc, frameElements, options, parentWindowId, sessionId) {
265
+ function processFramesAsync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames) {
266
266
  const frames = [];
267
267
  let requestTimeouts;
268
268
  if (sessions.get(sessionId)) {
@@ -280,17 +280,18 @@ function processFramesAsync(doc, frameElements, options, parentWindowId, session
280
280
  frameElements.forEach((frameElement, frameIndex) => {
281
281
  const windowId = parentWindowId + WINDOW_ID_SEPARATOR + frameIndex;
282
282
  try {
283
- sendMessage(frameElement.contentWindow, { method: INIT_REQUEST_MESSAGE, windowId, sessionId, options, scrolling: frameElement.scrolling });
283
+ sendMessage(frameElement.contentWindow, { method: INIT_REQUEST_MESSAGE, windowId, sessionId, options, scrolling: frameElement.scrolling, waitForFrames });
284
284
  // eslint-disable-next-line no-unused-vars
285
285
  } catch (error) {
286
286
  // ignored
287
287
  }
288
- requestTimeouts[windowId] = globalThis.setTimeout(() => sendInitResponse({ frames: [{ windowId, processed: true }], sessionId }), TIMEOUT_INIT_REQUEST_MESSAGE);
288
+ setFrameFallback(sessionId, windowId, () => getSrcdocFrameData(frameElement, windowId, options, sessionId));
289
+ requestTimeouts[windowId] = globalThis.setTimeout(() => sendInitResponse({ frames: [getFrameFallback(sessionId, windowId) || { windowId, processed: true }], sessionId }), TIMEOUT_INIT_REQUEST_MESSAGE);
289
290
  });
290
291
  delete doc.documentElement.dataset.requestedFrameId;
291
292
  }
292
293
 
293
- function processFramesSync(doc, frameElements, options, parentWindowId, sessionId) {
294
+ function processFramesSync(doc, frameElements, options, parentWindowId, sessionId, waitForFrames) {
294
295
  const frames = [];
295
296
  frameElements.forEach((frameElement, frameIndex) => {
296
297
  const windowId = parentWindowId + WINDOW_ID_SEPARATOR + frameIndex;
@@ -303,21 +304,25 @@ function processFramesSync(doc, frameElements, options, parentWindowId, sessionI
303
304
  } catch (error) {
304
305
  // ignored
305
306
  }
306
- const srcdoc = frameElement.getAttribute("srcdoc");
307
- if (!frameDoc && srcdoc) {
308
- const doc = new DOMParser().parseFromString(srcdoc, "text/html");
309
- frameDoc = doc;
310
- frameWindow = globalThis;
311
- }
312
307
  if (frameDoc) {
313
308
  try {
314
309
  clearFrameTimeout("requestTimeouts", sessionId, windowId);
315
- processFrames(frameDoc, options, windowId, sessionId);
310
+ processFrames(frameDoc, options, windowId, sessionId, waitForFrames);
316
311
  frames.push(getFrameData(frameDoc, frameWindow, windowId, options, frameElement.scrolling));
317
312
  // eslint-disable-next-line no-unused-vars
318
313
  } catch (error) {
319
314
  frames.push({ windowId, processed: true });
320
315
  }
316
+ } else if (!waitForFrames) {
317
+ // the frame is cross-origin or sandboxed, so its document is out of reach. Re-parsing
318
+ // srcdoc is the only source left here, and it is markup only: no script has run and
319
+ // nothing is rendered. When there is time to wait, the fallback is kept for the frames
320
+ // that never answer instead, so a frame that does answer wins with its rendered data
321
+ const fallbackFrameData = getFrameFallback(sessionId, windowId);
322
+ if (fallbackFrameData) {
323
+ clearFrameTimeout("requestTimeouts", sessionId, windowId);
324
+ frames.push(fallbackFrameData);
325
+ }
321
326
  }
322
327
  });
323
328
  sendInitResponse({ frames, sessionId, requestedFrameId: doc.documentElement.dataset.requestedFrameId && parentWindowId });
@@ -338,7 +343,39 @@ function clearFrameTimeout(type, sessionId, windowId) {
338
343
  function createFrameResponseTimeout(sessionId, windowId) {
339
344
  const session = sessions.get(sessionId);
340
345
  if (session && session.responseTimeouts) {
341
- session.responseTimeouts[windowId] = globalThis.setTimeout(() => sendInitResponse({ frames: [{ windowId: windowId, processed: true }], sessionId: sessionId }), TIMEOUT_INIT_RESPONSE_MESSAGE);
346
+ session.responseTimeouts[windowId] = globalThis.setTimeout(() => sendInitResponse({ frames: [getFrameFallback(sessionId, windowId) || { windowId, processed: true }], sessionId: sessionId }), TIMEOUT_INIT_RESPONSE_MESSAGE);
347
+ }
348
+ }
349
+
350
+ function setFrameFallback(sessionId, windowId, getFallbackFrameData) {
351
+ const session = sessions.get(sessionId);
352
+ if (session) {
353
+ if (!session.frameFallbacks) {
354
+ session.frameFallbacks = {};
355
+ }
356
+ session.frameFallbacks[windowId] = getFallbackFrameData;
357
+ }
358
+ }
359
+
360
+ function getFrameFallback(sessionId, windowId) {
361
+ const session = sessions.get(sessionId);
362
+ const getFallbackFrameData = session && session.frameFallbacks && session.frameFallbacks[windowId];
363
+ if (getFallbackFrameData) {
364
+ return getFallbackFrameData();
365
+ }
366
+ }
367
+
368
+ function getSrcdocFrameData(frameElement, windowId, options, sessionId) {
369
+ const srcdoc = frameElement.getAttribute("srcdoc");
370
+ if (srcdoc) {
371
+ try {
372
+ const frameDoc = new DOMParser().parseFromString(srcdoc, "text/html");
373
+ processFrames(frameDoc, options, windowId, sessionId, false);
374
+ return getFrameData(frameDoc, globalThis, windowId, options, frameElement.scrolling);
375
+ // eslint-disable-next-line no-unused-vars
376
+ } catch (error) {
377
+ // ignored
378
+ }
342
379
  }
343
380
  }
344
381
 
@@ -242,39 +242,46 @@
242
242
  }
243
243
  return boundingRect;
244
244
  };
245
+ Element.prototype.getBoundingClientRect.toString = function () { return "function getBoundingClientRect() { [native code] }"; };
246
+ setFunctionName(Element.prototype.getBoundingClientRect, "getBoundingClientRect");
245
247
  }
246
248
  }
247
249
  if (!globalThis._singleFileImage) {
248
- const Image = globalThis.Image;
249
- globalThis._singleFileImage = globalThis.Image;
250
- globalThis.__defineGetter__("Image", function () {
251
- return function () {
252
- const image = new Image(...arguments);
253
- const result = new Image(...arguments);
254
- result.__defineSetter__("src", value => {
255
- image.src = value;
256
- document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT, { detail: image.src }));
257
- });
258
- result.__defineGetter__("src", () => image.src);
259
- result.__defineSetter__("srcset", value => {
260
- document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT));
261
- image.srcset = value;
262
- });
263
- result.__defineGetter__("srcset", () => image.srcset);
264
- result.__defineGetter__("height", () => image.height);
265
- result.__defineGetter__("width", () => image.width);
266
- result.__defineGetter__("naturalHeight", () => image.naturalHeight);
267
- result.__defineGetter__("naturalWidth", () => image.naturalWidth);
268
- if (image.decode) {
269
- result.__defineGetter__("decode", () => () => image.decode());
270
- }
271
- image.onload = image.onloadend = image.onerror = event => {
272
- document.dispatchEvent(new CustomEvent(IMAGE_LOADED_EVENT, { detail: image.src }));
273
- result.dispatchEvent(new Event(event.type, event));
274
- };
275
- return result;
250
+ const NativeImage = globalThis.Image;
251
+ globalThis._singleFileImage = NativeImage;
252
+ const ImageWrapper = function Image() {
253
+ const image = new NativeImage(...arguments);
254
+ const result = new NativeImage(...arguments);
255
+ result.__defineSetter__("src", value => {
256
+ image.src = value;
257
+ document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT, { detail: image.src }));
258
+ });
259
+ result.__defineGetter__("src", () => image.src);
260
+ result.__defineSetter__("srcset", value => {
261
+ document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT));
262
+ image.srcset = value;
263
+ });
264
+ result.__defineGetter__("srcset", () => image.srcset);
265
+ result.__defineGetter__("height", () => image.height);
266
+ result.__defineGetter__("width", () => image.width);
267
+ result.__defineGetter__("naturalHeight", () => image.naturalHeight);
268
+ result.__defineGetter__("naturalWidth", () => image.naturalWidth);
269
+ if (image.decode) {
270
+ const decode = function decode() { return image.decode(); };
271
+ decode.toString = function () { return "function decode() { [native code] }"; };
272
+ setFunctionName(decode, "decode");
273
+ result.__defineGetter__("decode", () => decode);
274
+ }
275
+ image.onload = image.onloadend = image.onerror = event => {
276
+ document.dispatchEvent(new CustomEvent(IMAGE_LOADED_EVENT, { detail: image.src }));
277
+ result.dispatchEvent(new Event(event.type, event));
276
278
  };
277
- });
279
+ return result;
280
+ };
281
+ ImageWrapper.prototype = NativeImage.prototype;
282
+ ImageWrapper.toString = function () { return "function Image() { [native code] }"; };
283
+ setFunctionName(ImageWrapper, "Image");
284
+ globalThis.__defineGetter__("Image", () => ImageWrapper);
278
285
  }
279
286
  const verticalZoomFactor = clientHeight / scrollHeight;
280
287
  const horizontalZoomFactor = clientWidth / scrollWidth;
@@ -371,9 +378,10 @@
371
378
 
372
379
  if (globalThis.CSS && globalThis.CSS.paintWorklet && globalThis.CSS.paintWorklet.addModule) {
373
380
  const addModule = globalThis.CSS.paintWorklet.addModule;
374
- globalThis.CSS.paintWorklet.addModule = function (moduleURL, options) {
381
+ globalThis.CSS.paintWorklet.addModule = function (moduleURL) {
375
382
  try {
376
383
  const result = addModule.apply(globalThis.CSS.paintWorklet, arguments);
384
+ const options = arguments[1];
377
385
  moduleURL = new URL(moduleURL, document.baseURI).href;
378
386
  document.dispatchEvent(new CustomEvent(NEW_WORKLET_EVENT, { detail: { moduleURL, options } }));
379
387
  return result;
@@ -382,6 +390,8 @@
382
390
  throw error;
383
391
  }
384
392
  };
393
+ globalThis.CSS.paintWorklet.addModule.toString = function () { return "function addModule() { [native code] }"; };
394
+ setFunctionName(globalThis.CSS.paintWorklet.addModule, "addModule");
385
395
  }
386
396
 
387
397
  if (globalThis.FontFace) {
@@ -412,6 +422,7 @@
412
422
  }
413
423
  };
414
424
  document.fonts.delete.toString = function () { return "function delete() { [native code] }"; };
425
+ setFunctionName(document.fonts.delete, "delete");
415
426
  const clearFonts = document.fonts.clear;
416
427
  document.fonts.clear = function () {
417
428
  try {
@@ -423,16 +434,16 @@
423
434
  }
424
435
  };
425
436
  document.fonts.clear.toString = function () { return "function clear() { [native code] }"; };
437
+ setFunctionName(document.fonts.clear, "clear");
426
438
  }
427
439
 
428
440
  if (globalThis.IntersectionObserver) {
429
441
  const origIntersectionObserver = globalThis.IntersectionObserver;
430
- globalThis.IntersectionObserver = function IntersectionObserver() {
442
+ globalThis.IntersectionObserver = function IntersectionObserver(callback) {
431
443
  try {
432
444
  const intersectionObserver = new origIntersectionObserver(...arguments);
433
445
  const observeIntersection = origIntersectionObserver.prototype.observe || intersectionObserver.observe;
434
446
  const unobserveIntersection = origIntersectionObserver.prototype.unobserve || intersectionObserver.unobserve;
435
- const callback = arguments[0];
436
447
  const options = arguments[1];
437
448
  if (observeIntersection) {
438
449
  intersectionObserver.observe = function (targetElement) {
@@ -450,6 +461,7 @@
450
461
  }
451
462
  };
452
463
  intersectionObserver.observe.toString = function () { return "function observe() { [native code] }"; };
464
+ setFunctionName(intersectionObserver.observe, "observe");
453
465
  }
454
466
  if (unobserveIntersection) {
455
467
  intersectionObserver.unobserve = function (targetElement) {
@@ -471,6 +483,7 @@
471
483
  }
472
484
  };
473
485
  intersectionObserver.unobserve.toString = function () { return "function unobserve() { [native code] }"; };
486
+ setFunctionName(intersectionObserver.unobserve, "unobserve");
474
487
  }
475
488
  observers.set(intersectionObserver, { callback, options });
476
489
  return intersectionObserver;
@@ -496,6 +509,7 @@
496
509
  }
497
510
  };
498
511
  CSSStyleSheet.prototype.replaceSync.toString = function () { return "function replaceSync() { [native code] }"; };
512
+ setFunctionName(CSSStyleSheet.prototype.replaceSync, "replaceSync");
499
513
  const orginalReplace = CSSStyleSheet.prototype.replace;
500
514
  CSSStyleSheet.prototype.replace = async function (text) {
501
515
  try {
@@ -508,10 +522,11 @@
508
522
  }
509
523
  };
510
524
  CSSStyleSheet.prototype.replace.toString = function () { return "function replace() { [native code] }"; };
525
+ setFunctionName(CSSStyleSheet.prototype.replace, "replace");
511
526
  const originalInsertRule = CSSStyleSheet.prototype.insertRule;
512
- CSSStyleSheet.prototype.insertRule = function (rule, index) {
527
+ CSSStyleSheet.prototype.insertRule = function (rule) {
513
528
  try {
514
- const result = originalInsertRule.apply(this, [rule, index]);
529
+ const result = originalInsertRule.apply(this, [rule, arguments[1]]);
515
530
  adoptedStylesheetsData.delete(this);
516
531
  return result;
517
532
  } catch (error) {
@@ -520,6 +535,7 @@
520
535
  }
521
536
  };
522
537
  CSSStyleSheet.prototype.insertRule.toString = function () { return "function insertRule() { [native code] }"; };
538
+ setFunctionName(CSSStyleSheet.prototype.insertRule, "insertRule");
523
539
  const originalDeleteRule = CSSStyleSheet.prototype.deleteRule;
524
540
  CSSStyleSheet.prototype.deleteRule = function (index) {
525
541
  try {
@@ -532,6 +548,7 @@
532
548
  }
533
549
  };
534
550
  CSSStyleSheet.prototype.deleteRule.toString = function () { return "function deleteRule() { [native code] }"; };
551
+ setFunctionName(CSSStyleSheet.prototype.deleteRule, "deleteRule");
535
552
 
536
553
  // the listener below is reached through the host element, and a closed shadow root is
537
554
  // not reachable from it, so the roots are recorded as they are created
@@ -27,7 +27,7 @@ any check failed.
27
27
 
28
28
  | Script | What it covers |
29
29
  |---|---|
30
- | `format-rules.js` | The rules of the format: charset round trip, the wrapper-tag ladder and its selection tests, the identifier, appended-data placement and declaration, password scope, the PDF and PNG faces. |
30
+ | `format-rules.js` | The rules of the format: the charset declaration and the doctype cap that keeps it inside the scan window, the wrapper-tag ladder and its selection tests, the identifier, appended-data placement and declaration, password scope, the PDF and PNG faces. |
31
31
  | `stored-trigger.js` | That a stored (uncompressed) entry whose bytes contain a rung's pattern moves the writer to the right rung. |
32
32
  | `check-determinism.js` | That the same inputs produce the same bytes, and that the levers which should change the output do. |
33
33
  | `option-wiring.js` | That every option `compression.js` reads is either declared as a caller option or classified as internal, and that `single-file.js` still builds its argument from that declaration. Guards the layer the other suites sit below. |
@@ -40,6 +40,7 @@ any check failed.
40
40
  | `filename-characters.js` | That `getValidFilename` maps a full-width lookalike one character at a time — `C++` used to be saved as `C+` — while a run of characters with no lookalike still collapses to a single replacement. |
41
41
  | `zip64.js` | That the `page.pdf` record injection accounts for the zip64 end of central directory record (§5.7): all four EOCD fields left at their sentinels, the entry counts and directory size carried in the zip64 record, the directory offset pointing at the injected record, and the archive still readable. The branch runs only past 4 GiB or 65535 entries, so nothing reached it before; the suite forces zip64 through `zipWriter.options` from inside the `writeEntries` callback, with no production lever. |
42
42
  | `byte-map.js` | That the byte offsets §8.2 of the specification prints still describe what the writer emits: the prologue order, the doctype and root tag with nothing between them, the identifier's length ahead of the region, absolute EOCD offsets, and the entry order. The specimen §8.2 documents is saved from a live URL and has never been in this repository, so none of its numbers could be checked; three of them were wrong. This builds an equivalent with no network. |
43
+ | `charset-round-trip.js` | That the encoding tables §8.4 prints still describe the WHATWG index: which 20 of the 38 encodings carry all 256 byte values through a decode injectively, the sizes of the reverse tables they need, and the five windows-1252 positions a platform codec of the same name leaves undefined. It also re-derives the reverse table the extractor ships as a literal, which no build step checks and which corrupts one byte per occurrence when wrong. |
43
44
  | `css-fonts-minifier.js` | That `removeUnusedFonts` reads the font families it prunes on correctly: a `var()` family resolved from the values the document declares and not only from the ones the body inherits, every font kept when the value is genuinely undetermined, and a multi-word family name that does not also claim a font named after its own tail. |
44
45
 
45
46
  ## The tools
@@ -0,0 +1,161 @@
1
+ /*
2
+ * Copyright 2010-2026 Gildas Lormeau
3
+ * contact : gildas.lormeau <at> gmail.com
4
+ *
5
+ * This file is part of SingleFile.
6
+ *
7
+ * The code in this file is free software: you can redistribute it and/or
8
+ * modify it under the terms of the GNU Affero General Public License
9
+ * (GNU AGPL) as published by the Free Software Foundation, either version 3
10
+ * of the License, or (at your option) any later version.
11
+ *
12
+ * The code in this file is distributed in the hope that it will be useful,
13
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
14
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU Affero
15
+ * General Public License for more details.
16
+ *
17
+ * As additional permission under GNU AGPL version 3 section 7, you may
18
+ * distribute UNMODIFIED VERSIONS OF THIS file without the copy of the GNU
19
+ * AGPL normally required by section 4, provided you include this license
20
+ * notice and a URL through which recipients can access the Corresponding
21
+ * Source.
22
+ */
23
+
24
+ // Universal mode recovers the ZIP region from the characters the HTML parser produced, so the
25
+ // declared charset has to carry all 256 byte values through a decode injectively (§2.1). Which
26
+ // encodings do is a property of the WHATWG index, not of this repository, and §8.4 prints the
27
+ // answer as a table: 20 qualify, 18 do not, and each qualifying one needs a reverse table of a
28
+ // stated size. Nothing re-derived that table -- it was measured once, by hand, outside the repo,
29
+ // and would go stale silently if an index changed or the prose were edited.
30
+ //
31
+ // The last check is the one with teeth. §5.5 requires the reverse table to be derived from the
32
+ // WHATWG index and NOT from a platform codec of the same name, because most platform codecs
33
+ // leave five windows-1252 positions undefined and those bytes occur in ordinary compressed data.
34
+ // The extractor ships that table as a literal, so it is derived once at authoring time and never
35
+ // again; here it is re-derived from the runtime's own decoder and compared entry by entry.
36
+
37
+ const BYTES = new Uint8Array(256).map((_, index) => index);
38
+
39
+ // the 20 of §8.4, in the order the section lists them
40
+ const QUALIFYING = [
41
+ "windows-1252", "iso-8859-2", "iso-8859-4", "iso-8859-5", "iso-8859-10", "iso-8859-13",
42
+ "iso-8859-14", "iso-8859-15", "iso-8859-16", "koi8-r", "koi8-u", "macintosh", "windows-1250",
43
+ "windows-1251", "windows-1254", "windows-1256", "windows-1258", "x-mac-cyrillic", "ibm866",
44
+ "x-user-defined"
45
+ ];
46
+ // the 18 that do not: eight single-byte encodings with undefined positions in their index, then
47
+ // the multi-byte ones, which decode a lone byte sequence to U+FFFD or to fewer than 256 characters
48
+ const DISQUALIFIED = [
49
+ "iso-8859-3", "iso-8859-6", "iso-8859-7", "iso-8859-8", "windows-874", "windows-1253",
50
+ "windows-1255", "windows-1257",
51
+ "utf-8", "utf-16le", "utf-16be", "gbk", "gb18030", "big5", "euc-jp", "shift_jis", "euc-kr",
52
+ "iso-2022-jp"
53
+ ];
54
+
55
+ let failures = 0;
56
+
57
+ function describe(label) {
58
+ const points = Array.from(new TextDecoder(label).decode(BYTES));
59
+ if (points.length != 256) {
60
+ return { qualifies: false, reason: points.length + " characters" };
61
+ }
62
+ const table = new Map();
63
+ const seen = new Set();
64
+ let identity = 0;
65
+ for (let byte = 0; byte < 256; byte++) {
66
+ const codePoint = points[byte].codePointAt(0);
67
+ if (codePoint == 0xFFFD) {
68
+ return { qualifies: false, reason: "U+FFFD at 0x" + byte.toString(16) };
69
+ }
70
+ if (seen.has(codePoint)) {
71
+ return { qualifies: false, reason: "collision at 0x" + byte.toString(16) };
72
+ }
73
+ seen.add(codePoint);
74
+ if (codePoint == byte) {
75
+ identity++;
76
+ } else {
77
+ table.set(codePoint, byte);
78
+ }
79
+ }
80
+ return { qualifies: true, identity, table, points };
81
+ }
82
+
83
+ const described = new Map([...QUALIFYING, ...DISQUALIFIED].map(label => [label, describe(label)]));
84
+
85
+ check("§8.4 covers the whole standard: 20 qualifying + 18 disqualified",
86
+ QUALIFYING.length + DISQUALIFIED.length == 38 && new Set([...QUALIFYING, ...DISQUALIFIED]).size == 38);
87
+
88
+ const wrongVerdict = [...described].filter(([label, result]) =>
89
+ result.qualifies != QUALIFYING.includes(label));
90
+ check("every encoding falls on the side of the table §8.4 puts it on", wrongVerdict.length == 0,
91
+ wrongVerdict.map(([label, result]) => label + " " + (result.reason || "qualifies")).join(", "));
92
+
93
+ // §8.4 quotes the extremes of the reverse-table sizes; they bound what an implementation has to
94
+ // carry to support any qualifying charset rather than only the one the reference writer declares
95
+ const qualifying = QUALIFYING.filter(label => described.get(label).qualifies);
96
+ const sizes = qualifying.map(label => [label, described.get(label).table.size]);
97
+ const smallest = Math.min(...sizes.map(([, size]) => size));
98
+ const largest = Math.max(...sizes.map(([, size]) => size));
99
+ check("the smallest reverse table is 8 entries, iso-8859-15", smallest == 8 &&
100
+ sizes.filter(([, size]) => size == smallest).map(([label]) => label).join() == "iso-8859-15");
101
+ check("the largest is 128, for koi8-r, koi8-u, ibm866 and x-user-defined", largest == 128 &&
102
+ sizes.filter(([, size]) => size == largest).map(([label]) => label).sort().join() ==
103
+ "ibm866,koi8-r,koi8-u,x-user-defined");
104
+
105
+ const windows1252 = described.get("windows-1252");
106
+ check("windows-1252 decodes 229 of the 256 values to themselves (§5.5 rule 1)",
107
+ windows1252.identity == 229, String(windows1252.identity));
108
+ check("its reverse table is the remaining 27 (§5.5 rule 2)", windows1252.table.size == 27,
109
+ String(windows1252.table.size));
110
+
111
+ // the trap of §5.5: iso-8859-1 is a LABEL of windows-1252, not the identity mapping its name
112
+ // suggests, so a reader that treats it as latin-1 builds a table with no entries at all
113
+ check("iso-8859-1 is a label of windows-1252, not a separate identity encoding",
114
+ new TextDecoder("iso-8859-1").encoding == "windows-1252" &&
115
+ new TextDecoder("latin1").encoding == "windows-1252");
116
+
117
+ // the five positions of the §5.5 table: the WHATWG index assigns them, most platform codecs do not
118
+ check("the WHATWG index assigns 0x81, 0x8D, 0x8F, 0x90 and 0x9D (§5.5)",
119
+ [0x81, 0x8D, 0x8F, 0x90, 0x9D].every(byte =>
120
+ windows1252.points[byte].codePointAt(0) == byte && !windows1252.table.has(byte)));
121
+
122
+ // §8.4's caveat on x-user-defined: it qualifies on the criterion and is still a poor choice
123
+ check("x-user-defined maps 0x80-0xFF into the Private Use Area, U+F780-U+F7FF",
124
+ [...Array(128).keys()].every(index =>
125
+ described.get("x-user-defined").points[128 + index].codePointAt(0) == 0xF780 + index));
126
+
127
+ // the round trip of §5.5 rules 1 and 2, on every byte value and every qualifying charset. NUL and
128
+ // the newlines need rules 3 and 4 in a browser, where the parser has replaced and normalized them;
129
+ // through a decoder alone they arrive intact, so the mapping is exact for all 256 values here
130
+ const broken = qualifying.filter(label => {
131
+ const { table, points } = described.get(label);
132
+ return points.some((character, byte) => {
133
+ const codePoint = character.codePointAt(0);
134
+ return (table.has(codePoint) ? table.get(codePoint) : codePoint) != byte;
135
+ });
136
+ });
137
+ check("all 256 byte values survive decode and reverse mapping, under every qualifying charset",
138
+ broken.length == 0, broken.join(", "));
139
+
140
+ // the extractor's own table, re-derived. It is inlined into every archive, so an error here is
141
+ // not caught by any build step and corrupts one byte per occurrence in the recovered region
142
+ const compression = await Deno.readTextFile(new URL("../../processors/compression/compression.js", import.meta.url));
143
+ const literal = compression.slice(compression.indexOf("const characterMap = new Map(["));
144
+ const shipped = new Map([...literal.slice(0, literal.indexOf("]);")).matchAll(/\[(\d+),\s*(\d+)\]/g)]
145
+ .map(([, codePoint, byte]) => [Number(codePoint), Number(byte)]));
146
+ const expected = new Map([[0xFFFD, 0], ...windows1252.table]);
147
+ const wrongEntries = [...expected].filter(([codePoint, byte]) => shipped.get(codePoint) !== byte);
148
+ const extraEntries = [...shipped].filter(([codePoint]) => !expected.has(codePoint));
149
+ check("the extractor's characterMap is the derived windows-1252 table plus U+FFFD (28 entries)",
150
+ shipped.size == 28 && wrongEntries.length == 0 && extraEntries.length == 0,
151
+ "missing/wrong " + JSON.stringify(wrongEntries) + " extra " + JSON.stringify(extraEntries));
152
+
153
+ console.log(failures ? `\n${failures} check(s) FAILED` : "\nall checks passed");
154
+ Deno.exit(failures ? 1 : 0);
155
+
156
+ function check(label, condition, detail) {
157
+ if (!condition) {
158
+ failures++;
159
+ }
160
+ console.log((condition ? "PASS" : "FAIL") + " " + label + (condition || !detail ? "" : ": " + detail));
161
+ }
@@ -1,5 +1,5 @@
1
1
  import "./dom-stub.js";
2
- import { makePageData, makeOptions, runProcess, mulberry32 } from "./common.js";
2
+ import { makePageData, makeOptions, runProcess, mulberry32, freezeDate } from "./common.js";
3
3
  import { ZipReader, ZipWriter, BlobReader } from "../../vendor/zip/zip.js";
4
4
 
5
5
  // the quote is there because the escaper the title shares with the table of contents encodes
@@ -239,6 +239,65 @@ function countIdentifiers(text) {
239
239
  check("a relocated archive declares no comment", view.getUint16(bytes.length - 2, true), 0);
240
240
  }
241
241
 
242
+ function makeNewlinePageData(seed, newlineCount) {
243
+ const pageData = makePageData(seed, 4 * 1024);
244
+ const rand = mulberry32(seed);
245
+ const newlines = ["\n", "\r", "\r\n"];
246
+ let content = "";
247
+ for (let index = 0; index < newlineCount; index++) {
248
+ content += newlines[(rand() * 3) | 0];
249
+ }
250
+ pageData.resources.stylesheets.push({ name: "newlines.txt", extension: ".txt", content, url: "https://example.com/newlines.txt" });
251
+ return pageData;
252
+ }
253
+
254
+ // libarchive gives up looking for the end of central directory record 16383 bytes from the end of
255
+ // the file, so the default budget is 16361 appended bytes: that window minus the 22-byte record.
256
+ // This fixture's payload lands between the default and the 65535-byte comment ceiling, which is
257
+ // the range the old budget kept appended and bsdtar could not open.
258
+ // The Date must be frozen: `archiveTime` (compression.js:155) puts an ISO timestamp in the
259
+ // archive, whose milliseconds move a few newline bytes in and out of the recovered range and
260
+ // therefore change the payload length build to build. The boundary checks below compare a budget
261
+ // against a run measured in an EARLIER build, so without freezing they are off by a few bytes one
262
+ // run in three. check-determinism.js asserts both halves of that
263
+ {
264
+ const restoreDate = freezeDate();
265
+ try {
266
+ const wide = makeOptions({ disableCompression: true, maxAppendedDataLength: 65535 });
267
+ const { bytes: wideBytes } = await runProcess(makeNewlinePageData(43, 80 * 1000), wide);
268
+ const wideTail = readAppendedData(wideBytes).trailing;
269
+ check("a wider budget keeps the payload appended", wide.extraDataSize, undefined);
270
+ check("the fixture overflows the default budget", wideTail > 16361, true);
271
+ check("the fixture fits the comment ceiling", wideTail <= 65535, true);
272
+
273
+ const byDefault = makeOptions({ disableCompression: true });
274
+ const { bytes } = await runProcess(makeNewlinePageData(43, 80 * 1000), byDefault);
275
+ check("the default budget relocates the payload", byDefault.extraDataSize > 0, true);
276
+ check("the default budget keeps the record in libarchive's window", readAppendedData(bytes).trailing <= 16361, true);
277
+
278
+ const fitting = makeOptions({ disableCompression: true, maxAppendedDataLength: wideTail });
279
+ await runProcess(makeNewlinePageData(43, 80 * 1000), fitting);
280
+ check("a budget matching the run to the byte keeps it appended", fitting.extraDataSize, undefined);
281
+
282
+ const tight = makeOptions({ disableCompression: true, maxAppendedDataLength: wideTail - 1 });
283
+ await runProcess(makeNewlinePageData(43, 80 * 1000), tight);
284
+ check("one byte below the run relocates it", tight.extraDataSize > 0, true);
285
+ } finally {
286
+ restoreDate();
287
+ }
288
+ }
289
+
290
+ // the budget and the comment ceiling are two different limits, and only the second is a property
291
+ // of the format. A budget raised past 65535 produces a run the comment-length field cannot hold,
292
+ // which setUint16 would write back modulo 65536: the writer leaves it undeclared instead
293
+ {
294
+ const options = makeOptions({ disableCompression: true, declareAppendedData: true, maxAppendedDataLength: Number.MAX_SAFE_INTEGER });
295
+ const { bytes } = await runProcess(makeNewlinePageData(44, 320 * 1000), options);
296
+ const { declared, trailing } = readAppendedData(bytes);
297
+ check("a run past the comment ceiling stays appended", trailing > 65535, true);
298
+ check("a run past the comment ceiling is left undeclared", declared, 0);
299
+ }
300
+
242
301
  {
243
302
  const options = makeOptions({ embeddedPdf: PDF });
244
303
  const pageData = makePageData(15, 4 * 1024);