single-file-core 1.5.125 → 1.5.126

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/core/index.js CHANGED
@@ -1361,6 +1361,11 @@ class Processor {
1361
1361
  }
1362
1362
  }
1363
1363
  }));
1364
+ frameElements.forEach(frameElement => {
1365
+ if (!frameElement.getAttribute("src") && !frameElement.getAttribute("srcdoc") && !frameElement.getAttribute("data")) {
1366
+ frameElement.removeAttribute("sandbox");
1367
+ }
1368
+ });
1364
1369
 
1365
1370
  async function initializeProcessor(frameData, frameElement, frameWindowId, batchRequest, options) {
1366
1371
  options.insertSingleFileComment = false;
@@ -243,9 +243,16 @@ class ProcessorHelperCommon {
243
243
  return serializeSrcset([Object.assign({}, srcsetValue, { url: resourceURL })]);
244
244
  }
245
245
  }));
246
- resourceElement.setAttribute("srcset", srcsetValues.filter(srcsetValue => srcsetValue).join(", "));
246
+ const newSrcset = srcsetValues.filter(srcsetValue => srcsetValue).join(", ");
247
+ if (newSrcset) {
248
+ resourceElement.setAttribute("srcset", newSrcset);
249
+ } else {
250
+ resourceElement.removeAttribute("srcset");
251
+ resourceElement.removeAttribute("sizes");
252
+ }
247
253
  } else {
248
- resourceElement.setAttribute("srcset", "");
254
+ resourceElement.removeAttribute("srcset");
255
+ resourceElement.removeAttribute("sizes");
249
256
  }
250
257
  }));
251
258
  }
@@ -267,6 +274,7 @@ class ProcessorHelperCommon {
267
274
  element.style.setProperty("background-size", style && style["background-size"] ? style["background-size"] : "100% 100%", "important");
268
275
  element.style.setProperty("background-origin", "content-box", "important");
269
276
  element.style.setProperty("background-repeat", "no-repeat", "important");
277
+ element.style.setProperty("background-attachment", "scroll", "important");
270
278
  }
271
279
 
272
280
  async getStylesheetContent(resourceURL, options) {
@@ -443,16 +451,10 @@ class ProcessorHelperCommon {
443
451
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
444
452
  const key = this.getFontKey(ruleData);
445
453
  const fontInfo = fontsDetails.fonts.get(key);
446
- if (fontInfo) {
447
- const ruleKey = key + " " + this.getPropertyValue(ruleData, "src");
448
- if (fontsDetails.emittedFonts.has(ruleKey)) {
454
+ if (fontInfo && fontsDetails.lastRules.get(key) == ruleData) {
455
+ const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
456
+ if (!keptRule) {
449
457
  removedRules.push(cssRule);
450
- } else {
451
- fontsDetails.emittedFonts.add(ruleKey);
452
- const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
453
- if (!keptRule) {
454
- removedRules.push(cssRule);
455
- }
456
458
  }
457
459
  } else {
458
460
  removedRules.push(cssRule);
@@ -490,16 +492,22 @@ class ProcessorHelperCommon {
490
492
  fontInfo = [];
491
493
  mediaFontsDetails.fonts.set(fontKey, fontInfo);
492
494
  }
495
+ mediaFontsDetails.lastRules.set(fontKey, ruleData);
493
496
  const src = this.getPropertyValue(ruleData, "src");
494
497
  if (src) {
495
498
  const fontSources = src.match(REGEXP_URL_FUNCTION);
496
499
  if (fontSources) {
500
+ const ruleSources = [];
497
501
  fontSources.forEach(source => {
498
502
  if (fontInfo.includes(source)) {
499
503
  fontInfo.splice(fontInfo.indexOf(source), 1);
500
504
  }
501
- fontInfo.unshift(source);
505
+ if (ruleSources.includes(source)) {
506
+ ruleSources.splice(ruleSources.indexOf(source), 1);
507
+ }
508
+ ruleSources.unshift(source);
502
509
  });
510
+ ruleSources.forEach(source => fontInfo.push(source));
503
511
  }
504
512
  }
505
513
  }
@@ -512,7 +520,7 @@ class ProcessorHelperCommon {
512
520
  medias: new Map(),
513
521
  supports: new Map(),
514
522
  layers: new Map(),
515
- emittedFonts: new Set()
523
+ lastRules: new Map()
516
524
  };
517
525
  }
518
526
 
package/core/util.js CHANGED
@@ -69,11 +69,6 @@ const EXPECTED_TYPES_MEDIA = ["font", "image", "video", "audio"];
69
69
  const URL = globalThis.URL;
70
70
  const DOMParser = globalThis.DOMParser;
71
71
  const Blob = globalThis.Blob;
72
- const fetch = (url, options) => {
73
- options.cache = "force-cache";
74
- options.referrerPolicy = "strict-origin-when-cross-origin";
75
- return globalThis.fetch(url, options);
76
- };
77
72
  const TextDecoder = globalThis.TextDecoder;
78
73
  const URLSearchParams = globalThis.URLSearchParams;
79
74
 
@@ -83,8 +78,8 @@ export {
83
78
 
84
79
  function getInstance(utilOptions) {
85
80
  utilOptions = utilOptions || {};
86
- utilOptions.fetch = utilOptions.fetch || fetch;
87
- utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch || fetch;
81
+ utilOptions.fetch = utilOptions.fetch || ((url, options) => globalThis.fetch(url, { ...options, cache: "force-cache", referrerPolicy: "strict-origin-when-cross-origin" }));
82
+ utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch;
88
83
  return {
89
84
  getDoctypeString,
90
85
  getFilenameExtension(resourceURL, replacedCharacters, replacementCharacter, replacementCharacters) {
@@ -264,7 +259,7 @@ function getInstance(utilOptions) {
264
259
  // eslint-disable-next-line no-unused-vars
265
260
  } catch (error) {
266
261
  response = await Promise.race([
267
- fetchResource(resourceURL, { headers: { accept } }),
262
+ fetchResource(resourceURL, { referrer: options.resourceReferrer, headers: { accept } }),
268
263
  networkTimeoutPromise
269
264
  ]);
270
265
  }
@@ -136,7 +136,7 @@ Three consequences shape everything below:
136
136
  | **region** | A byte range with a single producer, named in §3. Regions are the units the rest of this document reasons about; a region can appear in several pieces — `html-prologue` resumes after the embedded PDF document in the PDF variants, and after the `tEXt "ZIP"` chunk header in the PNG ones, so with all four faces it comes in three. |
137
137
  | **universal mode** | The variant whose HTML face can extract the archive from the *parsed page text*, the text and comment nodes the HTML parser produced, and therefore needs no access to its own raw bytes. Named "universal" because it works from any location, including the `file:` protocol. |
138
138
  | **wrapper tag** | The HTML construct that hides a binary region from the HTML parser, `<!--`…`-->` by default (§5.1). |
139
- | **appended data** | Bytes after the ZIP End Of Central Directory record. Readers tolerate them within the window their EOCD scan already covers: 65557 bytes from the end of the file (the 22-byte record plus the 65535-byte maximum comment length); "the 64 KB window" refers to this. It may be left undeclared or declared as the archive comment; both forms are valid ZIP and readers MUST accept both (§4.2). The recovery payload can be computed before that choice is made because it stops two bytes short of the record, excluding its comment-length field (see *recovered range* below). |
139
+ | **appended data** | Bytes after the ZIP End Of Central Directory record. A reader tolerates them as far back as its EOCD scan reaches, and how far that is varies by an order of magnitude: 65557 bytes from the end of the file for Python `zipfile` (the 22-byte record plus the 65535-byte maximum comment length), but 16383 for libarchive and 32768 for perl `Archive::Zip` (§8.1). No reader's window is guaranteed, so a writer keeps its own narrower budget (§5.2). The format's one hard limit is the 65535-byte comment field, and it binds only a run the writer declares (§4.2). It may be left undeclared or declared as the archive comment; both forms are valid ZIP and readers MUST accept both (§4.2). The recovery payload can be computed before that choice is made because it stops two bytes short of the record, excluding its comment-length field (see *recovered range* below). |
140
140
  | **ZIP region** | The contiguous byte range holding the archive proper: from the first local file header the ZIP writer emitted through the last byte of the End Of Central Directory record. It spans the `zip-entries`, `pdf-central-record` (when present) and `central-directory · eocd` blocks of §3, and in the HTML variants it is the content of the last wrapper, exactly so on the element rungs and preceded by the `sfz-data` identifier on the comment rung, which the extractor steps over. It does **not** include `pdf-local-header` or the PDF document, which sit earlier in the file. |
141
141
  | **archive** | The *logical* ZIP file: the set of entries the central directory describes, wherever their bytes lie. This is distinct from the ZIP region above, which is a contiguous byte range. Every entry but one has its bytes inside the region; `page.pdf` is the deliberate exception, an entry of the archive whose local header and data sit before the region (§4.2). "Archive" in this document always means the logical file, "ZIP region" always the byte range, and the two differ only in the PDF-with-HTML variants. |
142
142
  | **recovered range** | What the universal extractor reproduces (§4.5): the ZIP region minus its last two bytes, the comment-length field of the End Of Central Directory record. That field is the one part of the record whose value depends on what follows the region, so leaving it out is what lets a writer decide the appended-data form after the recovery payload is final (§4.2). The extractor supplies the two bytes itself, as zeroes — the recovered range carries no comment. |
@@ -174,8 +174,9 @@ every acquisition path including the ones that read raw bytes and could return i
174
174
  HTML face, so the option does not apply. The *Specimen* column names the measured
175
175
  reference files this document cites; §8 records how to regenerate them.
176
176
 
177
- Other writer options shape the file without adding a face: `preventAppendedData` and
178
- `declareAppendedData` (§4.2, §5.2), `includeBOM` (§3.1), `insertTextBody` (§4.6),
177
+ Other writer options shape the file without adding a face: `preventAppendedData`,
178
+ `declareAppendedData` and `maxAppendedDataLength` (§4.2, §5.2), `includeBOM` (§3.1),
179
+ `insertTextBody` (§4.6),
179
180
  `password` (§5.6), `createRootDirectory` (§7.1), and the head-element switches
180
181
  `insertCanonicalLink`, `insertMetaNoIndex` and `insertMetaCSP` (§3.1).
181
182
 
@@ -328,7 +329,8 @@ the same way: only a universal file carries an `<sfz-extra-data>` element.
328
329
 
329
330
  Unless a row states otherwise, the layouts below are measured from specimen files
330
331
  saved from `example.com` (the generation commands are in §8). The relocated row covers
331
- two cases with one layout, `preventAppendedData` and a payload over 64 KB: the first is
332
+ two cases with one layout, `preventAppendedData` and a payload over the appended-data
333
+ budget (§5.2): the first is
332
334
  measured on the relocated specimen, the second derived from the writer rules, because
333
335
  such a payload requires an archive too large for a readable specimen. The figure below shows
334
336
  the regions and their order; the glossary of §3.1 is the normative list, and it states
@@ -353,7 +355,7 @@ face adds, then the regions the PNG face adds.
353
355
  | `<!--` / `-->` | HTML | HTML face | The wrapper tag pair hiding a binary region from the HTML parser — comment tags by default, another pair when the hidden bytes defeat them — which `-->` is only the commonest way to do, the full test being `<!--`, `--!>`, a trailing `<!-` and, for the PNG payload, a leading `>` or `->` (§5.1). Drawn at each opening and closing position. The close tag is absent whenever the recovery payload is relocated (§5.2): under `preventAppendedData`, when the payload outgrows the appended-data budget, or on the `<plaintext>` wrapper which cannot close. No markup then follows the archive and the wrapper runs to end-of-file. That does not mean the file ends at the EOCD — the PNG face's tail still follows, inside the wrapper, where it parses as text (§5.1). |
354
356
  | `zip-entries` | ZIP | always | The archive's local file headers and entry data, written by the ZIP writer. The central directory of an archive written by the reference writer lists `index.html` (the page) first, then `manifest.json` (a JSON description of the archive: original URL, title, save time, resource-to-URL map — informative; the page displays without it), then the resources; the *physical* order of the local headers inside the region is not guaranteed to match, and readers MUST NOT rely on either order — entries are addressed by name (§7.1). |
355
357
  | `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
356
- | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the 64 KB appended-data window or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
358
+ | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the appended-data budget (§5.2) or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
357
359
  | `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
358
360
  | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag set as on every other entry — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
359
361
  | `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
@@ -366,7 +368,7 @@ face adds, then the regions the PNG face adds.
366
368
  | `crc · IEND` | PNG | PNG face | The `tEXt "ZIP"` chunk's CRC, computed once the archive bytes are final (§6), followed by the empty `IEND` chunk — the last bytes of the file (PNG requires `IEND` to end the stream, which is why the PNG variants drop the end tags). |
367
369
 
368
370
  The reader-by-reader interpretation of these regions is §4; the mechanics that keep
369
- them from colliding (wrapper-tag selection, checksums, offsets, the 64 KB budget) are
371
+ them from colliding (wrapper-tag selection, checksums, offsets, the appended-data budget) are
370
372
  §5.
371
373
 
372
374
  ## 4. Reader lenses
@@ -447,10 +449,12 @@ path the plain variant's error message describes.
447
449
  ### 4.2 The ZIP reader
448
450
 
449
451
  The ZIP face is read from the end. A reader locates the End Of Central Directory
450
- record by scanning backward from end-of-file; the format guarantees it lies within
451
- the window every reader must already scan to support archive comments (§1.3,
452
- *appended data*), because everything after it — wrapper close tag, extra-data, end
453
- tags, PNG tail — fits the appended-data budget (§5.2). Accepting *undeclared* bytes
452
+ record by scanning backward from end-of-file, and how far back it scans is the one
453
+ reader property the format cannot assume (§1.3, *appended data*). Everything the
454
+ writer emits after the record — wrapper close tag, extra-data, end tags, PNG tail —
455
+ fits the appended-data budget of §5.2, and the reference writer sizes that budget to
456
+ the narrowest scan measured in §8.1, so the record stays reachable for every reader
457
+ listed there. Accepting *undeclared* bytes
454
458
  in that window is itself a customary tolerance (§1.1): the ZIP specification
455
459
  documents the comment, not trailing junk. From the EOCD
456
460
  the reader jumps to the central directory and reads only what it references;
@@ -510,6 +514,14 @@ readers accept at all: `java.util.zip`, and therefore Android and most JVM tooli
510
514
  rejects an archive with undeclared trailing bytes outright (§8.1). A writer SHOULD
511
515
  offer both and default to raw.
512
516
 
517
+ The declared form carries a ceiling the raw form does not. The comment length is a
518
+ 16-bit field, so a run longer than 65535 bytes cannot be declared at all. A writer
519
+ whose appended-data budget (§5.2) is raised past that ceiling MUST leave such a run
520
+ undeclared rather than write its length back modulo 65536, and readers that accept
521
+ only the declared form then reject the file with no diagnostic. The budget and the
522
+ ceiling are two separate limits, and a writer that exposes the first as an option
523
+ SHOULD say so where it documents it.
524
+
513
525
  Neither form constrains the other faces, and universal mode supports both, because
514
526
  the recovery payload describes the recovered range rather than the whole region: the
515
527
  comment-length field is excluded (§1.3), so its value can be decided after the
@@ -1011,16 +1023,32 @@ re-check the field.
1011
1023
 
1012
1024
  ### 5.2 The appended-data budget
1013
1025
 
1014
- Everything the writer emits after the EOCD record MUST fit in 65535 bytes — the
1015
- maximum length a ZIP archive comment may declare, and therefore the distance beyond
1016
- the record that every reader's backward scan already covers (§1.3). The appended run is:
1026
+ The run the writer emits after the EOCD record has two limits, and only one of them
1027
+ comes from the format. A run *declared* as the archive comment MUST fit in 65535
1028
+ bytes, the largest value a comment-length field can hold (§4.2). A run left *raw* has
1029
+ no format limit at all: the bytes are outside the archive, and nothing in ZIP bounds
1030
+ them. What bounds both in practice is the reader. Locating the EOCD record means
1031
+ scanning backward from end-of-file, and the searches measured in §8.1 stop at 16383
1032
+ bytes for libarchive, 32768 for perl `Archive::Zip` and 65557 for Python `zipfile`, so
1033
+ a run sized to the comment ceiling is already invisible to the narrowest of them. A
1034
+ writer therefore keeps a *budget*, sized to the readers it means to satisfy rather
1035
+ than to the format. The appended run is:
1017
1036
 
1018
1037
  ```
1019
1038
  wrapper close tag + extra-data element + end tags + (PNG face: 4-byte chunk CRC + 12-byte IEND)
1020
1039
  ```
1021
1040
 
1022
- and the writer compares its total against 65535 before committing to it. The EOCD
1023
- record's own 22 bytes sit inside the window too, giving the 65557-byte figure of §1.3.
1041
+ and the writer compares its total against that budget before committing to it. The
1042
+ EOCD record's own 22 bytes sit inside a reader's window as well, which is what turns a
1043
+ 65535-byte run into the 65557 bytes of §1.3 and the reference writer's budget into
1044
+ libarchive's 16383.
1045
+
1046
+ The reference writer exposes the budget as `maxAppendedDataLength` and defaults it to
1047
+ 16361 bytes: libarchive's window less the 22 bytes of the record, which is the largest
1048
+ run behind which every reader of §8.1 still finds the record. A writer MAY choose
1049
+ another value. Raising it above 65535 leaves the run undeclarable: it is emitted, and
1050
+ it is still valid ZIP, but no comment length can cover it, and §4.2 says what that
1051
+ costs.
1024
1052
 
1025
1053
  Only the extra-data element can outgrow the budget: it carries one 2-bit code per
1026
1054
  newline sequence in the recovered range — the ZIP region without its comment-length
@@ -1028,9 +1056,9 @@ field (§4.5), CR LF counting once, for two bytes (§5.5) — so it
1028
1056
  grows with the archive. Newline bytes
1029
1057
  occur at their natural density in compressed and STOREd binary data — about two in
1030
1058
  every 256 bytes — and the codes are compressed and base64-encoded, which measures at
1031
- one byte of element per 650 bytes of archive at scale (§8). The budget is therefore
1032
- exhausted at an archive of roughly 40 MB, so the relocated placement is rare in
1033
- practice. That ratio is the large-archive limit and must not be used to size a
1059
+ one byte of element per 650 bytes of archive at scale (§8). The default budget is
1060
+ therefore exhausted at an archive of roughly 10 MB, and the 65535-byte ceiling at
1061
+ roughly 40 MB, so the relocated placement is uncommon in practice. That ratio is the large-archive limit and must not be used to size a
1034
1062
  particular file: deflate's overhead is a fixed cost spread over a growing payload, so
1035
1063
  small archives are far less efficient. Measured on exact byte counts, a 6099-byte region
1036
1064
  needs 69 bytes of element — a ratio of 88 — and a 74057-byte region needs 189, a ratio
@@ -1051,6 +1079,14 @@ trailing bytes open it (§8.1). The parser closes the open
1051
1079
  comment or element at end of file, and `</body></html>` are implied, so the page
1052
1080
  renders the same.
1053
1081
 
1082
+ Relocation is not a move at constant size, and it can end either way. Two effects pull
1083
+ against each other: the room the writer sets aside, which in the reference writer is
1084
+ `Math.ceil(length * 1.01) + 32` bytes — a percentage of the payload plus a constant, so
1085
+ the constant dominates a small payload and the percentage a large one — and the 17 bytes
1086
+ of wrapper terminator and end tags it stops emitting. Measured on three files the net ran
1087
+ from 9 bytes saved to 190 bytes spent, so a writer sizing a file should quote that range
1088
+ rather than a single figure.
1089
+
1054
1090
  ### 5.3 Offset bookkeeping
1055
1091
 
1056
1092
  Three coordinate systems coexist in one file, and the format's job is to keep each
@@ -1467,7 +1503,7 @@ terminate because each of them advances a monotone quantity:
1467
1503
  There is no converse of the second: a pass that reserved room never discards it,
1468
1504
  even when the relocated payload would have fit the appended window. Relocation moves
1469
1505
  the archive, which changes the offsets, which changes the payload that made the
1470
- relocation necessary, so a payload lying on the 65535-byte boundary can be too large
1506
+ relocation necessary, so a payload lying on the budget boundary can be too large
1471
1507
  appended and small enough relocated, and a writer that dropped the reservation could
1472
1508
  rebuild the two placements forever. Relocation is therefore final (§5.2), and the file
1473
1509
  keeps at most the reservation's own margin of dead padding.
@@ -1622,7 +1658,7 @@ only if it affects the bytes the page is built from:
1622
1658
  | An entry's CRC-32 or AES authentication code does not match | **SHOULD** fail for that entry, and MUST NOT present a page rebuilt from it as intact |
1623
1659
  | `page.pdf` was reconstructed from the parsed page and its CRC-32 does not match | **MUST** discard the reconstruction (§4.5). The bytes are a guess about newlines the recovery payload does not describe, and the checksum is the only thing that tests it — unlike the row above, there is no read to have gone wrong, only an inference |
1624
1660
  | Bytes outside the archive proper — before the first local file header, after the EOCD record, or between an entry's data and the next header | **MUST** tolerate: they are the other faces (§7.1). The gap in the middle is not hypothetical: with the PDF face the bootstrap lies between `page.pdf`'s data and the ZIP region |
1625
- | The appended run exceeds the 65535-byte budget (§5.2) | Not a reader's problem: if the EOCD record was found, the archive is readable. Readers MAY warn |
1661
+ | The appended run exceeds the 65535-byte ceiling (§5.2) | Not a reader's problem: if the EOCD record was found, the archive is readable. Readers MAY warn |
1626
1662
  | A `tEXt` chunk CRC does not match, or a chunk holds bytes PNG does not permit (§4.4) | Irrelevant to extraction; a reader of the archive MAY ignore both |
1627
1663
  | `page.pdf` is present but its data does not begin with `%PDF-` | Not an error. The entry is data like any other |
1628
1664
  | `index.html` is present without `manifest.json` | **MUST** still extract (§7.1) |
@@ -1662,11 +1698,20 @@ and is class C.
1662
1698
  | Info-ZIP `unzip`, `zipinfo` | ✔ | ✔ | ✔ | Lists and extracts every variant. AES entries are skipped — `need PK compat. v5.1 (can do v4.5)` — a limitation of the tool, not of the file; `page.pdf` still extracts because it is never encrypted |
1663
1699
  | Python `zipfile` | ✔ | ✔ | ✔ | Lists and extracts every variant |
1664
1700
  | 7-Zip (`7zz`) | ✔ | ✔ | ✔ | Lists and extracts every variant, AES included |
1665
- | libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant |
1701
+ | libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant. Its EOCD scan is the narrowest measured, so a class-C file whose appended run pushes the record past 16383 bytes from the end is rejected with `Unrecognized archive format`; §5.2's default budget is sized to this window, and the ✔ holds for files that respect it |
1666
1702
  | libarchive `bsdtar`, piped input | ✔ | ✘ | ✘ | `Unrecognized archive format` — the forward-only case of §1.2, measured |
1667
1703
  | Java `java.util.zip` (`jar tf`) | ✔ | ✔ | ✘ | `zip END header not found` whenever bytes follow the EOCD undeclared. Declaring them as the archive comment makes the same file open, measured on every class-C variant (§4.2) |
1668
1704
  | macOS `ditto -x -k` | ✔ | ✘ | ✘ | `Couldn't read PKZip signature` — requires a local file header at offset 0, so prepended data alone defeats it |
1669
1705
 
1706
+ The backward scans behind the class-C column differ by an order of magnitude, and they
1707
+ are what §5.2's budget is sized against. Measured by padding a working archive until
1708
+ the record fell out of reach, the largest distance from end-of-file at which each
1709
+ reader still finds the EOCD record is: libarchive 16383, perl `Archive::Zip` 32768,
1710
+ Python `zipfile` 65557, zip.js 65536, Info-ZIP `unzip` 68000, and 7-Zip beyond 1 MiB,
1711
+ which scans the whole file. libarchive binds, and its window less the 22-byte record
1712
+ is the 16361-byte default budget of §5.2. macOS `ditto` is not on this axis at all: it
1713
+ requires a local file header at offset 0 whatever the tail looks like.
1714
+
1670
1715
  The cost of the declared form was measured on the same tools: it is a display cost, not
1671
1716
  a compatibility one. An archive whose trailing bytes are declared as the comment has
1672
1717
  them printed back on ordinary listings; `unzip -l` reproduces the whole run — in
@@ -1741,7 +1786,7 @@ specimen without a network.
1741
1786
  | 123006 | `<sfz-extra-data>` … `</sfz-extra-data>` | recovery payload, appended placement (§5.2); 24 base64 characters for this archive |
1742
1787
  | 123063 | `</body></html>` | end tags; end of file at 123077 |
1743
1788
 
1744
- The appended run is 74 bytes, well inside the 65535-byte budget (§5.2). The ZIP region
1789
+ The appended run is 74 bytes, well inside the 16361-byte default budget (§5.2). The ZIP region
1745
1790
  is the 998 bytes from 122005 to 123003; the universal extractor reproduces the first 996
1746
1791
  of them and supplies the last two itself (§1.3).
1747
1792
 
@@ -1775,7 +1820,7 @@ These specimens are deliberately small, and a reader tested only against them is
1775
1820
  undertested: they are all flat archives of two or three entries. None
1776
1821
  exercises a root directory, `frames/<n>/` nesting, a second `index.html`, a `data:`-URL
1777
1822
  entry comment, the optional text body (§4.6), a UTF-8 BOM, zip64
1778
- (§5.7), a payload past the 64 KB budget, or a relocated reservation with padding left
1823
+ (§5.7), a payload past the appended-data budget, or a relocated reservation with padding left
1779
1824
  in it. Two omissions matter more than the rest, because they are the parts of §5.1 a
1780
1825
  writer is most likely to get wrong: no specimen defeats a rung by its **start**
1781
1826
  pattern, and none defeats one with an **upper-case** pattern. A writer that tested only
@@ -1847,6 +1892,7 @@ predicts.
1847
1892
  | August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
1848
1893
  | August 2026 | Core 1.5.119: the inlined ZIP library is built ASCII-only, and §2.1 now requires it of any bootstrap. Its CP437 table had been emitted as literal characters, which the page re-decoded as windows-1252, growing the table from 256 entries to 508 and shifting every lookup by 60 — so the one entry read without the UTF-8 flag, `page.pdf`, came back mangled and no archive with a PDF face extracted in any engine (§5.8) |
1849
1894
  | August 2026 | Core 1.5.120: the hand-built `page.pdf` records set the language encoding flag, like every entry the ZIP writer produces (§5.8). Its name is ASCII, so no decoded name changes; what changes is that no entry in an archive is read through CP437 any more, closing the path the 1.5.119 defect surfaced on |
1895
+ | September 2026 | Core 1.5.126: the appended-data budget becomes the `maxAppendedDataLength` writer option and its default drops from 65535 to 16361 bytes, so the EOCD record stays inside libarchive's scan and `bsdtar` opens archives it used to reject (§5.2, §8.1). The 65535-byte comment ceiling is now a separate limit, stated in §4.2: a budget raised past it produces a run that cannot be declared |
1850
1896
 
1851
1897
  This document was itself revised in August 2026, against core 1.5.108, after several
1852
1898
  independent reviews. One of them was a reader built from this specification alone, with
@@ -83,15 +83,15 @@ function getSourceSrcData(sources) {
83
83
  function setSrc(srcData, imgElement, pictureElement) {
84
84
  if (srcData.src) {
85
85
  imgElement.setAttribute("src", srcData.src);
86
- imgElement.setAttribute("srcset", "");
87
- imgElement.setAttribute("sizes", "");
86
+ imgElement.removeAttribute("srcset");
87
+ imgElement.removeAttribute("sizes");
88
88
  } else {
89
89
  imgElement.setAttribute("src", EMPTY_RESOURCE);
90
90
  if (srcData.srcset) {
91
91
  imgElement.setAttribute("srcset", srcData.srcset);
92
92
  } else {
93
- imgElement.setAttribute("srcset", "");
94
- imgElement.setAttribute("sizes", "");
93
+ imgElement.removeAttribute("srcset");
94
+ imgElement.removeAttribute("sizes");
95
95
  }
96
96
  }
97
97
  if (pictureElement) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "single-file-core",
3
- "version": "1.5.125",
3
+ "version": "1.5.126",
4
4
  "description": "SingleFile Core",
5
5
  "author": "Gildas Lormeau",
6
6
  "license": "AGPL-3.0-or-later",
@@ -89,7 +89,8 @@ const PNG_CHUNK_CRC_LENGTH = 4;
89
89
  const PNG_SIGNATURE_LENGTH = 8;
90
90
  const PNG_IHDR_LENGTH = 25;
91
91
  const COMMENT_LENGTH_FIELD_LENGTH = 2;
92
- const MAX_APPENDED_DATA_LENGTH = 65535;
92
+ const MAX_ZIP_COMMENT_LENGTH = 65535;
93
+ const DEFAULT_MAX_APPENDED_DATA_LENGTH = 16361;
93
94
  const PDF_ENTRY_FILENAME = "page.pdf";
94
95
  const PRESCAN_WINDOW_LENGTH = 1024;
95
96
  const PNG_TEXT_CHUNK_HEADER_LENGTH = 12;
@@ -121,6 +122,7 @@ const PROCESS_OPTION_NAMES = [
121
122
  "insertMetaCSP",
122
123
  "insertMetaNoIndex",
123
124
  "insertTextBody",
125
+ "maxAppendedDataLength",
124
126
  "password",
125
127
  "preventAppendedData",
126
128
  "selfExtractingArchive",
@@ -278,7 +280,7 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
278
280
  const payloadView = new DataView(payload.buffer);
279
281
  words.forEach((word, indexWord) => payloadView.setUint32(indexWord * 4, word, true));
280
282
  extraData = "<sfz-extra-data>" + base64Encode(deflateRaw(payload)) + "</sfz-extra-data>";
281
- if (options.preventAppendedData || extraData.length > MAX_APPENDED_DATA_LENGTH - pageContent.length - endTags.length - (options.embeddedImage ? PNG_IEND_LENGTH + PNG_CHUNK_CRC_LENGTH : 0)) {
283
+ if (options.preventAppendedData || extraData.length > getMaxAppendedDataLength(options) - pageContent.length - endTags.length - (options.embeddedImage ? PNG_IEND_LENGTH + PNG_CHUNK_CRC_LENGTH : 0)) {
282
284
  if (!options.extraDataSize) {
283
285
  options.preventAppendedData = true;
284
286
  options.extraDataSize = getReservationSize(extraData.length);
@@ -304,7 +306,7 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
304
306
  if (options.declareAppendedData) {
305
307
  const appendedDataLength = pageContent.length - data.length +
306
308
  (options.embeddedImage ? PNG_CHUNK_CRC_LENGTH + PNG_IEND_LENGTH : 0);
307
- if (appendedDataLength && appendedDataLength <= MAX_APPENDED_DATA_LENGTH && isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options)) {
309
+ if (appendedDataLength && appendedDataLength <= MAX_ZIP_COMMENT_LENGTH && isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options)) {
308
310
  new DataView(pageContent.buffer, pageContent.byteOffset).setUint16(zipDataEnd, appendedDataLength, true);
309
311
  }
310
312
  }
@@ -324,6 +326,10 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
324
326
  }
325
327
  }
326
328
 
329
+ function getMaxAppendedDataLength(options) {
330
+ return options.maxAppendedDataLength === undefined ? DEFAULT_MAX_APPENDED_DATA_LENGTH : options.maxAppendedDataLength;
331
+ }
332
+
327
333
  function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options) {
328
334
  if (options.extractDataFromPageTags && options.extractDataFromPageTags[0] == "<plaintext>") {
329
335
  return true;
@@ -747,11 +753,7 @@ async function getContent() {
747
753
  });
748
754
  return new Promise((resolve, reject) => {
749
755
  let aborted = false;
750
- if (location.protocol == "file:") {
751
- extractDataFromDocument();
752
- } else {
753
- getPageData();
754
- }
756
+ getPageData();
755
757
 
756
758
  async function extractDataFromDocument() {
757
759
  try {
@@ -242,39 +242,46 @@
242
242
  }
243
243
  return boundingRect;
244
244
  };
245
+ Element.prototype.getBoundingClientRect.toString = function () { return "function getBoundingClientRect() { [native code] }"; };
246
+ setFunctionName(Element.prototype.getBoundingClientRect, "getBoundingClientRect");
245
247
  }
246
248
  }
247
249
  if (!globalThis._singleFileImage) {
248
- const Image = globalThis.Image;
249
- globalThis._singleFileImage = globalThis.Image;
250
- globalThis.__defineGetter__("Image", function () {
251
- return function () {
252
- const image = new Image(...arguments);
253
- const result = new Image(...arguments);
254
- result.__defineSetter__("src", value => {
255
- image.src = value;
256
- document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT, { detail: image.src }));
257
- });
258
- result.__defineGetter__("src", () => image.src);
259
- result.__defineSetter__("srcset", value => {
260
- document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT));
261
- image.srcset = value;
262
- });
263
- result.__defineGetter__("srcset", () => image.srcset);
264
- result.__defineGetter__("height", () => image.height);
265
- result.__defineGetter__("width", () => image.width);
266
- result.__defineGetter__("naturalHeight", () => image.naturalHeight);
267
- result.__defineGetter__("naturalWidth", () => image.naturalWidth);
268
- if (image.decode) {
269
- result.__defineGetter__("decode", () => () => image.decode());
270
- }
271
- image.onload = image.onloadend = image.onerror = event => {
272
- document.dispatchEvent(new CustomEvent(IMAGE_LOADED_EVENT, { detail: image.src }));
273
- result.dispatchEvent(new Event(event.type, event));
274
- };
275
- return result;
250
+ const NativeImage = globalThis.Image;
251
+ globalThis._singleFileImage = NativeImage;
252
+ const ImageWrapper = function Image() {
253
+ const image = new NativeImage(...arguments);
254
+ const result = new NativeImage(...arguments);
255
+ result.__defineSetter__("src", value => {
256
+ image.src = value;
257
+ document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT, { detail: image.src }));
258
+ });
259
+ result.__defineGetter__("src", () => image.src);
260
+ result.__defineSetter__("srcset", value => {
261
+ document.dispatchEvent(new CustomEvent(LOAD_IMAGE_EVENT));
262
+ image.srcset = value;
263
+ });
264
+ result.__defineGetter__("srcset", () => image.srcset);
265
+ result.__defineGetter__("height", () => image.height);
266
+ result.__defineGetter__("width", () => image.width);
267
+ result.__defineGetter__("naturalHeight", () => image.naturalHeight);
268
+ result.__defineGetter__("naturalWidth", () => image.naturalWidth);
269
+ if (image.decode) {
270
+ const decode = function decode() { return image.decode(); };
271
+ decode.toString = function () { return "function decode() { [native code] }"; };
272
+ setFunctionName(decode, "decode");
273
+ result.__defineGetter__("decode", () => decode);
274
+ }
275
+ image.onload = image.onloadend = image.onerror = event => {
276
+ document.dispatchEvent(new CustomEvent(IMAGE_LOADED_EVENT, { detail: image.src }));
277
+ result.dispatchEvent(new Event(event.type, event));
276
278
  };
277
- });
279
+ return result;
280
+ };
281
+ ImageWrapper.prototype = NativeImage.prototype;
282
+ ImageWrapper.toString = function () { return "function Image() { [native code] }"; };
283
+ setFunctionName(ImageWrapper, "Image");
284
+ globalThis.__defineGetter__("Image", () => ImageWrapper);
278
285
  }
279
286
  const verticalZoomFactor = clientHeight / scrollHeight;
280
287
  const horizontalZoomFactor = clientWidth / scrollWidth;
@@ -371,9 +378,10 @@
371
378
 
372
379
  if (globalThis.CSS && globalThis.CSS.paintWorklet && globalThis.CSS.paintWorklet.addModule) {
373
380
  const addModule = globalThis.CSS.paintWorklet.addModule;
374
- globalThis.CSS.paintWorklet.addModule = function (moduleURL, options) {
381
+ globalThis.CSS.paintWorklet.addModule = function (moduleURL) {
375
382
  try {
376
383
  const result = addModule.apply(globalThis.CSS.paintWorklet, arguments);
384
+ const options = arguments[1];
377
385
  moduleURL = new URL(moduleURL, document.baseURI).href;
378
386
  document.dispatchEvent(new CustomEvent(NEW_WORKLET_EVENT, { detail: { moduleURL, options } }));
379
387
  return result;
@@ -382,6 +390,8 @@
382
390
  throw error;
383
391
  }
384
392
  };
393
+ globalThis.CSS.paintWorklet.addModule.toString = function () { return "function addModule() { [native code] }"; };
394
+ setFunctionName(globalThis.CSS.paintWorklet.addModule, "addModule");
385
395
  }
386
396
 
387
397
  if (globalThis.FontFace) {
@@ -1,5 +1,5 @@
1
1
  import "./dom-stub.js";
2
- import { makePageData, makeOptions, runProcess, mulberry32 } from "./common.js";
2
+ import { makePageData, makeOptions, runProcess, mulberry32, freezeDate } from "./common.js";
3
3
  import { ZipReader, ZipWriter, BlobReader } from "../../vendor/zip/zip.js";
4
4
 
5
5
  // the quote is there because the escaper the title shares with the table of contents encodes
@@ -239,6 +239,65 @@ function countIdentifiers(text) {
239
239
  check("a relocated archive declares no comment", view.getUint16(bytes.length - 2, true), 0);
240
240
  }
241
241
 
242
+ function makeNewlinePageData(seed, newlineCount) {
243
+ const pageData = makePageData(seed, 4 * 1024);
244
+ const rand = mulberry32(seed);
245
+ const newlines = ["\n", "\r", "\r\n"];
246
+ let content = "";
247
+ for (let index = 0; index < newlineCount; index++) {
248
+ content += newlines[(rand() * 3) | 0];
249
+ }
250
+ pageData.resources.stylesheets.push({ name: "newlines.txt", extension: ".txt", content, url: "https://example.com/newlines.txt" });
251
+ return pageData;
252
+ }
253
+
254
+ // libarchive gives up looking for the end of central directory record 16383 bytes from the end of
255
+ // the file, so the default budget is 16361 appended bytes: that window minus the 22-byte record.
256
+ // This fixture's payload lands between the default and the 65535-byte comment ceiling, which is
257
+ // the range the old budget kept appended and bsdtar could not open.
258
+ // The Date must be frozen: `archiveTime` (compression.js:155) puts an ISO timestamp in the
259
+ // archive, whose milliseconds move a few newline bytes in and out of the recovered range and
260
+ // therefore change the payload length build to build. The boundary checks below compare a budget
261
+ // against a run measured in an EARLIER build, so without freezing they are off by a few bytes one
262
+ // run in three. check-determinism.js asserts both halves of that
263
+ {
264
+ const restoreDate = freezeDate();
265
+ try {
266
+ const wide = makeOptions({ disableCompression: true, maxAppendedDataLength: 65535 });
267
+ const { bytes: wideBytes } = await runProcess(makeNewlinePageData(43, 80 * 1000), wide);
268
+ const wideTail = readAppendedData(wideBytes).trailing;
269
+ check("a wider budget keeps the payload appended", wide.extraDataSize, undefined);
270
+ check("the fixture overflows the default budget", wideTail > 16361, true);
271
+ check("the fixture fits the comment ceiling", wideTail <= 65535, true);
272
+
273
+ const byDefault = makeOptions({ disableCompression: true });
274
+ const { bytes } = await runProcess(makeNewlinePageData(43, 80 * 1000), byDefault);
275
+ check("the default budget relocates the payload", byDefault.extraDataSize > 0, true);
276
+ check("the default budget keeps the record in libarchive's window", readAppendedData(bytes).trailing <= 16361, true);
277
+
278
+ const fitting = makeOptions({ disableCompression: true, maxAppendedDataLength: wideTail });
279
+ await runProcess(makeNewlinePageData(43, 80 * 1000), fitting);
280
+ check("a budget matching the run to the byte keeps it appended", fitting.extraDataSize, undefined);
281
+
282
+ const tight = makeOptions({ disableCompression: true, maxAppendedDataLength: wideTail - 1 });
283
+ await runProcess(makeNewlinePageData(43, 80 * 1000), tight);
284
+ check("one byte below the run relocates it", tight.extraDataSize > 0, true);
285
+ } finally {
286
+ restoreDate();
287
+ }
288
+ }
289
+
290
+ // the budget and the comment ceiling are two different limits, and only the second is a property
291
+ // of the format. A budget raised past 65535 produces a run the comment-length field cannot hold,
292
+ // which setUint16 would write back modulo 65536: the writer leaves it undeclared instead
293
+ {
294
+ const options = makeOptions({ disableCompression: true, declareAppendedData: true, maxAppendedDataLength: Number.MAX_SAFE_INTEGER });
295
+ const { bytes } = await runProcess(makeNewlinePageData(44, 320 * 1000), options);
296
+ const { declared, trailing } = readAppendedData(bytes);
297
+ check("a run past the comment ceiling stays appended", trailing > 65535, true);
298
+ check("a run past the comment ceiling is left undeclared", declared, 0);
299
+ }
300
+
242
301
  {
243
302
  const options = makeOptions({ embeddedPdf: PDF });
244
303
  const pageData = makePageData(15, 4 * 1024);