single-file-core 1.5.125 → 1.5.127

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/core/index.js CHANGED
@@ -1361,6 +1361,11 @@ class Processor {
1361
1361
  }
1362
1362
  }
1363
1363
  }));
1364
+ frameElements.forEach(frameElement => {
1365
+ if (!frameElement.getAttribute("src") && !frameElement.getAttribute("srcdoc") && !frameElement.getAttribute("data")) {
1366
+ frameElement.removeAttribute("sandbox");
1367
+ }
1368
+ });
1364
1369
 
1365
1370
  async function initializeProcessor(frameData, frameElement, frameWindowId, batchRequest, options) {
1366
1371
  options.insertSingleFileComment = false;
@@ -243,9 +243,16 @@ class ProcessorHelperCommon {
243
243
  return serializeSrcset([Object.assign({}, srcsetValue, { url: resourceURL })]);
244
244
  }
245
245
  }));
246
- resourceElement.setAttribute("srcset", srcsetValues.filter(srcsetValue => srcsetValue).join(", "));
246
+ const newSrcset = srcsetValues.filter(srcsetValue => srcsetValue).join(", ");
247
+ if (newSrcset) {
248
+ resourceElement.setAttribute("srcset", newSrcset);
249
+ } else {
250
+ resourceElement.removeAttribute("srcset");
251
+ resourceElement.removeAttribute("sizes");
252
+ }
247
253
  } else {
248
- resourceElement.setAttribute("srcset", "");
254
+ resourceElement.removeAttribute("srcset");
255
+ resourceElement.removeAttribute("sizes");
249
256
  }
250
257
  }));
251
258
  }
@@ -267,6 +274,7 @@ class ProcessorHelperCommon {
267
274
  element.style.setProperty("background-size", style && style["background-size"] ? style["background-size"] : "100% 100%", "important");
268
275
  element.style.setProperty("background-origin", "content-box", "important");
269
276
  element.style.setProperty("background-repeat", "no-repeat", "important");
277
+ element.style.setProperty("background-attachment", "scroll", "important");
270
278
  }
271
279
 
272
280
  async getStylesheetContent(resourceURL, options) {
@@ -441,10 +449,9 @@ class ProcessorHelperCommon {
441
449
  await this.processFontFaceRules(ruleData.block.children, sheetIndex, fontsDetails.layers.get("layer-" + sheetIndex + "-" + layerIndex + "-" + layerText), fonts, fontTests, stats);
442
450
  layerIndex++;
443
451
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
444
- const key = this.getFontKey(ruleData);
445
- const fontInfo = fontsDetails.fonts.get(key);
452
+ const fontInfo = fontsDetails.fonts.get(ruleData);
446
453
  if (fontInfo) {
447
- const ruleKey = key + " " + this.getPropertyValue(ruleData, "src");
454
+ const ruleKey = this.getFontKey(ruleData) + " " + this.getPropertyValue(ruleData, "src");
448
455
  if (fontsDetails.emittedFonts.has(ruleKey)) {
449
456
  removedRules.push(cssRule);
450
457
  } else {
@@ -484,17 +491,14 @@ class ProcessorHelperCommon {
484
491
  layerIndex++;
485
492
  this.getFontsDetails(doc, ruleData.block.children, sheetIndex, fontsDetails);
486
493
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face" && ruleData.block && ruleData.block.children) {
487
- const fontKey = this.getFontKey(ruleData);
488
- let fontInfo = mediaFontsDetails.fonts.get(fontKey);
489
- if (!fontInfo) {
490
- fontInfo = [];
491
- mediaFontsDetails.fonts.set(fontKey, fontInfo);
492
- }
494
+ const fontInfo = [];
495
+ mediaFontsDetails.fonts.set(ruleData, fontInfo);
493
496
  const src = this.getPropertyValue(ruleData, "src");
494
497
  if (src) {
495
498
  const fontSources = src.match(REGEXP_URL_FUNCTION);
496
499
  if (fontSources) {
497
- fontSources.forEach(source => {
500
+ fontSources.forEach(fontSource => {
501
+ const source = fontSource.match(REGEXP_FONT_SRC)[1];
498
502
  if (fontInfo.includes(source)) {
499
503
  fontInfo.splice(fontInfo.indexOf(source), 1);
500
504
  }
package/core/util.js CHANGED
@@ -69,11 +69,6 @@ const EXPECTED_TYPES_MEDIA = ["font", "image", "video", "audio"];
69
69
  const URL = globalThis.URL;
70
70
  const DOMParser = globalThis.DOMParser;
71
71
  const Blob = globalThis.Blob;
72
- const fetch = (url, options) => {
73
- options.cache = "force-cache";
74
- options.referrerPolicy = "strict-origin-when-cross-origin";
75
- return globalThis.fetch(url, options);
76
- };
77
72
  const TextDecoder = globalThis.TextDecoder;
78
73
  const URLSearchParams = globalThis.URLSearchParams;
79
74
 
@@ -83,8 +78,8 @@ export {
83
78
 
84
79
  function getInstance(utilOptions) {
85
80
  utilOptions = utilOptions || {};
86
- utilOptions.fetch = utilOptions.fetch || fetch;
87
- utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch || fetch;
81
+ utilOptions.fetch = utilOptions.fetch || ((url, options) => globalThis.fetch(url, { ...options, cache: "force-cache", referrerPolicy: "strict-origin-when-cross-origin" }));
82
+ utilOptions.frameFetch = utilOptions.frameFetch || utilOptions.fetch;
88
83
  return {
89
84
  getDoctypeString,
90
85
  getFilenameExtension(resourceURL, replacedCharacters, replacementCharacter, replacementCharacters) {
@@ -264,7 +259,7 @@ function getInstance(utilOptions) {
264
259
  // eslint-disable-next-line no-unused-vars
265
260
  } catch (error) {
266
261
  response = await Promise.race([
267
- fetchResource(resourceURL, { headers: { accept } }),
262
+ fetchResource(resourceURL, { referrer: options.resourceReferrer, headers: { accept } }),
268
263
  networkTimeoutPromise
269
264
  ]);
270
265
  }
@@ -136,7 +136,7 @@ Three consequences shape everything below:
136
136
  | **region** | A byte range with a single producer, named in §3. Regions are the units the rest of this document reasons about; a region can appear in several pieces — `html-prologue` resumes after the embedded PDF document in the PDF variants, and after the `tEXt "ZIP"` chunk header in the PNG ones, so with all four faces it comes in three. |
137
137
  | **universal mode** | The variant whose HTML face can extract the archive from the *parsed page text*, the text and comment nodes the HTML parser produced, and therefore needs no access to its own raw bytes. Named "universal" because it works from any location, including the `file:` protocol. |
138
138
  | **wrapper tag** | The HTML construct that hides a binary region from the HTML parser, `<!--`…`-->` by default (§5.1). |
139
- | **appended data** | Bytes after the ZIP End Of Central Directory record. Readers tolerate them within the window their EOCD scan already covers: 65557 bytes from the end of the file (the 22-byte record plus the 65535-byte maximum comment length); "the 64 KB window" refers to this. It may be left undeclared or declared as the archive comment; both forms are valid ZIP and readers MUST accept both (§4.2). The recovery payload can be computed before that choice is made because it stops two bytes short of the record, excluding its comment-length field (see *recovered range* below). |
139
+ | **appended data** | Bytes after the ZIP End Of Central Directory record. A reader tolerates them as far back as its EOCD scan reaches, and how far that is varies by an order of magnitude: 65557 bytes from the end of the file for Python `zipfile` (the 22-byte record plus the 65535-byte maximum comment length), but 16383 for libarchive and 32768 for perl `Archive::Zip` (§8.1). No reader's window is guaranteed, so a writer keeps its own narrower budget (§5.2). The format's one hard limit is the 65535-byte comment field, and it binds only a run the writer declares (§4.2). It may be left undeclared or declared as the archive comment; both forms are valid ZIP and readers MUST accept both (§4.2). The recovery payload can be computed before that choice is made because it stops two bytes short of the record, excluding its comment-length field (see *recovered range* below). |
140
140
  | **ZIP region** | The contiguous byte range holding the archive proper: from the first local file header the ZIP writer emitted through the last byte of the End Of Central Directory record. It spans the `zip-entries`, `pdf-central-record` (when present) and `central-directory · eocd` blocks of §3, and in the HTML variants it is the content of the last wrapper, exactly so on the element rungs and preceded by the `sfz-data` identifier on the comment rung, which the extractor steps over. It does **not** include `pdf-local-header` or the PDF document, which sit earlier in the file. |
141
141
  | **archive** | The *logical* ZIP file: the set of entries the central directory describes, wherever their bytes lie. This is distinct from the ZIP region above, which is a contiguous byte range. Every entry but one has its bytes inside the region; `page.pdf` is the deliberate exception, an entry of the archive whose local header and data sit before the region (§4.2). "Archive" in this document always means the logical file, "ZIP region" always the byte range, and the two differ only in the PDF-with-HTML variants. |
142
142
  | **recovered range** | What the universal extractor reproduces (§4.5): the ZIP region minus its last two bytes, the comment-length field of the End Of Central Directory record. That field is the one part of the record whose value depends on what follows the region, so leaving it out is what lets a writer decide the appended-data form after the recovery payload is final (§4.2). The extractor supplies the two bytes itself, as zeroes — the recovered range carries no comment. |
@@ -174,8 +174,9 @@ every acquisition path including the ones that read raw bytes and could return i
174
174
  HTML face, so the option does not apply. The *Specimen* column names the measured
175
175
  reference files this document cites; §8 records how to regenerate them.
176
176
 
177
- Other writer options shape the file without adding a face: `preventAppendedData` and
178
- `declareAppendedData` (§4.2, §5.2), `includeBOM` (§3.1), `insertTextBody` (§4.6),
177
+ Other writer options shape the file without adding a face: `preventAppendedData`,
178
+ `declareAppendedData` and `maxAppendedDataLength` (§4.2, §5.2), `includeBOM` (§3.1),
179
+ `insertTextBody` (§4.6),
179
180
  `password` (§5.6), `createRootDirectory` (§7.1), and the head-element switches
180
181
  `insertCanonicalLink`, `insertMetaNoIndex` and `insertMetaCSP` (§3.1).
181
182
 
@@ -328,7 +329,8 @@ the same way: only a universal file carries an `<sfz-extra-data>` element.
328
329
 
329
330
  Unless a row states otherwise, the layouts below are measured from specimen files
330
331
  saved from `example.com` (the generation commands are in §8). The relocated row covers
331
- two cases with one layout, `preventAppendedData` and a payload over 64 KB: the first is
332
+ two cases with one layout, `preventAppendedData` and a payload over the appended-data
333
+ budget (§5.2): the first is
332
334
  measured on the relocated specimen, the second derived from the writer rules, because
333
335
  such a payload requires an archive too large for a readable specimen. The figure below shows
334
336
  the regions and their order; the glossary of §3.1 is the normative list, and it states
@@ -353,9 +355,9 @@ face adds, then the regions the PNG face adds.
353
355
  | `<!--` / `-->` | HTML | HTML face | The wrapper tag pair hiding a binary region from the HTML parser — comment tags by default, another pair when the hidden bytes defeat them — which `-->` is only the commonest way to do, the full test being `<!--`, `--!>`, a trailing `<!-` and, for the PNG payload, a leading `>` or `->` (§5.1). Drawn at each opening and closing position. The close tag is absent whenever the recovery payload is relocated (§5.2): under `preventAppendedData`, when the payload outgrows the appended-data budget, or on the `<plaintext>` wrapper which cannot close. No markup then follows the archive and the wrapper runs to end-of-file. That does not mean the file ends at the EOCD — the PNG face's tail still follows, inside the wrapper, where it parses as text (§5.1). |
354
356
  | `zip-entries` | ZIP | always | The archive's local file headers and entry data, written by the ZIP writer. The central directory of an archive written by the reference writer lists `index.html` (the page) first, then `manifest.json` (a JSON description of the archive: original URL, title, save time, resource-to-URL map — informative; the page displays without it), then the resources; the *physical* order of the local headers inside the region is not guaranteed to match, and readers MUST NOT rely on either order — entries are addressed by name (§7.1). |
355
357
  | `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
356
- | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the 64 KB appended-data window or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
358
+ | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the appended-data budget (§5.2) or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
357
359
  | `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
358
- | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag set as on every other entry — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
360
+ | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag left clear as on every other ASCII name — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
359
361
  | `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
360
362
  | `pdf-central-record` | ZIP | PDF face with HTML | The central-directory record for `page.pdf`, injected *before* the writer's own central directory. The start of the central directory is the one place a record can be added without moving any offset the writer already committed, and it makes `page.pdf` the first entry ZIP tools list (§6). |
361
363
  | `png-signature · IHDR` | PNG | PNG face | The 8-byte PNG signature and the `IHDR` chunk declaring the source image's dimensions — the first 33 bytes of the file. |
@@ -366,7 +368,7 @@ face adds, then the regions the PNG face adds.
366
368
  | `crc · IEND` | PNG | PNG face | The `tEXt "ZIP"` chunk's CRC, computed once the archive bytes are final (§6), followed by the empty `IEND` chunk — the last bytes of the file (PNG requires `IEND` to end the stream, which is why the PNG variants drop the end tags). |
367
369
 
368
370
  The reader-by-reader interpretation of these regions is §4; the mechanics that keep
369
- them from colliding (wrapper-tag selection, checksums, offsets, the 64 KB budget) are
371
+ them from colliding (wrapper-tag selection, checksums, offsets, the appended-data budget) are
370
372
  §5.
371
373
 
372
374
  ## 4. Reader lenses
@@ -447,10 +449,12 @@ path the plain variant's error message describes.
447
449
  ### 4.2 The ZIP reader
448
450
 
449
451
  The ZIP face is read from the end. A reader locates the End Of Central Directory
450
- record by scanning backward from end-of-file; the format guarantees it lies within
451
- the window every reader must already scan to support archive comments (§1.3,
452
- *appended data*), because everything after it — wrapper close tag, extra-data, end
453
- tags, PNG tail — fits the appended-data budget (§5.2). Accepting *undeclared* bytes
452
+ record by scanning backward from end-of-file, and how far back it scans is the one
453
+ reader property the format cannot assume (§1.3, *appended data*). Everything the
454
+ writer emits after the record — wrapper close tag, extra-data, end tags, PNG tail —
455
+ fits the appended-data budget of §5.2, and the reference writer sizes that budget to
456
+ the narrowest scan measured in §8.1, so the record stays reachable for every reader
457
+ listed there. Accepting *undeclared* bytes
454
458
  in that window is itself a customary tolerance (§1.1): the ZIP specification
455
459
  documents the comment, not trailing junk. From the EOCD
456
460
  the reader jumps to the central directory and reads only what it references;
@@ -510,6 +514,14 @@ readers accept at all: `java.util.zip`, and therefore Android and most JVM tooli
510
514
  rejects an archive with undeclared trailing bytes outright (§8.1). A writer SHOULD
511
515
  offer both and default to raw.
512
516
 
517
+ The declared form carries a ceiling the raw form does not. The comment length is a
518
+ 16-bit field, so a run longer than 65535 bytes cannot be declared at all. A writer
519
+ whose appended-data budget (§5.2) is raised past that ceiling MUST leave such a run
520
+ undeclared rather than write its length back modulo 65536, and readers that accept
521
+ only the declared form then reject the file with no diagnostic. The budget and the
522
+ ceiling are two separate limits, and a writer that exposes the first as an option
523
+ SHOULD say so where it documents it.
524
+
513
525
  Neither form constrains the other faces, and universal mode supports both, because
514
526
  the recovery payload describes the recovered range rather than the whole region: the
515
527
  comment-length field is excluded (§1.3), so its value can be decided after the
@@ -1011,16 +1023,32 @@ re-check the field.
1011
1023
 
1012
1024
  ### 5.2 The appended-data budget
1013
1025
 
1014
- Everything the writer emits after the EOCD record MUST fit in 65535 bytes — the
1015
- maximum length a ZIP archive comment may declare, and therefore the distance beyond
1016
- the record that every reader's backward scan already covers (§1.3). The appended run is:
1026
+ The run the writer emits after the EOCD record has two limits, and only one of them
1027
+ comes from the format. A run *declared* as the archive comment MUST fit in 65535
1028
+ bytes, the largest value a comment-length field can hold (§4.2). A run left *raw* has
1029
+ no format limit at all: the bytes are outside the archive, and nothing in ZIP bounds
1030
+ them. What bounds both in practice is the reader. Locating the EOCD record means
1031
+ scanning backward from end-of-file, and the searches measured in §8.1 stop at 16383
1032
+ bytes for libarchive, 32768 for perl `Archive::Zip` and 65557 for Python `zipfile`, so
1033
+ a run sized to the comment ceiling is already invisible to the narrowest of them. A
1034
+ writer therefore keeps a *budget*, sized to the readers it means to satisfy rather
1035
+ than to the format. The appended run is:
1017
1036
 
1018
1037
  ```
1019
1038
  wrapper close tag + extra-data element + end tags + (PNG face: 4-byte chunk CRC + 12-byte IEND)
1020
1039
  ```
1021
1040
 
1022
- and the writer compares its total against 65535 before committing to it. The EOCD
1023
- record's own 22 bytes sit inside the window too, giving the 65557-byte figure of §1.3.
1041
+ and the writer compares its total against that budget before committing to it. The
1042
+ EOCD record's own 22 bytes sit inside a reader's window as well, which is what turns a
1043
+ 65535-byte run into the 65557 bytes of §1.3 and the reference writer's budget into
1044
+ libarchive's 16383.
1045
+
1046
+ The reference writer exposes the budget as `maxAppendedDataLength` and defaults it to
1047
+ 16361 bytes: libarchive's window less the 22 bytes of the record, which is the largest
1048
+ run behind which every reader of §8.1 still finds the record. A writer MAY choose
1049
+ another value. Raising it above 65535 leaves the run undeclarable: it is emitted, and
1050
+ it is still valid ZIP, but no comment length can cover it, and §4.2 says what that
1051
+ costs.
1024
1052
 
1025
1053
  Only the extra-data element can outgrow the budget: it carries one 2-bit code per
1026
1054
  newline sequence in the recovered range — the ZIP region without its comment-length
@@ -1028,9 +1056,9 @@ field (§4.5), CR LF counting once, for two bytes (§5.5) — so it
1028
1056
  grows with the archive. Newline bytes
1029
1057
  occur at their natural density in compressed and STOREd binary data — about two in
1030
1058
  every 256 bytes — and the codes are compressed and base64-encoded, which measures at
1031
- one byte of element per 650 bytes of archive at scale (§8). The budget is therefore
1032
- exhausted at an archive of roughly 40 MB, so the relocated placement is rare in
1033
- practice. That ratio is the large-archive limit and must not be used to size a
1059
+ one byte of element per 650 bytes of archive at scale (§8). The default budget is
1060
+ therefore exhausted at an archive of roughly 10 MB, and the 65535-byte ceiling at
1061
+ roughly 40 MB, so the relocated placement is uncommon in practice. That ratio is the large-archive limit and must not be used to size a
1034
1062
  particular file: deflate's overhead is a fixed cost spread over a growing payload, so
1035
1063
  small archives are far less efficient. Measured on exact byte counts, a 6099-byte region
1036
1064
  needs 69 bytes of element — a ratio of 88 — and a 74057-byte region needs 189, a ratio
@@ -1051,6 +1079,14 @@ trailing bytes open it (§8.1). The parser closes the open
1051
1079
  comment or element at end of file, and `</body></html>` are implied, so the page
1052
1080
  renders the same.
1053
1081
 
1082
+ Relocation is not a move at constant size, and it can end either way. Two effects pull
1083
+ against each other: the room the writer sets aside, which in the reference writer is
1084
+ `Math.ceil(length * 1.01) + 32` bytes — a percentage of the payload plus a constant, so
1085
+ the constant dominates a small payload and the percentage a large one — and the 17 bytes
1086
+ of wrapper terminator and end tags it stops emitting. Measured on three files the net ran
1087
+ from 9 bytes saved to 190 bytes spent, so a writer sizing a file should quote that range
1088
+ rather than a single figure.
1089
+
1054
1090
  ### 5.3 Offset bookkeeping
1055
1091
 
1056
1092
  Three coordinate systems coexist in one file, and the format's job is to keep each
@@ -1255,18 +1291,26 @@ How a name is encoded is ZIP's own business, not this format's: bit 11 of the ge
1255
1291
  purpose bit flag selects UTF-8, and its absence selects the legacy code page. This
1256
1292
  document adds two requirements to that and specifies nothing else about it.
1257
1293
 
1258
- **A writer MUST set bit 11 on every entry**, not only on the entries whose names need
1259
- it. The two encodings agree over printable ASCII, so setting it unconditionally costs
1260
- nothing, and it means no name in the archive is decoded through the legacy path at all.
1294
+ **A writer MUST set bit 11 whenever a name or a comment needs it**, and the rule for
1295
+ when it does is ZIP's, not this format's: an encoded name or comment holding a byte
1296
+ outside printable ASCII needs it, one holding only printable ASCII does not, since the
1297
+ two encodings agree there. Control characters count as needing it, the legacy code page
1298
+ mapping them to graphic characters rather than to themselves. Setting it on names that
1299
+ do not need it is allowed and used to be required here; it was dropped because readers
1300
+ disagree about the flag more than they disagree about ASCII, so the safest name is the
1301
+ one that does not exercise the question. A writer MUST NOT set it on a name it then
1302
+ encodes in the legacy code page, which is the one combination that is simply wrong.
1261
1303
 
1262
1304
  **A reader MUST honor the flag** rather than assume one encoding, and MUST expect to
1263
- meet a clear one: the hand-built `page.pdf` records (§3.1, §6) are the only ones the
1264
- reference writer does not produce through its ZIP writer, and an archive may carry them
1265
- with no flag set at all. That single entry is then decoded as legacy while every other
1266
- name in the same file is UTF-8.
1267
- Its name is ASCII, where the two encodings agree, so a correct reader sees `page.pdf`
1268
- either way — but a reader that hardcodes UTF-8 on the strength of the other entries has
1269
- not covered it.
1305
+ meet a clear one — which, in an archive from the reference writer, is most of them:
1306
+ that writer percent-encodes every name it produces, so every name is printable ASCII
1307
+ and carries no flag, while an entry comment holding the original URL of a resource can
1308
+ carry one when that URL is not ASCII. The hand-built `page.pdf` records (§3.1, §6) are
1309
+ the only ones the reference writer does not produce through its ZIP writer, and they
1310
+ follow the same rule: `page.pdf` is ASCII, so they carry no flag either, and no entry
1311
+ in the archive is decoded differently from the rest. Archives written before this rule
1312
+ was relaxed carry the flag on every entry instead. Both decode identically, which is
1313
+ the point, but a reader that hardcodes either answer meets the other one eventually.
1270
1314
 
1271
1315
  A name is not a path. §7.3's rule that entry names are untrusted applies to the decoded
1272
1316
  name, and decoding is the step before that check, not a substitute for it.
@@ -1363,7 +1407,7 @@ pages can stop at the first row; the files it produces are accepted by every rea
1363
1407
  wrapper start tag chosen for the PDF payload (§5.1), the hand-built `page.pdf`
1364
1408
  local file header, the PDF document, the wrapper end tag, and record the local
1365
1409
  header's absolute position; then resume the prologue. The reference writer's
1366
- header declares version 2.0, the language encoding flag alone, method STORE, the
1410
+ header declares version 2.0, no general purpose bit flag, method STORE, the
1367
1411
  build's modification date in DOS form, the precomputed CRC-32, the document's
1368
1412
  length as both sizes, and no extra field; its central record adds a Unix
1369
1413
  "made by" version and external attributes of a regular file, mode 0644.
@@ -1467,7 +1511,7 @@ terminate because each of them advances a monotone quantity:
1467
1511
  There is no converse of the second: a pass that reserved room never discards it,
1468
1512
  even when the relocated payload would have fit the appended window. Relocation moves
1469
1513
  the archive, which changes the offsets, which changes the payload that made the
1470
- relocation necessary, so a payload lying on the 65535-byte boundary can be too large
1514
+ relocation necessary, so a payload lying on the budget boundary can be too large
1471
1515
  appended and small enough relocated, and a writer that dropped the reservation could
1472
1516
  rebuild the two placements forever. Relocation is therefore final (§5.2), and the file
1473
1517
  keeps at most the reservation's own margin of dead padding.
@@ -1622,7 +1666,7 @@ only if it affects the bytes the page is built from:
1622
1666
  | An entry's CRC-32 or AES authentication code does not match | **SHOULD** fail for that entry, and MUST NOT present a page rebuilt from it as intact |
1623
1667
  | `page.pdf` was reconstructed from the parsed page and its CRC-32 does not match | **MUST** discard the reconstruction (§4.5). The bytes are a guess about newlines the recovery payload does not describe, and the checksum is the only thing that tests it — unlike the row above, there is no read to have gone wrong, only an inference |
1624
1668
  | Bytes outside the archive proper — before the first local file header, after the EOCD record, or between an entry's data and the next header | **MUST** tolerate: they are the other faces (§7.1). The gap in the middle is not hypothetical: with the PDF face the bootstrap lies between `page.pdf`'s data and the ZIP region |
1625
- | The appended run exceeds the 65535-byte budget (§5.2) | Not a reader's problem: if the EOCD record was found, the archive is readable. Readers MAY warn |
1669
+ | The appended run exceeds the 65535-byte ceiling (§5.2) | Not a reader's problem: if the EOCD record was found, the archive is readable. Readers MAY warn |
1626
1670
  | A `tEXt` chunk CRC does not match, or a chunk holds bytes PNG does not permit (§4.4) | Irrelevant to extraction; a reader of the archive MAY ignore both |
1627
1671
  | `page.pdf` is present but its data does not begin with `%PDF-` | Not an error. The entry is data like any other |
1628
1672
  | `index.html` is present without `manifest.json` | **MUST** still extract (§7.1) |
@@ -1662,11 +1706,20 @@ and is class C.
1662
1706
  | Info-ZIP `unzip`, `zipinfo` | ✔ | ✔ | ✔ | Lists and extracts every variant. AES entries are skipped — `need PK compat. v5.1 (can do v4.5)` — a limitation of the tool, not of the file; `page.pdf` still extracts because it is never encrypted |
1663
1707
  | Python `zipfile` | ✔ | ✔ | ✔ | Lists and extracts every variant |
1664
1708
  | 7-Zip (`7zz`) | ✔ | ✔ | ✔ | Lists and extracts every variant, AES included |
1665
- | libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant |
1709
+ | libarchive `bsdtar`, seekable input | ✔ | ✔ | ✔ | Lists and extracts every variant. Its EOCD scan is the narrowest measured, so a class-C file whose appended run pushes the record past 16383 bytes from the end is rejected with `Unrecognized archive format`; §5.2's default budget is sized to this window, and the ✔ holds for files that respect it |
1666
1710
  | libarchive `bsdtar`, piped input | ✔ | ✘ | ✘ | `Unrecognized archive format` — the forward-only case of §1.2, measured |
1667
1711
  | Java `java.util.zip` (`jar tf`) | ✔ | ✔ | ✘ | `zip END header not found` whenever bytes follow the EOCD undeclared. Declaring them as the archive comment makes the same file open, measured on every class-C variant (§4.2) |
1668
1712
  | macOS `ditto -x -k` | ✔ | ✘ | ✘ | `Couldn't read PKZip signature` — requires a local file header at offset 0, so prepended data alone defeats it |
1669
1713
 
1714
+ The backward scans behind the class-C column differ by an order of magnitude, and they
1715
+ are what §5.2's budget is sized against. Measured by padding a working archive until
1716
+ the record fell out of reach, the largest distance from end-of-file at which each
1717
+ reader still finds the EOCD record is: libarchive 16383, perl `Archive::Zip` 32768,
1718
+ Python `zipfile` 65557, zip.js 65536, Info-ZIP `unzip` 68000, and 7-Zip beyond 1 MiB,
1719
+ which scans the whole file. libarchive binds, and its window less the 22-byte record
1720
+ is the 16361-byte default budget of §5.2. macOS `ditto` is not on this axis at all: it
1721
+ requires a local file header at offset 0 whatever the tail looks like.
1722
+
1670
1723
  The cost of the declared form was measured on the same tools: it is a display cost, not
1671
1724
  a compatibility one. An archive whose trailing bytes are declared as the comment has
1672
1725
  them printed back on ordinary listings; `unzip -l` reproduces the whole run — in
@@ -1741,7 +1794,7 @@ specimen without a network.
1741
1794
  | 123006 | `<sfz-extra-data>` … `</sfz-extra-data>` | recovery payload, appended placement (§5.2); 24 base64 characters for this archive |
1742
1795
  | 123063 | `</body></html>` | end tags; end of file at 123077 |
1743
1796
 
1744
- The appended run is 74 bytes, well inside the 65535-byte budget (§5.2). The ZIP region
1797
+ The appended run is 74 bytes, well inside the 16361-byte default budget (§5.2). The ZIP region
1745
1798
  is the 998 bytes from 122005 to 123003; the universal extractor reproduces the first 996
1746
1799
  of them and supplies the last two itself (§1.3).
1747
1800
 
@@ -1775,7 +1828,7 @@ These specimens are deliberately small, and a reader tested only against them is
1775
1828
  undertested: they are all flat archives of two or three entries. None
1776
1829
  exercises a root directory, `frames/<n>/` nesting, a second `index.html`, a `data:`-URL
1777
1830
  entry comment, the optional text body (§4.6), a UTF-8 BOM, zip64
1778
- (§5.7), a payload past the 64 KB budget, or a relocated reservation with padding left
1831
+ (§5.7), a payload past the appended-data budget, or a relocated reservation with padding left
1779
1832
  in it. Two omissions matter more than the rest, because they are the parts of §5.1 a
1780
1833
  writer is most likely to get wrong: no specimen defeats a rung by its **start**
1781
1834
  pattern, and none defeats one with an **upper-case** pattern. A writer that tested only
@@ -1847,6 +1900,8 @@ predicts.
1847
1900
  | August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
1848
1901
  | August 2026 | Core 1.5.119: the inlined ZIP library is built ASCII-only, and §2.1 now requires it of any bootstrap. Its CP437 table had been emitted as literal characters, which the page re-decoded as windows-1252, growing the table from 256 entries to 508 and shifting every lookup by 60 — so the one entry read without the UTF-8 flag, `page.pdf`, came back mangled and no archive with a PDF face extracted in any engine (§5.8) |
1849
1902
  | August 2026 | Core 1.5.120: the hand-built `page.pdf` records set the language encoding flag, like every entry the ZIP writer produces (§5.8). Its name is ASCII, so no decoded name changes; what changes is that no entry in an archive is read through CP437 any more, closing the path the 1.5.119 defect surfaced on |
1903
+ | September 2026 | §5.8 no longer requires bit 11 on every entry, deferring to ZIP's own rule: the flag is set when a name or a comment holds a byte outside printable ASCII, and left clear otherwise, because readers disagree about the flag more than they disagree about ASCII. The reference writer's names are all percent-encoded, so in practice none of them carries it now, and the hand-built `page.pdf` records follow the writer instead of overriding it — reversing the 1.5.120 row below, whose reason was that `page.pdf` would otherwise be the only entry read through the legacy path. It no longer is: every name in the archive takes the same path again, the other one |
1904
+ | September 2026 | Core 1.5.126: the appended-data budget becomes the `maxAppendedDataLength` writer option and its default drops from 65535 to 16361 bytes, so the EOCD record stays inside libarchive's scan and `bsdtar` opens archives it used to reject (§5.2, §8.1). The 65535-byte comment ceiling is now a separate limit, stated in §4.2: a budget raised past it produces a run that cannot be declared |
1850
1905
 
1851
1906
  This document was itself revised in August 2026, against core 1.5.108, after several
1852
1907
  independent reviews. One of them was a reader built from this specification alone, with
@@ -83,15 +83,15 @@ function getSourceSrcData(sources) {
83
83
  function setSrc(srcData, imgElement, pictureElement) {
84
84
  if (srcData.src) {
85
85
  imgElement.setAttribute("src", srcData.src);
86
- imgElement.setAttribute("srcset", "");
87
- imgElement.setAttribute("sizes", "");
86
+ imgElement.removeAttribute("srcset");
87
+ imgElement.removeAttribute("sizes");
88
88
  } else {
89
89
  imgElement.setAttribute("src", EMPTY_RESOURCE);
90
90
  if (srcData.srcset) {
91
91
  imgElement.setAttribute("srcset", srcData.srcset);
92
92
  } else {
93
- imgElement.setAttribute("srcset", "");
94
- imgElement.setAttribute("sizes", "");
93
+ imgElement.removeAttribute("srcset");
94
+ imgElement.removeAttribute("sizes");
95
95
  }
96
96
  }
97
97
  if (pictureElement) {
package/package.json CHANGED
@@ -1,11 +1,11 @@
1
1
  {
2
2
  "name": "single-file-core",
3
- "version": "1.5.125",
3
+ "version": "1.5.127",
4
4
  "description": "SingleFile Core",
5
5
  "author": "Gildas Lormeau",
6
6
  "license": "AGPL-3.0-or-later",
7
7
  "scripts": {
8
- "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
8
+ "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/font-face-composite.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
9
9
  "bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
10
10
  "bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
11
11
  "bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
@@ -100,16 +100,12 @@ async function createPagesArchive(pages, options) {
100
100
  const zipReader = new ZipReader(new Uint8ArrayReader(await pages[pageIndex].getData()));
101
101
  for (const entry of await zipReader.getEntries()) {
102
102
  const filename = pagePath + entry.filename;
103
- const rawData = await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
103
+ const rawData = entry.directory ? undefined : await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
104
104
  const canonicalFilename = writtenEntries && findDuplicate(writtenEntries, filename, entry, rawData);
105
105
  if (canonicalFilename === undefined) {
106
- await zipWriter.add(filename, new Uint8ArrayReader(rawData), {
106
+ await zipWriter.add(filename, entry.directory ? null : new Uint8ArrayReader(rawData), {
107
107
  passThrough: true,
108
- compressionMethod: entry.compressionMethod,
109
- uncompressedSize: entry.uncompressedSize,
110
- crc32: entry.crc32,
111
- comment: entry.comment,
112
- lastModDate: entry.lastModDate
108
+ entry
113
109
  });
114
110
  } else {
115
111
  aliases[filename] = canonicalFilename;