single-file-core 1.5.126 → 1.5.127

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -449,12 +449,17 @@ class ProcessorHelperCommon {
449
449
  await this.processFontFaceRules(ruleData.block.children, sheetIndex, fontsDetails.layers.get("layer-" + sheetIndex + "-" + layerIndex + "-" + layerText), fonts, fontTests, stats);
450
450
  layerIndex++;
451
451
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
452
- const key = this.getFontKey(ruleData);
453
- const fontInfo = fontsDetails.fonts.get(key);
454
- if (fontInfo && fontsDetails.lastRules.get(key) == ruleData) {
455
- const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
456
- if (!keptRule) {
452
+ const fontInfo = fontsDetails.fonts.get(ruleData);
453
+ if (fontInfo) {
454
+ const ruleKey = this.getFontKey(ruleData) + " " + this.getPropertyValue(ruleData, "src");
455
+ if (fontsDetails.emittedFonts.has(ruleKey)) {
457
456
  removedRules.push(cssRule);
457
+ } else {
458
+ fontsDetails.emittedFonts.add(ruleKey);
459
+ const keptRule = await this.processFontFaceRule(ruleData, fontInfo, fonts, fontTests, stats);
460
+ if (!keptRule) {
461
+ removedRules.push(cssRule);
462
+ }
458
463
  }
459
464
  } else {
460
465
  removedRules.push(cssRule);
@@ -486,28 +491,19 @@ class ProcessorHelperCommon {
486
491
  layerIndex++;
487
492
  this.getFontsDetails(doc, ruleData.block.children, sheetIndex, fontsDetails);
488
493
  } else if (ruleData.type == "Atrule" && ruleData.name == "font-face" && ruleData.block && ruleData.block.children) {
489
- const fontKey = this.getFontKey(ruleData);
490
- let fontInfo = mediaFontsDetails.fonts.get(fontKey);
491
- if (!fontInfo) {
492
- fontInfo = [];
493
- mediaFontsDetails.fonts.set(fontKey, fontInfo);
494
- }
495
- mediaFontsDetails.lastRules.set(fontKey, ruleData);
494
+ const fontInfo = [];
495
+ mediaFontsDetails.fonts.set(ruleData, fontInfo);
496
496
  const src = this.getPropertyValue(ruleData, "src");
497
497
  if (src) {
498
498
  const fontSources = src.match(REGEXP_URL_FUNCTION);
499
499
  if (fontSources) {
500
- const ruleSources = [];
501
- fontSources.forEach(source => {
500
+ fontSources.forEach(fontSource => {
501
+ const source = fontSource.match(REGEXP_FONT_SRC)[1];
502
502
  if (fontInfo.includes(source)) {
503
503
  fontInfo.splice(fontInfo.indexOf(source), 1);
504
504
  }
505
- if (ruleSources.includes(source)) {
506
- ruleSources.splice(ruleSources.indexOf(source), 1);
507
- }
508
- ruleSources.unshift(source);
505
+ fontInfo.unshift(source);
509
506
  });
510
- ruleSources.forEach(source => fontInfo.push(source));
511
507
  }
512
508
  }
513
509
  }
@@ -520,7 +516,7 @@ class ProcessorHelperCommon {
520
516
  medias: new Map(),
521
517
  supports: new Map(),
522
518
  layers: new Map(),
523
- lastRules: new Map()
519
+ emittedFonts: new Set()
524
520
  };
525
521
  }
526
522
 
@@ -357,7 +357,7 @@ face adds, then the regions the PNG face adds.
357
357
  | `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
358
358
  | `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the appended-data budget (§5.2) or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
359
359
  | `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
360
- | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag set as on every other entry — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
360
+ | `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag left clear as on every other ASCII name — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
361
361
  | `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
362
362
  | `pdf-central-record` | ZIP | PDF face with HTML | The central-directory record for `page.pdf`, injected *before* the writer's own central directory. The start of the central directory is the one place a record can be added without moving any offset the writer already committed, and it makes `page.pdf` the first entry ZIP tools list (§6). |
363
363
  | `png-signature · IHDR` | PNG | PNG face | The 8-byte PNG signature and the `IHDR` chunk declaring the source image's dimensions — the first 33 bytes of the file. |
@@ -1291,18 +1291,26 @@ How a name is encoded is ZIP's own business, not this format's: bit 11 of the ge
1291
1291
  purpose bit flag selects UTF-8, and its absence selects the legacy code page. This
1292
1292
  document adds two requirements to that and specifies nothing else about it.
1293
1293
 
1294
- **A writer MUST set bit 11 on every entry**, not only on the entries whose names need
1295
- it. The two encodings agree over printable ASCII, so setting it unconditionally costs
1296
- nothing, and it means no name in the archive is decoded through the legacy path at all.
1294
+ **A writer MUST set bit 11 whenever a name or a comment needs it**, and the rule for
1295
+ when it does is ZIP's, not this format's: an encoded name or comment holding a byte
1296
+ outside printable ASCII needs it, one holding only printable ASCII does not, since the
1297
+ two encodings agree there. Control characters count as needing it, the legacy code page
1298
+ mapping them to graphic characters rather than to themselves. Setting it on names that
1299
+ do not need it is allowed and used to be required here; it was dropped because readers
1300
+ disagree about the flag more than they disagree about ASCII, so the safest name is the
1301
+ one that does not exercise the question. A writer MUST NOT set it on a name it then
1302
+ encodes in the legacy code page, which is the one combination that is simply wrong.
1297
1303
 
1298
1304
  **A reader MUST honor the flag** rather than assume one encoding, and MUST expect to
1299
- meet a clear one: the hand-built `page.pdf` records (§3.1, §6) are the only ones the
1300
- reference writer does not produce through its ZIP writer, and an archive may carry them
1301
- with no flag set at all. That single entry is then decoded as legacy while every other
1302
- name in the same file is UTF-8.
1303
- Its name is ASCII, where the two encodings agree, so a correct reader sees `page.pdf`
1304
- either way — but a reader that hardcodes UTF-8 on the strength of the other entries has
1305
- not covered it.
1305
+ meet a clear one — which, in an archive from the reference writer, is most of them:
1306
+ that writer percent-encodes every name it produces, so every name is printable ASCII
1307
+ and carries no flag, while an entry comment holding the original URL of a resource can
1308
+ carry one when that URL is not ASCII. The hand-built `page.pdf` records (§3.1, §6) are
1309
+ the only ones the reference writer does not produce through its ZIP writer, and they
1310
+ follow the same rule: `page.pdf` is ASCII, so they carry no flag either, and no entry
1311
+ in the archive is decoded differently from the rest. Archives written before this rule
1312
+ was relaxed carry the flag on every entry instead. Both decode identically, which is
1313
+ the point, but a reader that hardcodes either answer meets the other one eventually.
1306
1314
 
1307
1315
  A name is not a path. §7.3's rule that entry names are untrusted applies to the decoded
1308
1316
  name, and decoding is the step before that check, not a substitute for it.
@@ -1399,7 +1407,7 @@ pages can stop at the first row; the files it produces are accepted by every rea
1399
1407
  wrapper start tag chosen for the PDF payload (§5.1), the hand-built `page.pdf`
1400
1408
  local file header, the PDF document, the wrapper end tag, and record the local
1401
1409
  header's absolute position; then resume the prologue. The reference writer's
1402
- header declares version 2.0, the language encoding flag alone, method STORE, the
1410
+ header declares version 2.0, no general purpose bit flag, method STORE, the
1403
1411
  build's modification date in DOS form, the precomputed CRC-32, the document's
1404
1412
  length as both sizes, and no extra field; its central record adds a Unix
1405
1413
  "made by" version and external attributes of a regular file, mode 0644.
@@ -1892,6 +1900,7 @@ predicts.
1892
1900
  | August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
1893
1901
  | August 2026 | Core 1.5.119: the inlined ZIP library is built ASCII-only, and §2.1 now requires it of any bootstrap. Its CP437 table had been emitted as literal characters, which the page re-decoded as windows-1252, growing the table from 256 entries to 508 and shifting every lookup by 60 — so the one entry read without the UTF-8 flag, `page.pdf`, came back mangled and no archive with a PDF face extracted in any engine (§5.8) |
1894
1902
  | August 2026 | Core 1.5.120: the hand-built `page.pdf` records set the language encoding flag, like every entry the ZIP writer produces (§5.8). Its name is ASCII, so no decoded name changes; what changes is that no entry in an archive is read through CP437 any more, closing the path the 1.5.119 defect surfaced on |
1903
+ | September 2026 | §5.8 no longer requires bit 11 on every entry, deferring to ZIP's own rule: the flag is set when a name or a comment holds a byte outside printable ASCII, and left clear otherwise, because readers disagree about the flag more than they disagree about ASCII. The reference writer's names are all percent-encoded, so in practice none of them carries it now, and the hand-built `page.pdf` records follow the writer instead of overriding it — reversing the 1.5.120 row below, whose reason was that `page.pdf` would otherwise be the only entry read through the legacy path. It no longer is: every name in the archive takes the same path again, the other one |
1895
1904
  | September 2026 | Core 1.5.126: the appended-data budget becomes the `maxAppendedDataLength` writer option and its default drops from 65535 to 16361 bytes, so the EOCD record stays inside libarchive's scan and `bsdtar` opens archives it used to reject (§5.2, §8.1). The 65535-byte comment ceiling is now a separate limit, stated in §4.2: a budget raised past it produces a run that cannot be declared |
1896
1905
 
1897
1906
  This document was itself revised in August 2026, against core 1.5.108, after several
package/package.json CHANGED
@@ -1,11 +1,11 @@
1
1
  {
2
2
  "name": "single-file-core",
3
- "version": "1.5.126",
3
+ "version": "1.5.127",
4
4
  "description": "SingleFile Core",
5
5
  "author": "Gildas Lormeau",
6
6
  "license": "AGPL-3.0-or-later",
7
7
  "scripts": {
8
- "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
8
+ "test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/font-face-composite.js && deno run --allow-read test/sfz-harness/inlined-functions.js && deno run --allow-read test/sfz-harness/pages-archive.js && deno run --allow-read test/sfz-harness/filename-max-length.js && deno run --allow-read test/sfz-harness/entry-compression.js && deno run --allow-read test/sfz-harness/filename-characters.js && deno run --allow-read test/sfz-harness/byte-map.js && deno run --allow-read test/sfz-harness/zip64.js && deno run --allow-read test/sfz-harness/charset-round-trip.js",
9
9
  "bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
10
10
  "bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
11
11
  "bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
@@ -100,16 +100,12 @@ async function createPagesArchive(pages, options) {
100
100
  const zipReader = new ZipReader(new Uint8ArrayReader(await pages[pageIndex].getData()));
101
101
  for (const entry of await zipReader.getEntries()) {
102
102
  const filename = pagePath + entry.filename;
103
- const rawData = await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
103
+ const rawData = entry.directory ? undefined : await entry.getData(new Uint8ArrayWriter(), { passThrough: true, checkCrc32: false });
104
104
  const canonicalFilename = writtenEntries && findDuplicate(writtenEntries, filename, entry, rawData);
105
105
  if (canonicalFilename === undefined) {
106
- await zipWriter.add(filename, new Uint8ArrayReader(rawData), {
106
+ await zipWriter.add(filename, entry.directory ? null : new Uint8ArrayReader(rawData), {
107
107
  passThrough: true,
108
- compressionMethod: entry.compressionMethod,
109
- uncompressedSize: entry.uncompressedSize,
110
- crc32: entry.crc32,
111
- comment: entry.comment,
112
- lastModDate: entry.lastModDate
108
+ entry
113
109
  });
114
110
  } else {
115
111
  aliases[filename] = canonicalFilename;
@@ -40,9 +40,7 @@ import {
40
40
  router
41
41
  } from "./compression-router.js";
42
42
 
43
- const { Blob, fetch, TextEncoder, TextDecoder, DOMParser } = globalThis;
44
-
45
- const TEXT_DECODER = new TextDecoder("windows-1252");
43
+ const { Blob, fetch, TextEncoder, DOMParser } = globalThis;
46
44
 
47
45
  const COMPRESSIBLE_CONTENT_TYPES = ["application/javascript", "application/x-javascript", "application/ecmascript", "application/json", "application/ld+json", "application/manifest+json", "application/xml", "application/xhtml+xml", "application/rss+xml", "application/atom+xml", "image/svg+xml"];
48
46
  const TEXT_CONTENT_TYPE_PREFIX = "text/";
@@ -63,19 +61,20 @@ const EMBEDDED_DATA_TAGS = [
63
61
  ...EXTRA_DATA_TAGS,
64
62
  ];
65
63
  const DATA_IDENTIFIER = "sfz-data";
66
- const EXTRA_DATA_REGEXPS = [
67
- [/<script/i, /<\/script[\t\n\f\r />]/i],
68
- [/<style/i, /<\/style[\t\n\f\r />]/i],
69
- [/<noframes/i, /<\/noframes[\t\n\f\r />]/i],
70
- [/<noembed/i, /<\/noembed[\t\n\f\r />]/i],
71
- [/<iframe/i, /<\/iframe[\t\n\f\r />]/i],
72
- [/<xmp/i, /<\/xmp[\t\n\f\r />]/i],
73
- [/<!\[CDATA\[/i, /\]\]>/],
74
- [/<plaintext/i, /<\/plaintext[\t\n\f\r />]/i]
64
+ const TAG_NAME_TERMINATORS = "\t\n\f\r />";
65
+ const EXTRA_DATA_PATTERNS = [
66
+ [["<script"], ["</script", TAG_NAME_TERMINATORS]],
67
+ [["<style"], ["</style", TAG_NAME_TERMINATORS]],
68
+ [["<noframes"], ["</noframes", TAG_NAME_TERMINATORS]],
69
+ [["<noembed"], ["</noembed", TAG_NAME_TERMINATORS]],
70
+ [["<iframe"], ["</iframe", TAG_NAME_TERMINATORS]],
71
+ [["<xmp"], ["</xmp", TAG_NAME_TERMINATORS]],
72
+ [["<![CDATA["], ["]]>"]],
73
+ [["<plaintext"], ["</plaintext", TAG_NAME_TERMINATORS]]
75
74
  ];
76
- const EMBEDDED_DATA_REGEXPS = [
77
- [/<!--/i, /--!?>|<!-$/i],
78
- ...EXTRA_DATA_REGEXPS,
75
+ const EMBEDDED_DATA_PATTERNS = [
76
+ [["<!--"], ["-->"], ["--!>"], ["<!-", undefined, true]],
77
+ ...EXTRA_DATA_PATTERNS,
79
78
  ];
80
79
  const CRC32_TABLE = new Uint32Array(256).map((_, indexTable) => {
81
80
  let crc = indexTable;
@@ -106,7 +105,6 @@ const CENTRAL_FILE_HEADER_SIGNATURE = 0x02014b50;
106
105
  const END_OF_CENTRAL_DIR_SIGNATURE = 0x06054b50;
107
106
  const ZIP64_END_OF_CENTRAL_DIR_SIGNATURE = 0x06064b50;
108
107
  const ZIP64_END_OF_CENTRAL_DIR_LOCATOR_SIGNATURE = 0x07064b50;
109
- const LANGUAGE_ENCODING_FLAG = 0x0800;
110
108
 
111
109
  const browser = globalThis.browser;
112
110
 
@@ -229,18 +227,14 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
229
227
  const lfCodes = [];
230
228
  let crc32 = -1;
231
229
  if (!options.extractDataFromPageTags || options.extractDataFromPageTags[0] != "<plaintext>") {
232
- const textContent = TEXT_DECODER.decode(data.subarray(startOffset));
230
+ const zipData = data.subarray(startOffset);
233
231
  if (options.extractDataFromPageTags) {
234
232
  const tagIndex = getExtraDataTagIndex(options.extractDataFromPageTags);
235
- const regExpsTag = EXTRA_DATA_REGEXPS[tagIndex];
236
- if (textContent.match(regExpsTag[0]) || textContent.match(regExpsTag[1])) {
237
- return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, tagIndex + 1);
238
- }
239
- } else {
240
- const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[0];
241
- if (textContent.match(startRegExp) || textContent.match(endRegExp)) {
242
- return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions);
233
+ if (containsDataPattern(zipData, EXTRA_DATA_PATTERNS[tagIndex])) {
234
+ return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, tagIndex + 1);
243
235
  }
236
+ } else if (containsDataPattern(zipData, EMBEDDED_DATA_PATTERNS[0])) {
237
+ return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions);
244
238
  }
245
239
  }
246
240
  if (options.extractDataFromPage) {
@@ -337,9 +331,7 @@ function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, opt
337
331
  const tail = pageContent.slice(zipDataEnd - WRAPPER_PATTERN_WINDOW_LENGTH, zipDataEnd + COMMENT_LENGTH_FIELD_LENGTH);
338
332
  new DataView(tail.buffer).setUint16(WRAPPER_PATTERN_WINDOW_LENGTH, appendedDataLength, true);
339
333
  const tagIndex = options.extractDataFromPageTags ? getExtraDataTagIndex(options.extractDataFromPageTags) + 1 : 0;
340
- const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[tagIndex];
341
- const tailText = TEXT_DECODER.decode(tail);
342
- return !tailText.match(startRegExp) && !tailText.match(endRegExp);
334
+ return !containsDataPattern(tail, EMBEDDED_DATA_PATTERNS[tagIndex]);
343
335
  }
344
336
 
345
337
  function getCRC32(data, indexData = 0) {
@@ -365,7 +357,6 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
365
357
  const localHeaderView = new DataView(localHeader.buffer);
366
358
  localHeaderView.setUint32(0, LOCAL_FILE_HEADER_SIGNATURE, true);
367
359
  localHeaderView.setUint16(4, 20, true);
368
- localHeaderView.setUint16(6, LANGUAGE_ENCODING_FLAG, true);
369
360
  localHeaderView.setUint16(10, dosTime, true);
370
361
  localHeaderView.setUint16(12, dosDate, true);
371
362
  localHeaderView.setUint32(14, crc32, true);
@@ -378,7 +369,6 @@ function getPDFEntry(embeddedPdf, lastModDate = new Date()) {
378
369
  centralRecordView.setUint32(0, CENTRAL_FILE_HEADER_SIGNATURE, true);
379
370
  centralRecordView.setUint16(4, 0x0300, true);
380
371
  centralRecordView.setUint16(6, 20, true);
381
- centralRecordView.setUint16(8, LANGUAGE_ENCODING_FLAG, true);
382
372
  centralRecordView.setUint16(12, dosTime, true);
383
373
  centralRecordView.setUint16(14, dosDate, true);
384
374
  centralRecordView.setUint32(16, crc32, true);
@@ -552,8 +542,7 @@ function getStartHTMLArray(pageData, options, lastModDate, startTag = "") {
552
542
  const embeddedPdf = new Uint8Array(options.embeddedPdf);
553
543
  pdfEntry = options.preventEmbeddedPdfEntry ? undefined : getPDFEntry(embeddedPdf, lastModDate);
554
544
  const localHeader = pdfEntry ? pdfEntry.localHeader : new Uint8Array(0);
555
- const embeddedPdfText = TEXT_DECODER.decode(localHeader) + TEXT_DECODER.decode(embeddedPdf);
556
- const pdfTagIndex = findEmbeddedDataTagIndex(embeddedPdfText);
545
+ const pdfTagIndex = findEmbeddedDataTagIndex(concatArrays(localHeader, embeddedPdf));
557
546
  if (pdfTagIndex == -1) {
558
547
  dropUnhiddenFace(options, "embeddedPdf", EMBEDDED_PDF_LABEL);
559
548
  pdfEntry = undefined;
@@ -629,12 +618,11 @@ function getExtraDataTagIndex(extractDataFromPageTags) {
629
618
  return tagIndex;
630
619
  }
631
620
 
632
- function findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags = 0) {
633
- const regExpsTag = EXTRA_DATA_REGEXPS[indexExtractDataFromPageTags];
621
+ function findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags = 0) {
634
622
  const plaintextTag = EXTRA_DATA_TAGS[indexExtractDataFromPageTags][0] == "<plaintext>";
635
- const matchTag = !plaintextTag && (textContent.match(regExpsTag[0]) || textContent.match(regExpsTag[1]));
623
+ const matchTag = !plaintextTag && containsDataPattern(zipData, EXTRA_DATA_PATTERNS[indexExtractDataFromPageTags]);
636
624
  if (matchTag) {
637
- return findExtraDataTags(textContent, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags + 1);
625
+ return findExtraDataTags(zipData, pageData, options, script, entriesData, zipWriterOptions, indexExtractDataFromPageTags + 1);
638
626
  } else {
639
627
  options.extractDataFromPageTags = EXTRA_DATA_TAGS[indexExtractDataFromPageTags];
640
628
  if (options.extractDataFromPageTags[0] == "<plaintext>") {
@@ -644,24 +632,93 @@ function findExtraDataTags(textContent, pageData, options, script, entriesData,
644
632
  }
645
633
  }
646
634
 
647
- function findEmbeddedDataTagIndex(text, fromIndex = 0) {
648
- const tagIndex = EMBEDDED_DATA_REGEXPS.slice(fromIndex, -1).findIndex(([startRegExp, endRegExp]) => !text.match(startRegExp) && !text.match(endRegExp));
635
+ function findEmbeddedDataTagIndex(data, fromIndex = 0) {
636
+ const tagIndex = EMBEDDED_DATA_PATTERNS.slice(fromIndex, -1).findIndex(patterns => !containsDataPattern(data, patterns));
649
637
  return tagIndex == -1 ? -1 : tagIndex + fromIndex;
650
638
  }
651
639
 
640
+ function containsDataPattern(data, patterns) {
641
+ const patternsByCharCode = new Map();
642
+ for (const pattern of patterns) {
643
+ const [text, , atEnd] = pattern;
644
+ if (atEnd) {
645
+ if (matchesDataPatternAt(data, pattern, data.length - text.length)) {
646
+ return true;
647
+ }
648
+ } else {
649
+ const charCode = text.charCodeAt(0);
650
+ indexPatternByCharCode(patternsByCharCode, charCode, pattern);
651
+ const alternateCharCode = getAlternateCharCode(charCode);
652
+ if (alternateCharCode != -1) {
653
+ indexPatternByCharCode(patternsByCharCode, alternateCharCode, pattern);
654
+ }
655
+ }
656
+ }
657
+ for (const [charCode, candidates] of patternsByCharCode) {
658
+ for (let index = data.indexOf(charCode); index != -1; index = data.indexOf(charCode, index + 1)) {
659
+ for (let indexCandidate = 0; indexCandidate < candidates.length; indexCandidate++) {
660
+ if (matchesDataPatternAt(data, candidates[indexCandidate], index)) {
661
+ return true;
662
+ }
663
+ }
664
+ }
665
+ }
666
+ return false;
667
+ }
668
+
669
+ function indexPatternByCharCode(patternsByCharCode, charCode, pattern) {
670
+ if (!patternsByCharCode.has(charCode)) {
671
+ patternsByCharCode.set(charCode, []);
672
+ }
673
+ patternsByCharCode.get(charCode).push(pattern);
674
+ }
675
+
676
+ function matchesDataPatternAt(data, [text, terminators], index) {
677
+ const textLength = text.length;
678
+ if (index < 0 || index + textLength > data.length) {
679
+ return false;
680
+ }
681
+ for (let indexText = 0; indexText < textLength; indexText++) {
682
+ const charCode = text.charCodeAt(indexText);
683
+ const code = data[index + indexText];
684
+ if (code != charCode && code != getAlternateCharCode(charCode)) {
685
+ return false;
686
+ }
687
+ }
688
+ return terminators === undefined || isTerminatorCode(terminators, data[index + textLength]);
689
+ }
690
+
691
+ function isTerminatorCode(terminators, code) {
692
+ return code !== undefined && terminators.includes(String.fromCharCode(code));
693
+ }
694
+
695
+ function getAlternateCharCode(charCode) {
696
+ const lowerCharCode = charCode | 0x20;
697
+ return lowerCharCode >= 0x61 && lowerCharCode <= 0x7a ? charCode ^ 0x20 : -1;
698
+ }
699
+
700
+ function concatArrays(...arrays) {
701
+ const result = new Uint8Array(arrays.reduce((length, array) => length + array.length, 0));
702
+ let offset = 0;
703
+ arrays.forEach(array => {
704
+ result.set(array, offset);
705
+ offset += array.length;
706
+ });
707
+ return result;
708
+ }
709
+
652
710
  function getImageHTMLChunk(pageData, options, lastModDate) {
653
- const embeddedImageText = TEXT_DECODER.decode(getEmbeddedImageData(options.embeddedImage)) +
654
- TEXT_DECODER.decode(new Uint8Array(4)) + TEXT_DECODER.decode(PNG_ZIP_CHUNK_TYPE_KEYWORD);
655
- let tagIndex = findEmbeddedDataTagIndex(embeddedImageText);
711
+ const embeddedImageData = concatArrays(getEmbeddedImageData(options.embeddedImage), new Uint8Array(4), PNG_ZIP_CHUNK_TYPE_KEYWORD);
712
+ let tagIndex = findEmbeddedDataTagIndex(embeddedImageData);
656
713
  while (tagIndex != -1) {
657
714
  const [startTag, endTag] = EMBEDDED_DATA_TAGS[tagIndex];
658
715
  const startHTMLData = getStartHTMLArray(pageData, options, lastModDate, startTag);
659
716
  const htmlData = new Uint8Array([...getLength(startHTMLData.htmlArray.length + 4), ...[0x74, 0x45, 0x58, 0x74, 0x50, 0x4e, 0x47, 0], ...startHTMLData.htmlArray]);
660
717
  const htmlDataCRC = getCRC32(htmlData, 4);
661
- const wrappedText = TEXT_DECODER.decode(htmlDataCRC) + embeddedImageText;
718
+ const wrappedData = concatArrays(htmlDataCRC, embeddedImageData);
662
719
  if ((tagIndex == 0 && (htmlDataCRC[0] == 0x3e || (htmlDataCRC[0] == 0x2d && htmlDataCRC[1] == 0x3e))) ||
663
- findEmbeddedDataTagIndex(wrappedText, tagIndex) != tagIndex) {
664
- tagIndex = findEmbeddedDataTagIndex(embeddedImageText, tagIndex + 1);
720
+ findEmbeddedDataTagIndex(wrappedData, tagIndex) != tagIndex) {
721
+ tagIndex = findEmbeddedDataTagIndex(embeddedImageData, tagIndex + 1);
665
722
  } else {
666
723
  return { endTag, startHTMLData, htmlData, htmlDataCRC };
667
724
  }
@@ -0,0 +1,135 @@
1
+ // Several @font-face rules that declare the same family with the same style descriptors are one
2
+ // composite face, not a stack where the last rule wins. CSS Fonts 4 §5.2: "When the matched face is
3
+ // a composite face, user agents must use the procedure above on each of the faces in the composite
4
+ // face in reverse order of @font-face rule definition", selecting a face only "if the effective
5
+ // character map supports the character in question". §4.5.1 says the same for the case where the
6
+ // ranges are identical rather than merely overlapping: "If the unicode ranges overlap for a set of
7
+ // @font-face rules with the same family and style descriptor values, the rules are ordered in the
8
+ // reverse order they were defined; the last rule defined is the first to be checked for a given
9
+ // character." So the later rule wins per character, and a character it has no glyph for falls back
10
+ // to the earlier rule of the same family.
11
+ //
12
+ // Dropping the earlier rule therefore loses glyphs. science.org declares icomoon twice, weight 400
13
+ // style normal in both, first a 103-codepoint icon set and then a 30-codepoint one holding only the
14
+ // slideshow arrows. The banner close button is content:"\e928", which lives in the first font only,
15
+ // so keeping the last rule alone rendered a tofu box reading "E9 28" where the X had been.
16
+ //
17
+ // Pooling the sources of every rule sharing a key breaks it the same way even when both rules are
18
+ // kept: one winning source gets written into all of them, so both rules end up naming the same font
19
+ // and the other one is gone just as surely. Each rule keeps its own sources.
20
+ //
21
+ // Rules that are duplicates outright, same key and same src, are still emitted once: they are the
22
+ // same member of the composite declared twice, so dropping one changes nothing.
23
+ import * as cssTree from "../../vendor/css-tree.js";
24
+
25
+ // helper.js reaches the frame hooks, which install themselves against window and document as they
26
+ // are evaluated: the stubs go in before the dynamic import
27
+ globalThis.window = globalThis;
28
+ globalThis.document = {};
29
+ globalThis.Document = class Document { };
30
+ globalThis.MutationObserver = class MutationObserver { observe() { } };
31
+ const { getProcessorHelperCommonClass } = await import("../../core/lib/processor-helper-common.js");
32
+
33
+ const ProcessorHelperCommon = getProcessorHelperCommonClass({}, cssTree);
34
+
35
+ // the real subclasses pick one source out of the list and rewrite the rule with it; the contract
36
+ // under test is which rules survive and which sources each one is handed, so this records that and
37
+ // keeps every rule
38
+ class TestProcessorHelper extends ProcessorHelperCommon {
39
+ constructor() {
40
+ super();
41
+ this.processedRules = [];
42
+ }
43
+ async processFontFaceRule(ruleData, fontInfo) {
44
+ this.processedRules.push({
45
+ family: this.getPropertyValue(ruleData, "font-family"),
46
+ sources: fontInfo.map(source => source.src)
47
+ });
48
+ return true;
49
+ }
50
+ }
51
+
52
+ async function run(css) {
53
+ const helper = new TestProcessorHelper();
54
+ const stylesheets = new Map([[0, { stylesheet: cssTree.parse(css) }]]);
55
+ await helper.removeAlternativeFonts({}, stylesheets, new Map(), new Map());
56
+ const remaining = [];
57
+ stylesheets.get(0).stylesheet.children.forEach(ruleData => {
58
+ if (ruleData.type == "Atrule" && ruleData.name == "font-face") {
59
+ remaining.push(helper.getPropertyValue(ruleData, "src"));
60
+ }
61
+ });
62
+ return { processed: helper.processedRules, remaining };
63
+ }
64
+
65
+ let failures = 0;
66
+
67
+ // a rule that was dropped when it should have been kept leaves a hole in the list, and reporting
68
+ // that as a failed comparison is more use than throwing on the way to the assertion
69
+ function sourcesOf(processedRules, index) {
70
+ const rule = processedRules[index];
71
+ return rule ? rule.sources : null;
72
+ }
73
+
74
+ function check(label, actual, expected) {
75
+ const pass = JSON.stringify(actual) == JSON.stringify(expected);
76
+ console.log((pass ? "PASS " : "FAIL ") + label + ": " + JSON.stringify(actual));
77
+ if (!pass) {
78
+ console.log(" expected: " + JSON.stringify(expected));
79
+ failures++;
80
+ }
81
+ }
82
+
83
+ const TWO_FACES = `
84
+ @font-face{font-family:icomoon;src:url(big.ttf) format("truetype"),url(big.woff) format("woff");font-weight:400;font-style:normal}
85
+ @font-face{font-family:icomoon;src:url(small.ttf) format("truetype"),url(small.woff) format("woff");font-weight:400;font-style:normal}`;
86
+
87
+ const two = await run(TWO_FACES);
88
+ check("both members of the composite face are kept", two.processed.length, 2);
89
+ check("the earlier rule keeps its own sources", sourcesOf(two.processed, 0).sort(), ["url(big.ttf)format(\"truetype\")", "url(big.woff)format(\"woff\")"]);
90
+ check("the later rule keeps its own sources", (sourcesOf(two.processed, 1) || []).sort(), ["url(small.ttf)format(\"truetype\")", "url(small.woff)format(\"woff\")"]);
91
+ check("neither rule is removed from the stylesheet", two.remaining.length, 2);
92
+
93
+ const DUPLICATE = `
94
+ @font-face{font-family:icomoon;src:url(one.woff) format("woff");font-weight:400;font-style:normal}
95
+ @font-face{font-family:icomoon;src:url(one.woff) format("woff");font-weight:400;font-style:normal}`;
96
+
97
+ const duplicate = await run(DUPLICATE);
98
+ check("an outright duplicate rule is emitted once", duplicate.processed.length, 1);
99
+ check("the duplicate is removed from the stylesheet", duplicate.remaining.length, 1);
100
+
101
+ // unicode-range is part of the font key, so the subsetting idiom was never affected by the
102
+ // shadowing bug; it is pinned here because the fix moved what the key is used for
103
+ const RANGES = `
104
+ @font-face{font-family:sub;src:url(latin.woff2) format("woff2");unicode-range:U+0-7F}
105
+ @font-face{font-family:sub;src:url(greek.woff2) format("woff2");unicode-range:U+370-3FF}`;
106
+
107
+ const ranges = await run(RANGES);
108
+ check("faces split by unicode-range are all kept", ranges.processed.length, 2);
109
+ check("each range keeps its own source", ranges.processed.map(rule => rule.sources), [["url(latin.woff2)format(\"woff2\")"], ["url(greek.woff2)format(\"woff2\")"]]);
110
+
111
+ // a rule declaring the same source twice contributes it once, in the position its later declaration
112
+ // gives it. The sources are compared after the separating comma is stripped, so a repeat is
113
+ // recognised wherever it sits: the value is split by a regexp that keeps that comma, and comparing
114
+ // the raw pieces made the last source in a list unequal to the same source anywhere before it
115
+ const REPEATED = `
116
+ @font-face{font-family:repeat;src:url(a.woff) format("woff"),url(b.woff) format("woff"),url(a.woff) format("woff"),url(c.woff) format("woff");font-weight:400}`;
117
+
118
+ const repeated = await run(REPEATED);
119
+ check("a source repeated inside one rule is listed once", (sourcesOf(repeated.processed, 0) || []).length, 3);
120
+
121
+ const REPEATED_LAST = `
122
+ @font-face{font-family:repeat;src:url(a.woff) format("woff"),url(b.woff) format("woff"),url(a.woff) format("woff");font-weight:400}`;
123
+
124
+ const repeatedLast = await run(REPEATED_LAST);
125
+ check("a repeat in last position is recognised too", (sourcesOf(repeatedLast.processed, 0) || []).length, 2);
126
+ // the list is held in reverse of the order it is written back in, so the entry the rule declares
127
+ // last comes first here: a.woff keeps the position its second declaration gives it
128
+ check("the repeat keeps the position of its later declaration", sourcesOf(repeatedLast.processed, 0), ["url(a.woff)format(\"woff\")", "url(b.woff)format(\"woff\")"]);
129
+
130
+ if (failures) {
131
+ console.log("\n" + failures + " check(s) failed");
132
+ Deno.exit(1);
133
+ } else {
134
+ console.log("\nall checks passed");
135
+ }
@@ -432,9 +432,13 @@ const ALL_FACE_RUNGS = "<!--sfz-data<script<style<noframes<noembed<iframe<xmp<![
432
432
  check("no page.pdf entry is left behind", entries.some(entry => entry.filename.endsWith("page.pdf")), false);
433
433
  }
434
434
 
435
- // page.pdf is the only record the writer builds by hand, so it is the only place the
436
- // language encoding flag can go missing: without it a reader decodes that one name
437
- // through CP437 while reading every other name in the same archive as UTF-8
435
+ // page.pdf is the only record the writer builds by hand, so it is the only place the language
436
+ // encoding flag can disagree with the rest of the archive, and a reader would then decode that one
437
+ // name through a different path than every other name in the same file. Which way the flag goes is
438
+ // the ZIP writer's business, not this format's: it sets bit 11 only when a name or a comment holds
439
+ // a byte outside printable ASCII, so every name in this fixture is flagless. What is asserted here
440
+ // is the agreement, so this still fails if the hand-built record diverges and still passes if the
441
+ // writer changes its rule again
438
442
  {
439
443
  const options = makeOptions({ embeddedPdf: PDF });
440
444
  const pageData = makePageData(23, 4 * 1024);
@@ -443,11 +447,14 @@ const ALL_FACE_RUNGS = "<!--sfz-data<script<style<noframes<noembed<iframe<xmp<![
443
447
  const zipReader = new ZipReader(new BlobReader(new Blob([bytes])));
444
448
  const entries = await zipReader.getEntries();
445
449
  await zipReader.close();
446
- check("the pdf entry is listed", entries.some(entry => entry.filename == "page.pdf"), true);
447
- check("every central record declares utf-8 names", entries.every(entry => entry.filenameUTF8), true);
450
+ const pdfEntry = entries.find(entry => entry.filename == "page.pdf");
451
+ check("the pdf entry is listed", Boolean(pdfEntry), true);
452
+ check("the hand-built central record declares its name like the records the writer produces",
453
+ entries.every(entry => Boolean(entry.filenameUTF8) == Boolean(pdfEntry.filenameUTF8)), true);
448
454
  const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
449
- const localFlags = entries.map(entry => view.getUint16(entry.offset + 6, true));
450
- check("every local header declares utf-8 names", localFlags.every(flags => Boolean(flags & 0x0800)), true);
455
+ const localFlags = new Map(entries.map(entry => [entry.filename, view.getUint16(entry.offset + 6, true) & 0x0800]));
456
+ check("the hand-built local header declares its name like the headers the writer produces",
457
+ [...localFlags.values()].every(flag => flag == localFlags.get("page.pdf")), true);
451
458
  }
452
459
 
453
460
  {