single-file-core 1.5.115 → 1.5.117
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/core/helper.js +34 -13
- package/core/index.js +19 -5
- package/core/lib/processor-helper-common.js +42 -18
- package/core/lib/processor-helper-inline.js +15 -10
- package/core/lib/processor-helper.js +44 -19
- package/doc/singlefile-archive.md +123 -26
- package/modules/css-fonts-minifier.js +191 -33
- package/modules/css-rules-minifier.js +9 -2
- package/package.json +2 -2
- package/processors/compression/compression-display.js +23 -2
- package/processors/compression/compression-packager.js +2 -11
- package/processors/compression/compression.js +33 -9
- package/processors/hooks/content/content-hooks-frames-web.js +27 -4
- package/test/sfz-harness/README.md +5 -2
- package/test/sfz-harness/adopted-stylesheets-hook.js +240 -0
- package/test/sfz-harness/css-fonts-minifier.js +235 -0
- package/test/sfz-harness/css-property-filter.js +85 -0
- package/test/sfz-harness/format-rules.js +31 -1
- package/test/sfz-harness/inlined-functions.js +82 -0
|
@@ -198,12 +198,54 @@ Notes on composition:
|
|
|
198
198
|
|
|
199
199
|
### 2.1 The charset rule
|
|
200
200
|
|
|
201
|
-
The HTML face declares `<meta charset=utf-8>` when universal mode is off
|
|
202
|
-
single-byte charset
|
|
203
|
-
declaration MUST appear within the first 1024 bytes of the file
|
|
204
|
-
encoding prescan finds it.
|
|
205
|
-
|
|
206
|
-
|
|
201
|
+
The HTML face declares `<meta charset=utf-8>` when universal mode is off. When
|
|
202
|
+
universal mode is on it declares a single-byte charset instead — `windows-1252` in the
|
|
203
|
+
reference writer. The declaration MUST appear within the first 1024 bytes of the file
|
|
204
|
+
so the parser's encoding prescan finds it. That bound is the HTML standard's own
|
|
205
|
+
authoring rule. The prescan it serves is weaker than the rule suggests: the standard
|
|
206
|
+
makes it optional, and only *encourages* scanning the first 1024 bytes. Treat the
|
|
207
|
+
number as a ceiling to write under, never as a budget a parser promises to read. The
|
|
208
|
+
whole `<meta>` tag has to fit: one that straddles the boundary is not seen, and the
|
|
209
|
+
parser falls back to its default encoding. Meeting the declaration later, during
|
|
210
|
+
tokenization, does not rescue the file. The parser does not resume the prescan. It
|
|
211
|
+
re-navigates the document under the new encoding instead, and a writer must not rely
|
|
212
|
+
on that.
|
|
213
|
+
|
|
214
|
+
The declaration decides the decoding only when nothing outranks it. Three things do,
|
|
215
|
+
each returning an encoding with the standard's *certain* confidence, all of them ahead
|
|
216
|
+
of the prescan: a byte order mark, a user's explicit encoding override, and a charset
|
|
217
|
+
stated by the transport layer, which over HTTP means a `Content-Type` header carrying
|
|
218
|
+
its own `charset`. Any of the three replaces the declared charset, the parsed text is
|
|
219
|
+
then not what the writer encoded, and the region cannot be recovered from it. This is
|
|
220
|
+
the one precondition universal mode has that the file cannot satisfy from within
|
|
221
|
+
itself.
|
|
222
|
+
|
|
223
|
+
Where it bites is narrower than that makes it sound, because the parsed text is the
|
|
224
|
+
last rung, not the first. The bootstrap reads the file's raw bytes whenever it can
|
|
225
|
+
(§4.1), and raw bytes carry no encoding; universal extraction is the fallback for when
|
|
226
|
+
they are out of reach. Taking the three in turn:
|
|
227
|
+
|
|
228
|
+
- A **transport charset** exists only over HTTP, and over HTTP the raw read is what
|
|
229
|
+
runs — the bootstrap requests its own URL and takes the response as bytes, which no
|
|
230
|
+
`Content-Type` can reinterpret. It reaches universal extraction only in a double
|
|
231
|
+
failure: the response has to defeat the raw read, through a network or CORS failure
|
|
232
|
+
or a non-200 status, *and* state a charset of its own.
|
|
233
|
+
- A **BOM** is the writer's own doing. It is why universal and PNG variants never carry
|
|
234
|
+
one (§3.1): the reference writer emits a BOM for the plain variant only
|
|
235
|
+
(`includeBOM`), where nothing depends on the declared charset.
|
|
236
|
+
- A **user override** is the one no software can prevent, and the rarest.
|
|
237
|
+
|
|
238
|
+
On `file:` URLs the bootstrap goes straight to page-text extraction, since no raw read
|
|
239
|
+
is available there (§4.1) — but there is also no transport layer, so the first of the
|
|
240
|
+
three cannot arise on the very path that depends on the charset most.
|
|
241
|
+
|
|
242
|
+
The failure is safe rather than silent, which is why the precondition is worth stating
|
|
243
|
+
at all. Decoded under the wrong charset the reconstructed bytes are wrong, the payload
|
|
244
|
+
checksum does not match, and the extractor MUST fail to the error message (§4.5)
|
|
245
|
+
instead of displaying a corrupt page. A reader MAY tell the case apart from ordinary
|
|
246
|
+
corruption by comparing the encoding the document was actually decoded with —
|
|
247
|
+
`document.characterSet` in a browser — against the declared one, and say so in the
|
|
248
|
+
error message. Nothing requires it, and the MUST is unaffected either way.
|
|
207
249
|
|
|
208
250
|
Universal mode works in two parts, and the charset carries the first. The archive
|
|
209
251
|
bytes themselves are recovered *from the parsed page text*: the browser decoded
|
|
@@ -267,7 +309,7 @@ face adds, then the regions the PNG face adds.
|
|
|
267
309
|
|
|
268
310
|
| Region | Producer | Present | Contents |
|
|
269
311
|
|---|---|---|---|
|
|
270
|
-
| `html-prologue` | HTML | HTML face | Doctype, the root element start tag, an optional implementation-defined comment, `<meta charset>`, title, optional head elements (canonical link, `robots` meta, viewport, Content-Security-Policy), minimal CSS, `<body hidden>`, wait/error messages, optional table of contents, optional text body (§4.6). In the plain variant an optional UTF-8 BOM MAY precede the doctype (`includeBOM`); universal and PNG variants never carry one. In the PNG variants the region is split: everything through `<body hidden>` is the data of the `tEXt "PNG"` chunk, while the messages, the optional table of contents and the optional text body follow the `tEXt "ZIP"` chunk header; the doctype and the leading comment are dropped. |
|
|
312
|
+
| `html-prologue` | HTML | HTML face | Doctype, the root element start tag, an optional implementation-defined comment, `<meta charset>`, title, optional head elements (canonical link, `robots` meta, viewport, Content-Security-Policy), minimal CSS, `<body hidden>`, wait/error messages, optional table of contents, optional text body (§4.6). The leading comment, the title, the canonical link and the text body are withheld when a password is set (§5.6). In the plain variant an optional UTF-8 BOM MAY precede the doctype (`includeBOM`); universal and PNG variants never carry one. In the PNG variants the region is split: everything through `<body hidden>` is the data of the `tEXt "PNG"` chunk, while the messages, the optional table of contents and the optional text body follow the `tEXt "ZIP"` chunk header; the doctype and the leading comment are dropped. |
|
|
271
313
|
| `bootstrap` | HTML | HTML face | One inline `<script>`: the embedded ZIP reader, the extractor, the display routine, and the content-acquisition logic (§4.1). The wrapper start tag that opens the ZIP region follows it, directly or after a relocated `extra-data`. |
|
|
272
314
|
| `<!--` / `-->` | HTML | HTML face | The wrapper tag pair hiding a binary region from the HTML parser — comment tags by default, another pair when the hidden bytes contain `-->` (§5.1). Drawn at each opening and closing position. The close tag is absent when appended data is prevented (`preventAppendedData`, or the `<plaintext>` wrapper which cannot close): no markup follows the archive and the wrapper runs to end-of-file. That does not mean the file ends at the EOCD — the PNG face's tail still follows, inside the wrapper, where it parses as text (§5.1). |
|
|
273
315
|
| `zip-entries` | ZIP | always | The archive's local file headers and entry data, written by the ZIP writer. The central directory of an archive written by the reference writer lists `index.html` (the page) first, then `manifest.json` (a JSON description of the archive: original URL, title, save time, resource-to-URL map — informative; the page displays without it), then the resources; the *physical* order of the local headers inside the region is not guaranteed to match, and readers MUST NOT rely on either order — entries are addressed by name (§7.1). |
|
|
@@ -307,7 +349,12 @@ change.
|
|
|
307
349
|
The HTML parser consumes the whole file as one document. Its encoding prescan finds
|
|
308
350
|
the `<meta charset>` declaration within the first 1024 bytes (§2.1) and the file is
|
|
309
351
|
decoded as a single text; every binary region therefore also exists as characters in
|
|
310
|
-
the parsed document, which is what universal mode exploits (§4.5).
|
|
352
|
+
the parsed document, which is what universal mode exploits (§4.5). This holds only
|
|
353
|
+
while the declaration is what decides the decoding: a BOM, a user override or a
|
|
354
|
+
transport-layer charset outranks it, and universal extraction then fails its checksum
|
|
355
|
+
rather than recovering anything (§2.1). The acquisition order below keeps that off the
|
|
356
|
+
common path — the raw bytes are read in preference to the parsed text wherever they
|
|
357
|
+
can be, and no encoding applies to them.
|
|
311
358
|
|
|
312
359
|
The binary regions are kept out of the rendered page by the wrapper tags. The
|
|
313
360
|
default wrapper is an HTML comment, and the HTML standard defines exactly which
|
|
@@ -494,8 +541,10 @@ decoded image is exactly that image.
|
|
|
494
541
|
|
|
495
542
|
The last reader is the format's own: the extraction path of universal mode, used
|
|
496
543
|
when the raw bytes are unreachable (§4.1). Its input is not the file but the *parsed
|
|
497
|
-
document* — the characters the HTML parser produced — and its output is the
|
|
498
|
-
|
|
544
|
+
document* — the characters the HTML parser produced — and its output is the ZIP
|
|
545
|
+
region reconstructed byte for byte, with one deliberate exception: the two bytes of
|
|
546
|
+
the EOCD comment-length field, which the payload does not describe and the extractor
|
|
547
|
+
always writes as zero (step 2 below, and the row in §7.4).
|
|
499
548
|
|
|
500
549
|
It works in three steps:
|
|
501
550
|
|
|
@@ -968,6 +1017,24 @@ The payload itself is a sequence of little-endian 32-bit words — checksum, rec
|
|
|
968
1017
|
range length, newline count, then the codes packed 16 per word, least-significant pair
|
|
969
1018
|
first — raw-deflated and base64-encoded with the standard alphabet and padding.
|
|
970
1019
|
|
|
1020
|
+
Those word widths cap what the payload can describe. A writer MUST NOT use universal
|
|
1021
|
+
mode for a ZIP region of 2^32 bytes or more, since the length field cannot express it.
|
|
1022
|
+
The cap is not enforced by the wire format itself: a writer that ignores it stores the
|
|
1023
|
+
length modulo 2^32 and produces a file that looks well-formed, and the mismatch
|
|
1024
|
+
surfaces only when a reader verifies the field (§4.5). The reference writer is in that
|
|
1025
|
+
position — it assigns the length into a `Uint32Array`, where the truncation is silent
|
|
1026
|
+
— and reaches the cap in no saved page. This bound and zip64 (§5.7) are separate
|
|
1027
|
+
things: zip64 is reachable at any archive size through the 65535-entry trigger and
|
|
1028
|
+
stays compatible with universal mode, and it is only a region large enough to need
|
|
1029
|
+
zip64's 64-bit *offsets* that runs past what the payload can describe.
|
|
1030
|
+
|
|
1031
|
+
An engine limit binds long before the format's. The extractor holds the region as one
|
|
1032
|
+
JavaScript string, and the maximum string length is engine-specific: V8 caps it at
|
|
1033
|
+
2^29 − 24 characters, 536870888, measured on V8 15.0.245. A universal-mode archive
|
|
1034
|
+
whose ZIP region approaches half a gigabyte is therefore already unreadable in Chrome,
|
|
1035
|
+
Edge and Node, whatever the payload declares. Other engines set the limit elsewhere.
|
|
1036
|
+
The practical ceiling on universal mode is this one, not the 4 GiB above.
|
|
1037
|
+
|
|
971
1038
|
### 5.6 Password scope
|
|
972
1039
|
|
|
973
1040
|
A password encrypts the *contents* of ZIP entries with AES, and nothing else. A reader
|
|
@@ -982,15 +1049,18 @@ gets no protection beyond that. Four consequences follow:
|
|
|
982
1049
|
- **Entry metadata is never encrypted.** Names, uncompressed sizes and dates remain
|
|
983
1050
|
readable in the central directory, so the resource list of an encrypted archive is
|
|
984
1051
|
public. This is standard ZIP behavior, not a property of this format; §7 restates it.
|
|
985
|
-
- **What the writer withholds instead.**
|
|
986
|
-
the format and so are withheld when a password is set
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
1052
|
+
- **What the writer withholds instead.** Five things are not forced into the clear by
|
|
1053
|
+
the format, and so are withheld when a password is set. Three of them state a URL:
|
|
1054
|
+
the entry comments, which publish every resource's source URL (§4.2), and two
|
|
1055
|
+
prologue fields carrying the address the page was saved from — the provenance
|
|
1056
|
+
comment an implementation may write there, and the canonical `<link>` among the head
|
|
1057
|
+
elements (§3.1). The other two are the `<title>` element's text, leaving an empty
|
|
1058
|
+
`<title></title>` in the prologue, and the optional text body, which repeats the
|
|
1059
|
+
whole page text outside the archive (§4.6). Nothing is lost by leaving any of them
|
|
1060
|
+
out: `manifest.json` holds the page URL, the title and the resource-URL map, and it
|
|
1061
|
+
is an encrypted entry like the rest. Unlike the PNG and PDF faces, none of the five
|
|
1062
|
+
is load-bearing for a reader, so a writer that emits them in a password-protected
|
|
1063
|
+
archive publishes what the password is meant to cover for no gain.
|
|
994
1064
|
|
|
995
1065
|
Encrypted entries are stamped AE-2, so their CRC-32 field is zero (§5.4). `page.pdf`
|
|
996
1066
|
stays unencrypted, so in a password-protected archive its checksum is the only one a
|
|
@@ -1018,6 +1088,12 @@ first by both Info-ZIP and the reference reader, the central directory offset in
|
|
|
1018
1088
|
zip64 record points at the injected record, and extraction produces the same page as
|
|
1019
1089
|
the non-zip64 build.
|
|
1020
1090
|
|
|
1091
|
+
zip64 does not conflict with universal mode. Its commonest trigger, 65535 entries or
|
|
1092
|
+
more, is reached at any archive size, and §4.5 gives the offset arithmetic for a
|
|
1093
|
+
recovered region whose EOCD fields are sentinels. What universal mode cannot carry is
|
|
1094
|
+
a ZIP region of 2^32 bytes or more, which the recovery payload's 32-bit length field
|
|
1095
|
+
cannot express (§5.5) — a size bound, not a zip64 one.
|
|
1096
|
+
|
|
1021
1097
|
## 6. Writer algorithm
|
|
1022
1098
|
|
|
1023
1099
|
This section specifies the reference writer's build order. It is normative in the
|
|
@@ -1049,9 +1125,11 @@ pages can stop at the first row; the files it produces are accepted by every rea
|
|
|
1049
1125
|
2. **HTML prologue.** With the HTML face, emit the doctype (omitted under the PNG
|
|
1050
1126
|
face, which owns the start of the file), the root element start tag, any comment the
|
|
1051
1127
|
implementation adds, the `<meta charset>` required by §2.1, the head elements (the
|
|
1052
|
-
`<title>`
|
|
1128
|
+
`<title>` and the canonical link among them), the CSS and `<body hidden>`,
|
|
1053
1129
|
the wait and error messages, the optional table of contents and text body, and the
|
|
1054
|
-
bootstrap script. With
|
|
1130
|
+
bootstrap script. With a password, five of those are left out: the comment, the
|
|
1131
|
+
title, the canonical link, the text body and the entry comments of step 6 (§5.6).
|
|
1132
|
+
With the PNG face the head of this region,
|
|
1055
1133
|
through `<body hidden>`, is the data of the `tEXt "PNG"` chunk and the remainder is
|
|
1056
1134
|
emitted after the `tEXt "ZIP"` chunk header in step 12; with the PDF face the
|
|
1057
1135
|
region is interrupted by step 3 as well.
|
|
@@ -1259,10 +1337,13 @@ alongside it.
|
|
|
1259
1337
|
Software that displays either MUST do so in a sandboxed context, and MUST NOT run
|
|
1260
1338
|
the bootstrap in a privileged one. The format's own display path replaces the
|
|
1261
1339
|
document with the extracted page, which is not an isolation boundary by itself.
|
|
1262
|
-
- **A password protects entry contents only** (§5.6). Entry names, sizes
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1340
|
+
- **A password protects entry contents only** (§5.6). Entry names, sizes and dates
|
|
1341
|
+
stay readable in the central directory, and a name commonly states the resource's
|
|
1342
|
+
filename. The PNG and PDF faces render the page regardless. A conforming writer
|
|
1343
|
+
withholds the five fields of §5.6, the source URLs among them, but a reader MUST NOT
|
|
1344
|
+
read their absence as protection: nothing in the format stops a writer from emitting
|
|
1345
|
+
any of them, so an archive of unknown provenance may state every URL in the clear.
|
|
1346
|
+
Software MUST NOT present a password-protected archive as an encrypted document.
|
|
1266
1347
|
- **Sniffing disagrees with itself on these files.** `file(1)` reports HTML, PNG, PDF
|
|
1267
1348
|
or "data" depending on the variant (§8.1), so a server that guesses the media type
|
|
1268
1349
|
from content may serve a saved page as an image. Software that serves SingleFile
|
|
@@ -1285,7 +1366,7 @@ only if it affects the bytes the page is built from:
|
|
|
1285
1366
|
| A `tEXt` chunk CRC does not match, or a chunk holds bytes PNG does not permit (§4.4) | Irrelevant to extraction; a reader of the archive MAY ignore both |
|
|
1286
1367
|
| `page.pdf` is present but its data does not begin with `%PDF-` | Not an error. The entry is data like any other |
|
|
1287
1368
|
| `index.html` is present without `manifest.json` | **MUST** still extract (§7.1) |
|
|
1288
|
-
| More than one
|
|
1369
|
+
| More than one candidate carries the `sfz-data` identifier once §4.5's tie-break has been applied | **MUST NOT** extract either silently. The tie-break comes first and settles the ordinary pairing: an id-bearing element that is one of §5.1's wrapper rungs wins over a comment, and one that is not a rung loses to it, since the `id` is then something else in the page. What this row forbids is what the tie-break does not reach — two elements, or two comments, or an element and a comment that both survive it. A conforming writer emits one candidate (§5.1), so a second is a payload that escaped its wrapper, most often a nested archive written by a writer that emitted a face bare. Both extract cleanly and check out, and the checksums say nothing about which one the file was built around |
|
|
1289
1370
|
| The recovered region (universal mode) disagrees with the same bytes read directly, in the EOCD's two comment-length bytes only | Expected, not an error. A recovered region always declares a zero-length comment (§4.5), so it differs here from any archive written in the declared form (§4.2). Compare the two only up to those bytes |
|
|
1290
1371
|
| The recovered region (universal mode) disagrees with the same bytes read directly, anywhere else | The file is not well-formed, whichever side is at fault, and a reader that has both MUST NOT silently merge them or pick per entry. Prefer the direct read — it is the writer's own output, where the recovered region is a reconstruction of it — and surface the disagreement rather than displaying either as intact |
|
|
1291
1372
|
|
|
@@ -1485,6 +1566,7 @@ predicts.
|
|
|
1485
1566
|
| August 2026 | Core 1.5.110: a PDF or PNG face whose payload names every rung is dropped instead of written bare (§5.1). Found by nesting an archive inside itself as both faces: the fifth level exhausts the ladder, and readers then extracted the fourth level's archive — checksums intact, no way to tell (§7.4) |
|
|
1486
1567
|
| August 2026 | Core 1.5.110: a PNG face leaving the comment rung on its checksum resumes the rung search instead of taking the next rung untested (§5.1). Taking it put a payload holding `</script>` on the script rung, where its own bytes closed the wrapper 93 bytes in and left the image data, the chunk framing and the whole ZIP region to the parser |
|
|
1487
1568
|
| August 2026 | Core 1.5.110: `<svg><![CDATA[` joins the ladder above `<plaintext>` (§5.1) — the one rung whose terminator, `]]>`, real payloads rarely carry. It gives a payload naming every element rung somewhere to go that does not cost the appended-data placement, and moves the self-nesting limit from the fifth level to the sixth |
|
|
1569
|
+
| August 2026 | Core 1.5.115: password-protected archives withhold the provenance comment and the canonical link as well (§5.6). Both wrote the page's own URL into the prologue, beside the title that was already withheld, so the address the archive was saved from stayed in the clear |
|
|
1488
1570
|
|
|
1489
1571
|
This document was itself revised in August 2026, against core 1.5.108, after several
|
|
1490
1572
|
independent reviews. One of them was a reader built from this specification alone, with
|
|
@@ -1506,3 +1588,18 @@ document; the limits of the reconstructed-`page.pdf` CRC check (§4.5); the dura
|
|
|
1506
1588
|
ranking of the faces (§1.1); what each face costs a writer (§6); and the silent loss of
|
|
1507
1589
|
the other faces to a pipeline that repacks the file (§7.2). One review found a live
|
|
1508
1590
|
defect rather than a documentation one, the non-monotone retry step recorded above.
|
|
1591
|
+
|
|
1592
|
+
A later pass found four places where the document contradicted itself or the standard
|
|
1593
|
+
it cites: §7.3 stated that source URLs stay readable under a password while §5.6 said
|
|
1594
|
+
the writer withholds them, §4.5 called the recovered region exact while excluding two
|
|
1595
|
+
bytes from it, §7.4 rejected a duplicate identifier that §4.5 resolves by tie-break,
|
|
1596
|
+
and §2.1 described the HTML encoding prescan as mandatory and 1024 bytes wide when the
|
|
1597
|
+
standard makes it optional and only encourages that bound. None of the four changes
|
|
1598
|
+
what a writer emits or a reader accepts.
|
|
1599
|
+
|
|
1600
|
+
The same pass added the two boundaries universal mode had left unstated: that it
|
|
1601
|
+
recovers the region only where the declared charset is what decided the decoding, a
|
|
1602
|
+
BOM, a user override and a transport-layer charset all outranking it — narrow in
|
|
1603
|
+
practice, since the raw read comes first and no encoding applies to it (§2.1) — and that
|
|
1604
|
+
the recovery payload's 32-bit length field caps the region below 2^32 bytes, with
|
|
1605
|
+
engine string limits binding well before that (§5.5).
|
|
@@ -41,7 +41,9 @@ const REGEXP_COMMA = /\s*,\s*/;
|
|
|
41
41
|
const REGEXP_DASH = /-/;
|
|
42
42
|
const REGEXP_QUESTION_MARK = /\?/g;
|
|
43
43
|
const REGEXP_STARTS_U_PLUS = /^U\+/i;
|
|
44
|
-
const REGEXP_CUSTOM_PROPERTY = /var\((--[^)
|
|
44
|
+
const REGEXP_CUSTOM_PROPERTY = /var\(\s*(--[^\s,)]+)\s*(?:,[^)]*)?\)/g;
|
|
45
|
+
const REGEXP_CUSTOM_PROPERTY_FAMILY = /^var\(\s*(--[^\s,)]+)\s*(?:,(.*))?\)$/;
|
|
46
|
+
const REGEXP_CUSTOM_PROPERTY_NAME = /^--/;
|
|
45
47
|
const VALID_FONT_STYLES = [/^normal$/, /^italic$/, /^oblique$/, /^oblique\s+/];
|
|
46
48
|
// a family name kept when the "font" shorthand cannot be read: it resolves to nothing, so it
|
|
47
49
|
// survives the substitution below and marks the fonts as undetermined instead of unused
|
|
@@ -54,9 +56,19 @@ export {
|
|
|
54
56
|
function process(doc, stylesheets, styles, options) {
|
|
55
57
|
const stats = { rules: { processed: 0, discarded: 0 }, fonts: { processed: 0, discarded: 0 } };
|
|
56
58
|
const fontsInfo = { declared: [], used: [] };
|
|
59
|
+
const customProperties = new Map();
|
|
57
60
|
const workStyleElement = doc.createElement("style");
|
|
58
61
|
let docContent = "";
|
|
59
62
|
doc.body.appendChild(workStyleElement);
|
|
63
|
+
// the custom properties are collected first: a family or a "font" shorthand can be resolved
|
|
64
|
+
// with a property that a later stylesheet, or a style attribute, declares
|
|
65
|
+
stylesheets.forEach(stylesheetInfo => {
|
|
66
|
+
if (stylesheetInfo.stylesheet && stylesheetInfo.stylesheet.children) {
|
|
67
|
+
getCustomPropertiesInfo(stylesheetInfo.stylesheet.children, customProperties);
|
|
68
|
+
}
|
|
69
|
+
});
|
|
70
|
+
styles.forEach(declarations => getCustomProperties(declarations.children, customProperties));
|
|
71
|
+
options = Object.assign({}, options, { customProperties });
|
|
60
72
|
stylesheets.forEach(stylesheetInfo => {
|
|
61
73
|
if (stylesheetInfo.stylesheet) {
|
|
62
74
|
const cssRules = stylesheetInfo.stylesheet.children;
|
|
@@ -77,31 +89,36 @@ function process(doc, stylesheets, styles, options) {
|
|
|
77
89
|
});
|
|
78
90
|
workStyleElement.remove();
|
|
79
91
|
docContent += doc.body.innerText;
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
const matchedVar = familyName.match(/^var\((--.*)\)$/);
|
|
83
|
-
if (matchedVar && matchedVar[1]) {
|
|
84
|
-
const computedFamilyName = globalThis.getComputedStyle(options.doc.body).getPropertyValue(matchedVar[1]);
|
|
85
|
-
return (computedFamilyName && computedFamilyName.split(",").map(name => helper.normalizeFontFamily(name))) || familyName;
|
|
86
|
-
}
|
|
87
|
-
return familyName;
|
|
88
|
-
}));
|
|
89
|
-
fontsInfo.used = fontsInfo.used.map(fontNames => helper.flatten(fontNames));
|
|
90
|
-
}
|
|
92
|
+
fontsInfo.used = fontsInfo.used.map(fontNames => fontNames.map(familyName => resolveFamilyName(familyName, options)));
|
|
93
|
+
fontsInfo.used = fontsInfo.used.map(fontNames => helper.flatten(fontNames));
|
|
91
94
|
const variableFound = fontsInfo.used.find(fontNames => fontNames.find(fontName => fontName.match(/^var\(--/)));
|
|
95
|
+
// an empty list of rendered fonts does not mean the document uses none: every rendered element
|
|
96
|
+
// has a computed font-family, so an empty list means the computed styles could not be read at
|
|
97
|
+
// all. A frame whose contentDocument is unreachable is re-parsed from its srcdoc with
|
|
98
|
+
// DOMParser, and that document is never rendered, so it reports nothing and every face it
|
|
99
|
+
// declares would be dropped
|
|
100
|
+
const usedFontsUnknown = !options.usedFonts || !options.usedFonts.length;
|
|
92
101
|
let unusedFonts, filteredUsedFonts;
|
|
93
|
-
if (
|
|
102
|
+
if (usedFontsUnknown) {
|
|
94
103
|
unusedFonts = [];
|
|
95
104
|
} else {
|
|
96
105
|
filteredUsedFonts = new Map();
|
|
97
|
-
fontsInfo.used.forEach(fontNames => fontNames.forEach(familyName =>
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
106
|
+
fontsInfo.used.forEach(fontNames => fontNames.forEach(familyName =>
|
|
107
|
+
keepDeclaredFontIfRendered(familyName, fontsInfo, filteredUsedFonts, options)));
|
|
108
|
+
// A family named through a value that could not be resolved cannot be looked up in the
|
|
109
|
+
// stylesheets — but the browser resolved it when it drew the page, so whatever it named is
|
|
110
|
+
// in the list of fonts actually rendered, and that list answers for it.
|
|
111
|
+
//
|
|
112
|
+
// Giving up on the whole document instead, which is what an unresolved name used to do,
|
|
113
|
+
// kept every declared face on any page holding one unreadable value anywhere while every
|
|
114
|
+
// other page pruned as usual. That was not a policy about uncertainty: a page that names
|
|
115
|
+
// its families plainly has always dropped a face it had not drawn yet, and only a page
|
|
116
|
+
// whose value happened not to parse was spared. The difference came from a parse failure,
|
|
117
|
+
// so it is the rendered list that decides in both cases now.
|
|
118
|
+
if (variableFound) {
|
|
119
|
+
fontsInfo.declared.forEach(fontInfo =>
|
|
120
|
+
keepDeclaredFontIfRendered(fontInfo.fontFamily, fontsInfo, filteredUsedFonts, options));
|
|
121
|
+
}
|
|
105
122
|
unusedFonts = fontsInfo.declared.filter(fontInfo => !filteredUsedFonts.has(fontInfo.fontFamily));
|
|
106
123
|
}
|
|
107
124
|
const docChars = Array.from(new Set(docContent)).map(char => char.charCodeAt(0)).sort((value1, value2) => value1 - value2);
|
|
@@ -117,6 +134,17 @@ function process(doc, stylesheets, styles, options) {
|
|
|
117
134
|
return stats;
|
|
118
135
|
}
|
|
119
136
|
|
|
137
|
+
// a face is kept when the page declares it and the browser reports having drawn with it: the
|
|
138
|
+
// stylesheets say what exists, the rendered list says what was needed
|
|
139
|
+
function keepDeclaredFontIfRendered(familyName, fontsInfo, filteredUsedFonts, options) {
|
|
140
|
+
if (fontsInfo.declared.find(fontInfo => fontInfo.fontFamily == familyName)) {
|
|
141
|
+
const optionalData = options.usedFonts.filter(fontInfo => fontInfo[0] == familyName);
|
|
142
|
+
if (optionalData.length) {
|
|
143
|
+
filteredUsedFonts.set(familyName, optionalData);
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
120
148
|
function getFontsInfo(cssRules, fontsInfo, options) {
|
|
121
149
|
cssRules.forEach(ruleData => {
|
|
122
150
|
if (ruleData.type == "Atrule" && (ruleData.name == "media" || ruleData.name == "supports" || ruleData.name == "layer" || ruleData.name == "container") && ruleData.block && ruleData.block.children) {
|
|
@@ -141,6 +169,126 @@ function getFontsInfo(cssRules, fontsInfo, options) {
|
|
|
141
169
|
});
|
|
142
170
|
}
|
|
143
171
|
|
|
172
|
+
function getCustomPropertiesInfo(cssRules, customProperties) {
|
|
173
|
+
cssRules.forEach(ruleData => {
|
|
174
|
+
if (ruleData.type == "Atrule" && ruleData.name == "import" && ruleData.prelude && ruleData.prelude.children && ruleData.prelude.children.head.data.importedChildren) {
|
|
175
|
+
getCustomPropertiesInfo(ruleData.prelude.children.head.data.importedChildren, customProperties);
|
|
176
|
+
} else if (ruleData.type == "Atrule" && (ruleData.name == "media" || ruleData.name == "supports" || ruleData.name == "layer" || ruleData.name == "container") && ruleData.block && ruleData.block.children) {
|
|
177
|
+
getCustomPropertiesInfo(ruleData.block.children, customProperties);
|
|
178
|
+
} else if (ruleData.type == "Rule" && ruleData.block && ruleData.block.children) {
|
|
179
|
+
getCustomProperties(ruleData.block.children, customProperties);
|
|
180
|
+
}
|
|
181
|
+
});
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
function getCustomProperties(declarations, customProperties) {
|
|
185
|
+
if (declarations) {
|
|
186
|
+
declarations.forEach(declaration => {
|
|
187
|
+
if (declaration.property && declaration.property.match(REGEXP_CUSTOM_PROPERTY_NAME)) {
|
|
188
|
+
try {
|
|
189
|
+
const value = cssTree.generate(declaration.value).trim();
|
|
190
|
+
if (value) {
|
|
191
|
+
let values = customProperties.get(declaration.property);
|
|
192
|
+
if (!values) {
|
|
193
|
+
values = new Set();
|
|
194
|
+
customProperties.set(declaration.property, values);
|
|
195
|
+
}
|
|
196
|
+
values.add(value);
|
|
197
|
+
}
|
|
198
|
+
// eslint-disable-next-line no-unused-vars
|
|
199
|
+
} catch (error) {
|
|
200
|
+
// ignored
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
});
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function getCustomPropertyValues(name, options) {
|
|
208
|
+
let values;
|
|
209
|
+
if (globalThis.getComputedStyle && options.doc) {
|
|
210
|
+
const computedValue = globalThis.getComputedStyle(options.doc.body).getPropertyValue(name);
|
|
211
|
+
if (computedValue && computedValue.trim()) {
|
|
212
|
+
values = [computedValue];
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
if (!values) {
|
|
216
|
+
// the property is not inherited by the body: it is declared on a descendant, or in a
|
|
217
|
+
// media query that does not apply, so the value seen by the element using it cannot be
|
|
218
|
+
// determined here. Every value declared for it in the document is taken as a candidate,
|
|
219
|
+
// which still discards the fonts named by none of them
|
|
220
|
+
const declaredValues = options.customProperties && options.customProperties.get(name);
|
|
221
|
+
if (declaredValues && declaredValues.size) {
|
|
222
|
+
values = Array.from(declaredValues);
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
return values;
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
function resolveFamilyName(familyName, options, resolvedProperties = new Set()) {
|
|
229
|
+
const matchedVar = familyName.match(REGEXP_CUSTOM_PROPERTY_FAMILY);
|
|
230
|
+
if (matchedVar) {
|
|
231
|
+
const propertyName = matchedVar[1];
|
|
232
|
+
const fallback = matchedVar[2];
|
|
233
|
+
// a property naming itself, directly or through another one, would resolve for ever: the
|
|
234
|
+
// chain already walked is carried down the branch so it stops instead
|
|
235
|
+
if (!resolvedProperties.has(propertyName)) {
|
|
236
|
+
const properties = new Set(resolvedProperties);
|
|
237
|
+
properties.add(propertyName);
|
|
238
|
+
const values = getCustomPropertyValues(propertyName, options);
|
|
239
|
+
if (values) {
|
|
240
|
+
const families = helper.flatten(values.map(value => splitFamilyNames(value, options, properties)));
|
|
241
|
+
const fallbackFamilies = fallback ? splitFamilyNames(fallback, options, properties) : [];
|
|
242
|
+
// the browser takes the property or the fallback, so knowing one branch is not
|
|
243
|
+
// knowing the value: a var() left unresolved in either one keeps the family
|
|
244
|
+
// undetermined, exactly as it was before the nested one could be read at all
|
|
245
|
+
if (!families.concat(fallbackFamilies).some(testUnresolvedFamilyName)) {
|
|
246
|
+
return families.concat(fallbackFamilies);
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
return familyName;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
function testUnresolvedFamilyName(familyName) {
|
|
255
|
+
return typeof familyName == "string" && familyName.startsWith("var(");
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
function splitFamilyNames(value, options, resolvedProperties) {
|
|
259
|
+
const familyNames = splitValues(value).map(familyName => normalizeFamilyName(familyName)).filter(familyName => familyName);
|
|
260
|
+
return options
|
|
261
|
+
? helper.flatten(familyNames.map(familyName => resolveFamilyName(familyName, options, resolvedProperties)))
|
|
262
|
+
: familyNames;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
// a family list is split on its top-level commas only: the commas inside a var() belong to that
|
|
266
|
+
// var(), and the ones inside a quoted name belong to the name. Splitting on every comma is what
|
|
267
|
+
// made a var() nested in a fallback unreadable, and it also broke a family named "Foo, Bar"
|
|
268
|
+
function splitValues(value) {
|
|
269
|
+
const values = [];
|
|
270
|
+
let depth = 0, quote, start = 0;
|
|
271
|
+
for (let index = 0; index < value.length; index++) {
|
|
272
|
+
const character = value.charAt(index);
|
|
273
|
+
if (quote) {
|
|
274
|
+
if (character == quote && value.charAt(index - 1) != "\\") {
|
|
275
|
+
quote = null;
|
|
276
|
+
}
|
|
277
|
+
} else if (character == "\"" || character == "'") {
|
|
278
|
+
quote = character;
|
|
279
|
+
} else if (character == "(") {
|
|
280
|
+
depth++;
|
|
281
|
+
} else if (character == ")") {
|
|
282
|
+
depth--;
|
|
283
|
+
} else if (character == "," && !depth) {
|
|
284
|
+
values.push(value.substring(start, index));
|
|
285
|
+
start = index + 1;
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
values.push(value.substring(start));
|
|
289
|
+
return values;
|
|
290
|
+
}
|
|
291
|
+
|
|
144
292
|
function filterUnusedFonts(cssRules, declaredFonts, unusedFonts, filteredUsedFonts, docChars) {
|
|
145
293
|
const removedRules = [];
|
|
146
294
|
for (let cssRule = cssRules.head; cssRule; cssRule = cssRule.next) {
|
|
@@ -266,7 +414,7 @@ function getFontFamilyNames(declarations, options) {
|
|
|
266
414
|
} else {
|
|
267
415
|
fontFamilyName = cssTree.generate(fontFamilyName.data.value);
|
|
268
416
|
if (fontFamilyName) {
|
|
269
|
-
fontFamilyNames.push(
|
|
417
|
+
fontFamilyNames.push(normalizeFamilyName(fontFamilyName));
|
|
270
418
|
}
|
|
271
419
|
}
|
|
272
420
|
}
|
|
@@ -280,7 +428,7 @@ function getFontFamilyNames(declarations, options) {
|
|
|
280
428
|
value = cssTree.parse(resolvedFontValue, { context: "value" });
|
|
281
429
|
}
|
|
282
430
|
const parsedFont = fontPropertyParser.parse(value);
|
|
283
|
-
parsedFont.family.forEach(familyName => fontFamilyNames.push(
|
|
431
|
+
parsedFont.family.forEach(familyName => fontFamilyNames.push(normalizeFamilyName(familyName)));
|
|
284
432
|
// eslint-disable-next-line no-unused-vars
|
|
285
433
|
} catch (error) {
|
|
286
434
|
// the shorthand is unreadable, and dropping it here would count the fonts it names as
|
|
@@ -294,12 +442,17 @@ function getFontFamilyNames(declarations, options) {
|
|
|
294
442
|
}
|
|
295
443
|
|
|
296
444
|
function resolveCustomProperties(value, options) {
|
|
297
|
-
|
|
298
|
-
const
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
return
|
|
302
|
-
}
|
|
445
|
+
return value.replace(REGEXP_CUSTOM_PROPERTY, (property, name) => {
|
|
446
|
+
const values = getCustomPropertyValues(name, options);
|
|
447
|
+
// the shorthand is parsed as a whole, so it can only be substituted with a single value:
|
|
448
|
+
// with several candidates the families it names stay undetermined
|
|
449
|
+
return values && values.length == 1 ? values[0] : property;
|
|
450
|
+
});
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
function normalizeFamilyName(familyName = "") {
|
|
454
|
+
// custom property names are case-sensitive, unlike family names
|
|
455
|
+
return familyName.match(REGEXP_CUSTOM_PROPERTY_FAMILY) ? familyName.trim() : helper.normalizeFontFamily(familyName);
|
|
303
456
|
}
|
|
304
457
|
|
|
305
458
|
function parseFamilyNames(fontFamilyNameTokenData, fontFamilyNames) {
|
|
@@ -307,16 +460,19 @@ function parseFamilyNames(fontFamilyNameTokenData, fontFamilyNames) {
|
|
|
307
460
|
while (nextToken) {
|
|
308
461
|
if (nextToken.data.type == "Identifier") {
|
|
309
462
|
let familyName = nextToken.data.name;
|
|
463
|
+
// an unquoted family name is a sequence of identifiers, and the walk has to resume
|
|
464
|
+
// after the last of them: resuming after the first pushes every word but that one
|
|
465
|
+
// again as a family of its own, so "Foo Bar" also claims a font-face named "Bar"
|
|
310
466
|
let nextIdentifierToken = nextToken.next;
|
|
311
|
-
while (nextIdentifierToken && nextIdentifierToken.data.type
|
|
467
|
+
while (nextIdentifierToken && nextIdentifierToken.data.type == "Identifier") {
|
|
312
468
|
familyName += " " + nextIdentifierToken.data.name;
|
|
313
469
|
nextIdentifierToken = nextIdentifierToken.next;
|
|
314
470
|
}
|
|
315
471
|
fontFamilyNames.push(helper.normalizeFontFamily(familyName));
|
|
316
|
-
nextToken =
|
|
472
|
+
nextToken = nextIdentifierToken;
|
|
317
473
|
} else if (nextToken.data.type == "Function" && nextToken.data.name == "var" && nextToken.data.children) {
|
|
318
474
|
const varName = nextToken.data.children.head.data.name;
|
|
319
|
-
fontFamilyNames.push(
|
|
475
|
+
fontFamilyNames.push("var(" + varName + ")");
|
|
320
476
|
let nextValueToken = nextToken.data.children.head.next;
|
|
321
477
|
while (nextValueToken && nextValueToken.data.type == "Operator" && nextValueToken.data.value == ",") {
|
|
322
478
|
nextValueToken = nextValueToken.next;
|
|
@@ -326,7 +482,9 @@ function parseFamilyNames(fontFamilyNameTokenData, fontFamilyNames) {
|
|
|
326
482
|
if (fallbackToken.data.children) {
|
|
327
483
|
parseFamilyNames(fallbackToken.data, fontFamilyNames);
|
|
328
484
|
} else {
|
|
329
|
-
|
|
485
|
+
// the fallback of a var() is parsed as a single raw token, so it still has to
|
|
486
|
+
// be split into the list of families it may hold
|
|
487
|
+
splitFamilyNames(String(fallbackToken.data.value)).forEach(familyName => fontFamilyNames.push(familyName));
|
|
330
488
|
}
|
|
331
489
|
}
|
|
332
490
|
nextToken = nextToken.next;
|
|
@@ -51,6 +51,7 @@ const SELECTOR_LIST_CONTEXT = "selectorList";
|
|
|
51
51
|
const STYLESHEET_CONTEXT = "stylesheet";
|
|
52
52
|
const SELECTOR_CONTEXT = "selector";
|
|
53
53
|
const DECLARATION_LIST_CONTEXT = "declarationList";
|
|
54
|
+
const UNKNOWN_PROPERTY_ERROR_NAME = "SyntaxReferenceError";
|
|
54
55
|
const PARSE_CSS_ERROR_MESSAGE = "Failed to parse CSS";
|
|
55
56
|
const QSA_ERROR_MESSAGE = "Failed to match selector";
|
|
56
57
|
const PRELUDE_SEPARATOR = ",";
|
|
@@ -69,9 +70,15 @@ const INVALID_CSS_ESCAPE_TEST = /\\(?![0-9a-fA-F]{1,6}\s|[^0-9a-zA-Z])/;
|
|
|
69
70
|
const ANONYMOUS_LAYER_PLACEHOLDER = "\u0000";
|
|
70
71
|
|
|
71
72
|
export {
|
|
72
|
-
process
|
|
73
|
+
process,
|
|
74
|
+
isUnsupportedPropertyValue
|
|
73
75
|
};
|
|
74
76
|
|
|
77
|
+
function isUnsupportedPropertyValue(property, value) {
|
|
78
|
+
const match = cssTree.lexer.matchProperty(property, value);
|
|
79
|
+
return Boolean(!match.matched && match.error && match.error.name !== UNKNOWN_PROPERTY_ERROR_NAME);
|
|
80
|
+
}
|
|
81
|
+
|
|
75
82
|
function process(doc, stylesheets) {
|
|
76
83
|
const docContext = {
|
|
77
84
|
doc,
|
|
@@ -493,7 +500,7 @@ function collectDeclarationItemsForElement(element, docContext) {
|
|
|
493
500
|
isInvalidValue = value.children.head.data.name.startsWith(VENDOR_PREFIX) || INVALID_CSS_ESCAPE_TEST.test(value.children.head.data.name);
|
|
494
501
|
} if (!property.startsWith(VENDOR_PREFIX) && value.children.head.data.value) {
|
|
495
502
|
try {
|
|
496
|
-
isInvalidValue =
|
|
503
|
+
isInvalidValue = isUnsupportedPropertyValue(property, value);
|
|
497
504
|
} catch {
|
|
498
505
|
// ignored
|
|
499
506
|
}
|
package/package.json
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "single-file-core",
|
|
3
|
-
"version": "1.5.
|
|
3
|
+
"version": "1.5.117",
|
|
4
4
|
"description": "SingleFile Core",
|
|
5
5
|
"author": "Gildas Lormeau",
|
|
6
6
|
"license": "AGPL-3.0-or-later",
|
|
7
7
|
"scripts": {
|
|
8
|
-
"test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js",
|
|
8
|
+
"test": "deno run --allow-read test/sfz-harness/format-rules.js && deno run --allow-read test/sfz-harness/stored-trigger.js && deno run --allow-read test/sfz-harness/check-determinism.js && deno run --allow-read test/sfz-harness/option-wiring.js && deno run --allow-read test/sfz-harness/css-property-filter.js && deno run --allow-read test/sfz-harness/adopted-stylesheets-hook.js && deno run --allow-read test/sfz-harness/css-fonts-minifier.js && deno run --allow-read test/sfz-harness/inlined-functions.js",
|
|
9
9
|
"bump-patch": "npm version patch --no-git-tag-version && npm run bump-commit",
|
|
10
10
|
"bump-minor": "npm version minor --no-git-tag-version && npm run bump-commit",
|
|
11
11
|
"bump-major": "npm version major --no-git-tag-version && npm run bump-commit",
|
|
@@ -23,13 +23,34 @@
|
|
|
23
23
|
|
|
24
24
|
/* global DOMParser, setTimeout */
|
|
25
25
|
|
|
26
|
-
import { getDoctypeString } from "./../../core/lib/doctype.js";
|
|
27
|
-
|
|
28
26
|
export {
|
|
29
27
|
display
|
|
30
28
|
};
|
|
31
29
|
|
|
30
|
+
// the functions below are serialized with toString() and pasted into the page the archive
|
|
31
|
+
// generates, where module scope does not exist: everything they call must be declared inside
|
|
32
|
+
// them. An import would survive bundling and then be an undefined free variable at runtime
|
|
32
33
|
async function display(document, docContent, { disableFramePointerEvents, inPlace } = {}) {
|
|
34
|
+
function getDoctypeString(doc) {
|
|
35
|
+
const docType = doc.doctype;
|
|
36
|
+
let docTypeString = "";
|
|
37
|
+
if (docType) {
|
|
38
|
+
docTypeString = "<!DOCTYPE " + docType.nodeName;
|
|
39
|
+
if (docType.publicId) {
|
|
40
|
+
docTypeString += " PUBLIC \"" + docType.publicId + "\"";
|
|
41
|
+
if (docType.systemId) {
|
|
42
|
+
docTypeString += " \"" + docType.systemId + "\"";
|
|
43
|
+
}
|
|
44
|
+
} else if (docType.systemId) {
|
|
45
|
+
docTypeString += " SYSTEM \"" + docType.systemId + "\"";
|
|
46
|
+
}
|
|
47
|
+
if (docType.internalSubset) {
|
|
48
|
+
docTypeString += " [" + docType.internalSubset + "]";
|
|
49
|
+
}
|
|
50
|
+
docTypeString += ">";
|
|
51
|
+
}
|
|
52
|
+
return docTypeString;
|
|
53
|
+
}
|
|
33
54
|
docContent = docContent.replace(/<noscript/gi, "<template disabled-noscript");
|
|
34
55
|
docContent = docContent.replace(/<\/noscript/gi, "</template");
|
|
35
56
|
const doc = (new DOMParser()).parseFromString(docContent, "text/html");
|