single-file-core 1.5.120 → 1.5.121
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
**Status: draft.** This document specifies the SingleFile archive, the polyglot file
|
|
4
4
|
format produced by [SingleFile](https://github.com/gildas-lormeau/SingleFile) when it
|
|
5
5
|
saves a page as a ZIP archive. It is written against the reference implementation,
|
|
6
|
-
[single-file-core](https://github.com/gildas-lormeau/single-file-core) 1.5.
|
|
6
|
+
[single-file-core](https://github.com/gildas-lormeau/single-file-core) 1.5.120
|
|
7
7
|
(`processors/compression/`), and every byte-level statement has been verified on
|
|
8
8
|
generated specimen files.
|
|
9
9
|
|
|
@@ -62,8 +62,9 @@ little software as possible. Each way of opening the file has a simpler fallback
|
|
|
62
62
|
is hidden by construction, so the browser displays neither the page nor raw
|
|
63
63
|
archive bytes (§4.1).
|
|
64
64
|
3. Renamed to `.zip`, the file opens in a ZIP tool; the page and each resource are
|
|
65
|
-
ordinary entries.
|
|
66
|
-
|
|
65
|
+
ordinary entries. Two measured readers refuse a self-extracting variant even from
|
|
66
|
+
seekable input; the ranking below names them. A forward-only reader refuses it too,
|
|
67
|
+
and is a non-goal (§1.2).
|
|
67
68
|
4. Renamed to `.pdf` or `.png` (when those faces are present), the file opens in a PDF
|
|
68
69
|
viewer or an image viewer.
|
|
69
70
|
|
|
@@ -119,7 +120,8 @@ Three consequences shape everything below:
|
|
|
119
120
|
- **Multi-page archives.** The reference implementation can bundle several saved
|
|
120
121
|
pages into one archive behind a routing bootstrap (`multiPageArchive`). This
|
|
121
122
|
version of the document specifies single-page archives only; the multi-page
|
|
122
|
-
layout is out of scope
|
|
123
|
+
layout is out of scope, and so are the regions it adds to the prologue, which this
|
|
124
|
+
document does not describe.
|
|
123
125
|
- **Confidentiality outside the ZIP entries.** A password encrypts ZIP entry contents
|
|
124
126
|
only (AES). The PDF and PNG faces render the page content and are plaintext by
|
|
125
127
|
design; the writer withholds what it can without breaking a face, as described in
|
|
@@ -172,6 +174,11 @@ every acquisition path including the ones that read raw bytes and could return i
|
|
|
172
174
|
HTML face, so the option does not apply. The *Specimen* column names the measured
|
|
173
175
|
reference files this document cites; §8 records how to regenerate them.
|
|
174
176
|
|
|
177
|
+
Other writer options shape the file without adding a face: `preventAppendedData` and
|
|
178
|
+
`declareAppendedData` (§4.2, §5.2), `includeBOM` (§3.1), `insertTextBody` (§4.6),
|
|
179
|
+
`password` (§5.6), `createRootDirectory` (§7.1), and the head-element switches
|
|
180
|
+
`insertCanonicalLink`, `insertMetaNoIndex` and `insertMetaCSP` (§3.1).
|
|
181
|
+
|
|
175
182
|
Notes on composition:
|
|
176
183
|
|
|
177
184
|
- **Universal mode requires the HTML face** (it is a property of the bootstrap) and is
|
|
@@ -238,8 +245,8 @@ they are out of reach. Taking the three in turn:
|
|
|
238
245
|
(`includeBOM`), where nothing depends on the declared charset.
|
|
239
246
|
- A **user override** is the one no software can prevent, and the rarest.
|
|
240
247
|
|
|
241
|
-
On `file:` URLs the bootstrap goes straight to page-text extraction, since
|
|
242
|
-
|
|
248
|
+
On `file:` URLs the bootstrap goes straight to page-text extraction, since it attempts
|
|
249
|
+
no raw read there (§4.1) — but there is also no transport layer, so the first of the
|
|
243
250
|
three cannot arise on the very path that depends on the charset most.
|
|
244
251
|
|
|
245
252
|
The failure is safe rather than silent, which is why the precondition is worth stating
|
|
@@ -281,9 +288,9 @@ The declared charset governs the **whole document**, not only the regions the fo
|
|
|
281
288
|
reasons about. Everything the parser reads is decoded with it, the bootstrap script
|
|
282
289
|
included, and the writer's own code is therefore subject to the same single-byte
|
|
283
290
|
decoding as the page it carries. In universal mode the bootstrap MUST contain no
|
|
284
|
-
character outside
|
|
291
|
+
character outside ASCII.
|
|
285
292
|
|
|
286
|
-
Unlike the `<title>`
|
|
293
|
+
Unlike the `<title>` (§4.6), it cannot be rescued by
|
|
287
294
|
character references. A `<script>` element's content is script data, a tokenizer state
|
|
288
295
|
that does not resolve them: `☺` written there stays seven literal characters and
|
|
289
296
|
reaches the program as seven characters. The escape has to happen one level down, in
|
|
@@ -311,7 +318,8 @@ conventions for humans and pickers; **readers MUST NOT rely on the file name**.
|
|
|
311
318
|
face is discoverable from the bytes alone: PNG and PDF by their signatures, the ZIP
|
|
312
319
|
face by its End Of Central Directory record, and the HTML face by an `<html` start tag
|
|
313
320
|
occurring before the first local file header — inside the first `tEXt` chunk's data in
|
|
314
|
-
the PNG variants, where the markup begins after the chunk's keyword
|
|
321
|
+
the PNG variants, where the markup begins after the chunk's keyword and its NUL
|
|
322
|
+
separator. Inside the
|
|
315
323
|
archive, the `index.html` and `manifest.json` entries mark it as a saved page; a
|
|
316
324
|
reader should identify it that way (§7.1). The self-extracting variants are told apart
|
|
317
325
|
the same way: only a universal file carries an `<sfz-extra-data>` element.
|
|
@@ -319,9 +327,10 @@ the same way: only a universal file carries an `<sfz-extra-data>` element.
|
|
|
319
327
|
## 3. The byte map
|
|
320
328
|
|
|
321
329
|
Unless a row states otherwise, the layouts below are measured from specimen files
|
|
322
|
-
saved from `example.com` (the generation commands are in §8)
|
|
323
|
-
|
|
324
|
-
|
|
330
|
+
saved from `example.com` (the generation commands are in §8). The relocated row covers
|
|
331
|
+
two cases with one layout, `preventAppendedData` and a payload over 64 KB: the first is
|
|
332
|
+
measured on the relocated specimen, the second derived from the writer rules, because
|
|
333
|
+
such a payload requires an archive too large for a readable specimen. The figure below shows
|
|
325
334
|
the regions and their order; the glossary of §3.1 is the normative list, and it states
|
|
326
335
|
in text everything the figure conveys.
|
|
327
336
|
|
|
@@ -339,21 +348,21 @@ face adds, then the regions the PNG face adds.
|
|
|
339
348
|
|
|
340
349
|
| Region | Producer | Present | Contents |
|
|
341
350
|
|---|---|---|---|
|
|
342
|
-
| `html-prologue` | HTML | HTML face | Doctype, the root element start tag, `<meta charset>`, an optional implementation-defined comment, title, optional head elements (canonical link, `robots` meta, viewport, Content-Security-Policy), minimal CSS, `<body hidden>`, wait/error messages, optional
|
|
351
|
+
| `html-prologue` | HTML | HTML face | Doctype, the root element start tag, `<meta charset>`, an optional implementation-defined comment, title, optional head elements (canonical link, `robots` meta, viewport, Content-Security-Policy), minimal CSS, `<body hidden>`, wait/error messages, optional text body (§4.6). The leading comment, the title, the canonical link and the text body are withheld when a password is set (§5.6). In the plain variant an optional UTF-8 BOM MAY precede the doctype (`includeBOM`); universal and PNG variants never carry one, and the reference writer ignores the option there. In the PNG variants the region is split: everything through `<body hidden>` is the data of the `tEXt "PNG"` chunk, while the messages and the optional text body follow the `tEXt "ZIP"` chunk header; the doctype and the leading comment are dropped. |
|
|
343
352
|
| `bootstrap` | HTML | HTML face | One inline `<script>`: the embedded ZIP reader, the extractor, the display routine, and the content-acquisition logic (§4.1). In universal mode its bytes MUST be pure ASCII, since the declared charset decodes this region like any other and character references do not apply inside script data (§2.1). The wrapper start tag that opens the ZIP region follows it, directly or after a relocated `extra-data`. |
|
|
344
|
-
| `<!--` / `-->` | HTML | HTML face | The wrapper tag pair hiding a binary region from the HTML parser — comment tags by default, another pair when the hidden bytes defeat them — which `-->` is only the commonest way to do, the full test being `<!--`,
|
|
353
|
+
| `<!--` / `-->` | HTML | HTML face | The wrapper tag pair hiding a binary region from the HTML parser — comment tags by default, another pair when the hidden bytes defeat them — which `-->` is only the commonest way to do, the full test being `<!--`, `--!>`, a trailing `<!-` and, for the PNG payload, a leading `>` or `->` (§5.1). Drawn at each opening and closing position. The close tag is absent whenever the recovery payload is relocated (§5.2): under `preventAppendedData`, when the payload outgrows the appended-data budget, or on the `<plaintext>` wrapper which cannot close. No markup then follows the archive and the wrapper runs to end-of-file. That does not mean the file ends at the EOCD — the PNG face's tail still follows, inside the wrapper, where it parses as text (§5.1). |
|
|
345
354
|
| `zip-entries` | ZIP | always | The archive's local file headers and entry data, written by the ZIP writer. The central directory of an archive written by the reference writer lists `index.html` (the page) first, then `manifest.json` (a JSON description of the archive: original URL, title, save time, resource-to-URL map — informative; the page displays without it), then the resources; the *physical* order of the local headers inside the region is not guaranteed to match, and readers MUST NOT rely on either order — entries are addressed by name (§7.1). |
|
|
346
355
|
| `central-directory · eocd` | ZIP | always | The central-directory records followed by the End Of Central Directory record. All offsets are absolute file positions (§5.3). In the HTML+PDF variants the EOCD accounts for the injected `pdf-central-record` (how the writer achieves that is §6). |
|
|
347
356
|
| `extra-data` | extractor | universal | `<sfz-extra-data>` element holding the base64, deflate-compressed recovery payload (§5.5). It always sits outside the wrapper, so it parses as a real element the extractor can address. Normal placement: after the EOCD, between the wrapper close tag and the end tags. Relocated placement, used when the payload exceeds the 64 KB appended-data window or `preventAppendedData` is set: immediately before the wrapper start tag. In the relocated form the element is followed by space padding: its room is reserved before the archive is written, because the region precedes the ZIP data and resizing it would shift every central-directory offset (§6). Neither placement carries positional meaning — the extractor finds the ZIP region by identifier, not relative to this element (§4.5). |
|
|
348
|
-
| `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted
|
|
357
|
+
| `</body></html>` | HTML | HTML face | The end tags closing the document after the wrapper close tag. Omitted whenever the recovery payload is relocated (§5.2), and in the PNG variants so the file can end with the PNG tail. |
|
|
349
358
|
| `pdf-local-header` | ZIP | PDF face with HTML | The hand-built local file header for `page.pdf` (STORE, checksum precomputed, language encoding flag set as on every other entry — §5.8), written immediately before the PDF document so ZIP readers see an ordinary entry whose data is the PDF (§6). |
|
|
350
359
|
| `pdf-document` | PDF | PDF face | The raw PDF bytes. With the HTML face, wrapped together with `pdf-local-header` in a wrapper tag pair inside `html-prologue`, placed so `%PDF-` starts at offset 1024 or lower — the range PDF readers search for the header, which is what lets a PDF document start after other bytes at all (§4.3). Without the HTML face and without the PNG face, the file simply *starts* with the PDF document, as prepended data the ZIP face tolerates; `page.pdf` is then not an archive entry at all — no local header, no central record. |
|
|
351
360
|
| `pdf-central-record` | ZIP | PDF face with HTML | The central-directory record for `page.pdf`, injected *before* the writer's own central directory. The start of the central directory is the one place a record can be added without moving any offset the writer already committed, and it makes `page.pdf` the first entry ZIP tools list (§6). |
|
|
352
361
|
| `png-signature · IHDR` | PNG | PNG face | The 8-byte PNG signature and the `IHDR` chunk declaring the source image's dimensions — the first 33 bytes of the file. |
|
|
353
|
-
| `tEXt "PNG"` | PNG | PNG face with HTML | The
|
|
354
|
-
| `tEXt "PDF"` | PNG | PNG + PDF faces without HTML | The length, type
|
|
355
|
-
| `pixel-data chunks` | PNG | PNG face |
|
|
356
|
-
| `tEXt "ZIP"` | PNG | PNG face | The length, type
|
|
362
|
+
| `tEXt "PNG"` | PNG | PNG face with HTML | The 12 header bytes of the first `tEXt` chunk: the 4-byte big-endian length, the type, the keyword and its NUL separator. Its data is `html-prologue` (with the PDF face, the embedded PDF document rides inside it too), ending with the wrapper start tag. |
|
|
363
|
+
| `tEXt "PDF"` | PNG | PNG + PDF faces without HTML | The 12 header bytes (length, type, keyword, NUL separator) of a `tEXt` chunk whose data is the raw PDF document. Written only when the PNG and PDF faces combine without HTML — with the HTML face the PDF rides inside `tEXt "PNG"` instead — and placed right after `IHDR` so `%PDF-` stays within the header scan window (§4.3). |
|
|
364
|
+
| `pixel-data chunks` | PNG | PNG face | Every chunk of the source image between `IHDR` and `IEND`, ancillary chunks included, copied unmodified. The reference writer takes `IHDR` as the 25 bytes after the signature and `IEND` as the last 12 bytes of the source, so a source image with bytes after `IEND` is not supported. With the HTML face the chunks sit inside the wrapper so the HTML parser skips them. |
|
|
365
|
+
| `tEXt "ZIP"` | PNG | PNG face | The 12 header bytes (length, type, keyword, NUL separator) of the archive's own `tEXt` chunk — the second one when the HTML face or the PDF face put a chunk ahead of it, the only one otherwise. Its declared length covers everything from there up to but not including the trailing chunk CRC, as a PNG chunk length always does, so the PNG decoder skips the archive — and, with the HTML face, the bootstrap and the appended data — as the data of one chunk. With the HTML face, the wrapper opened at the end of `tEXt "PNG"` closes immediately after these bytes: its content is the first chunk's CRC, the pixel-data chunks and this chunk's own header, and the prologue resumes as markup directly after the close tag. |
|
|
357
366
|
| `crc · IEND` | PNG | PNG face | The `tEXt "ZIP"` chunk's CRC, computed once the archive bytes are final (§6), followed by the empty `IEND` chunk — the last bytes of the file (PNG requires `IEND` to end the stream, which is why the PNG variants drop the end tags). |
|
|
358
367
|
|
|
359
368
|
The reader-by-reader interpretation of these regions is §4; the mechanics that keep
|
|
@@ -390,16 +399,17 @@ The binary regions are kept out of the rendered page by the wrapper tags. The
|
|
|
390
399
|
default wrapper is an HTML comment, and the HTML standard defines exactly which
|
|
391
400
|
character sequences terminate one (`-->`, and the recovery form `--!>`); the writer
|
|
392
401
|
MUST select a wrapper only after checking the bytes it must hide against that
|
|
393
|
-
wrapper's patterns (the
|
|
394
|
-
PNG payloads, §5.1), so hiding relies on
|
|
402
|
+
wrapper's patterns (the same test for every payload, with a shorter ladder for the
|
|
403
|
+
PDF and PNG payloads and one extra check for the PNG one, §5.1), so hiding relies on
|
|
404
|
+
normative parsing behavior. When no
|
|
395
405
|
wrapper fits a PDF or PNG payload, the face is dropped rather than emitted bare
|
|
396
406
|
(§5.1). Some binary content always sits *outside* a
|
|
397
407
|
wrapper: in the PNG variants, the signature, IHDR and chunk framing bytes that
|
|
398
408
|
precede the root element start tag decode to a short run of text that HTML error recovery
|
|
399
409
|
places in the (hidden) body. The backstop for all these cases is the prologue: it
|
|
400
|
-
declares `<body hidden
|
|
401
|
-
wait and error messages
|
|
402
|
-
with or without scripting.
|
|
410
|
+
declares `<body hidden>`, which only the bootstrap clears, and a stylesheet that
|
|
411
|
+
suppresses everything except the wait and error messages once the body is shown, so
|
|
412
|
+
the page comes up blank rather than showing raw bytes, with or without scripting.
|
|
403
413
|
|
|
404
414
|
The bootstrap script runs at parse time and proceeds in three stages:
|
|
405
415
|
|
|
@@ -412,7 +422,9 @@ The bootstrap script runs at parse time and proceeds in three stages:
|
|
|
412
422
|
large archive displays without downloading the ZIP region in full); otherwise it
|
|
413
423
|
downloads the whole file. When the header probe fails it falls back to page-text
|
|
414
424
|
extraction; so does a failure of the full download itself, past the probe, which
|
|
415
|
-
is why the probe leaves the document in place.
|
|
425
|
+
is why the probe leaves the document in place. A failure inside range reading,
|
|
426
|
+
past the probe, is not caught the same way: it goes to the error message. Only
|
|
427
|
+
when every applicable rung fails does the error
|
|
416
428
|
message appear, with recovery instructions that differ by variant (§2).
|
|
417
429
|
2. **Extract.** The embedded ZIP reader reads the archive through the ZIP lens
|
|
418
430
|
(§4.2) and rebuilds the page: text entries are decoded, binary entries become
|
|
@@ -462,9 +474,10 @@ A listing shows `page.pdf` first (when the PDF face is present with HTML), then
|
|
|
462
474
|
conventions below describe the reference writer rather than constraining the format —
|
|
463
475
|
readers address entries by name (§7.1):
|
|
464
476
|
|
|
465
|
-
- Entries for resources fetched from a URL carry that URL in their *comment* field
|
|
466
|
-
|
|
467
|
-
|
|
477
|
+
- Entries for resources fetched from a URL carry that URL in their *comment* field,
|
|
478
|
+
and so does each `index.html`, whose comment is the URL of the page or frame it
|
|
479
|
+
holds; a resource that came from a `data:` URL carries the literal marker `data:`
|
|
480
|
+
instead, and `manifest.json` and `page.pdf` have no comment. Comments are omitted entirely
|
|
468
481
|
from a password-protected archive, because the central directory is not encrypted
|
|
469
482
|
(§5.6).
|
|
470
483
|
- Entries whose content is already compressed (images, fonts, media, PDF) are STOREd
|
|
@@ -491,7 +504,8 @@ both. **Raw** — the EOCD declares a zero-length comment and the trailing bytes
|
|
|
491
504
|
simply outside the archive — is the default, because tools print a declared archive
|
|
492
505
|
comment on ordinary operations (§8.1), and in universal mode that comment is the
|
|
493
506
|
whole base64 recovery payload. **Declared** — the EOCD's comment length covers every
|
|
494
|
-
byte after the record
|
|
507
|
+
byte after the record, `declareAppendedData` in the reference writer — is the later
|
|
508
|
+
addition, and it is the only form some
|
|
495
509
|
readers accept at all: `java.util.zip`, and therefore Android and most JVM tooling,
|
|
496
510
|
rejects an archive with undeclared trailing bytes outright (§8.1). A writer SHOULD
|
|
497
511
|
offer both and default to raw.
|
|
@@ -534,7 +548,7 @@ of it would break the face. Without the HTML face the document needs no wrapper,
|
|
|
534
548
|
where it sits depends on the PNG face: alone with the ZIP face it simply starts the
|
|
535
549
|
file, at offset 0, exercising only the trailing-data tolerance; with the PNG face it is
|
|
536
550
|
the data of a `tEXt "PDF"` chunk placed right after `IHDR` (§3.1), which puts `%PDF-`
|
|
537
|
-
at
|
|
551
|
+
at offset 45 exactly, inside the window but not at its start.
|
|
538
552
|
|
|
539
553
|
### 4.4 The PNG decoder
|
|
540
554
|
|
|
@@ -588,7 +602,8 @@ It works in three steps:
|
|
|
588
602
|
document whose data starts with those characters. An element bearing the identifier
|
|
589
603
|
wins over a comment when both resolve; a reader that finds an id-bearing element
|
|
590
604
|
which is not one of §5.1's wrapper rungs SHOULD fall back to the comment, since the
|
|
591
|
-
`id` is then something else in the page
|
|
605
|
+
`id` is then something else in the page; the reference extractor does not, and
|
|
606
|
+
takes whatever element bears the identifier. The two placements (§3.1) need no
|
|
592
607
|
telling apart, and neither the region's position in the tree nor its depth carries
|
|
593
608
|
meaning — a document that moved the node before extraction resolves the same way,
|
|
594
609
|
which matters because the reference extractor relocates `meta` and `style` elements
|
|
@@ -627,7 +642,8 @@ It works in three steps:
|
|
|
627
642
|
256 values map to themselves — but the shortcut "any code point ≤ 255 is that
|
|
628
643
|
byte" is **not** a valid substitute: under other qualifying charsets (§2.1) code
|
|
629
644
|
points below 256 can belong to a different byte, 75 of them under `macintosh`, and
|
|
630
|
-
the shortcut would silently corrupt the region. The
|
|
645
|
+
the shortcut would silently corrupt the region. The reference extractor takes it,
|
|
646
|
+
and is correct only because it supports windows-1252 alone. The two things parsing destroyed
|
|
631
647
|
are restored from the payload: each parsed newline consumes the next 2-bit code to
|
|
632
648
|
reproduce the original byte sequence.
|
|
633
649
|
|
|
@@ -747,8 +763,8 @@ sits, in ordinary element text, and in attribute values alike — so the writer
|
|
|
747
763
|
every character outside printable ASCII, along with `&`, `<`, `>` and `"`, as a
|
|
748
764
|
numeric reference. Those bytes are therefore pure ASCII and the text survives the
|
|
749
765
|
single-byte declaration intact: a page titled 日本語 shows as 日本語 in the browser tab
|
|
750
|
-
and to any conforming parser. The reference writer passes the title
|
|
751
|
-
|
|
766
|
+
and to any conforming parser. The reference writer passes the title, the canonical
|
|
767
|
+
link's `href` and the viewport value — attribute values, which is what the `"` is for —
|
|
752
768
|
through one shared escaper. Writers that emit such text raw MUST NOT do so in
|
|
753
769
|
universal mode, where the same bytes decode as mojibake.
|
|
754
770
|
|
|
@@ -763,8 +779,11 @@ only one with no escape available at all. Comment data does not resolve characte
|
|
|
763
779
|
references either, and unlike script data it has no second language of its own to
|
|
764
780
|
escape in: `é` written in a comment stays `é` in every reader, so the escaper
|
|
765
781
|
does not restore the character, it replaces one unreadable form with another. The
|
|
766
|
-
reference writer therefore leaves the comment alone and serializes it as
|
|
767
|
-
the rest of the prologue, deliberately.
|
|
782
|
+
reference writer therefore leaves the comment's characters alone and serializes it as
|
|
783
|
+
UTF-8 with the rest of the prologue, deliberately. What it does rewrite is the
|
|
784
|
+
comment's own terminators: a space goes before the `>` of `-->` and `--!>`, before a
|
|
785
|
+
leading `>` or `->`, and after a trailing `<!-`, so the comment cannot close itself
|
|
786
|
+
(§5.1). Its audience is whoever opens the raw file in an
|
|
768
787
|
editor or runs a text tool over it, and those decode the bytes as UTF-8 whatever the
|
|
769
788
|
declaration says; only a browser's raw view, which honors the declared charset, shows
|
|
770
789
|
the text as mojibake in universal mode. The bytes are harmless to extraction, since
|
|
@@ -867,7 +886,9 @@ with (§4.5): an element rung takes it as an `id` attribute — `<script type=sf
|
|
|
867
886
|
id=sfz-data>`, `<noframes id=sfz-data>` — and the comment rung as the first characters of
|
|
868
887
|
its data, `<!--sfz-data`. The wrappers hiding the PDF and PNG faces MUST NOT carry it:
|
|
869
888
|
those payloads are found by byte structure, and a second node bearing the identifier
|
|
870
|
-
would shadow the archive.
|
|
889
|
+
would shadow the archive. For the same reason no comment ahead of the wrapper may
|
|
890
|
+
begin with those characters, the implementation-defined comment of §3.1 included,
|
|
891
|
+
since the lookup takes the first that does.
|
|
871
892
|
|
|
872
893
|
The reference writer walks the ladder from the top and takes the first rung the payload
|
|
873
894
|
does not defeat. The test it applies is the format's, and is the same for every payload;
|
|
@@ -956,6 +977,19 @@ writer hides and in every variant that hides one. Every rejection restarts the b
|
|
|
956
977
|
(§6): the wrapper choice changes the bytes preceding the archive, so the archive must
|
|
957
978
|
be rewritten at its new position.
|
|
958
979
|
|
|
980
|
+
Two fields are patched after that check. The EOCD comment-length field sits at the
|
|
981
|
+
end of the ZIP region and is patched under the declared form (§6.1, step 11); the
|
|
982
|
+
writer tests the bytes around it again with the final value in place and keeps the raw
|
|
983
|
+
form when that value would complete a pattern, since the raw form is always valid. The
|
|
984
|
+
`tEXt "ZIP"` length field sits inside the pixel-data wrapper, with the fixed `tEXt`
|
|
985
|
+
type and `ZIP` keyword after it, and is written last (step 12). The header is tested
|
|
986
|
+
with the rest of the payload, the length as zeros, which cannot join a pattern; the
|
|
987
|
+
real length is big-endian, so a pattern byte in it would have to be the most
|
|
988
|
+
significant byte of the chunk's size, and the smallest byte any pattern contains, `-`
|
|
989
|
+
at 0x2D, puts that size at 0x2D000000 bytes, about 755 MB. The writer refuses to
|
|
990
|
+
build a self-extracting PNG variant whose chunk reaches that size rather than
|
|
991
|
+
re-check the field.
|
|
992
|
+
|
|
959
993
|
### 5.2 The appended-data budget
|
|
960
994
|
|
|
961
995
|
Everything the writer emits after the EOCD record MUST fit in 65535 bytes — the
|
|
@@ -992,8 +1026,9 @@ ahead of the archive would shift every offset the ZIP writer has already committ
|
|
|
992
1026
|
the reservation is padded with spaces and the real payload is written into it once
|
|
993
1027
|
its final size is known (§6). Relocation is final for the build, and a relocated
|
|
994
1028
|
archive carries no appended run at all: the writer emits neither the wrapper's
|
|
995
|
-
terminator nor the end tags, so
|
|
996
|
-
|
|
1029
|
+
terminator nor the end tags, so outside the PNG face, whose tail still follows (§5.1),
|
|
1030
|
+
the file ends at the EOCD record like a plain ZIP file and the readers that reject
|
|
1031
|
+
trailing bytes open it (§8.1). The parser closes the open
|
|
997
1032
|
comment or element at end of file, and `</body></html>` are implied, so the page
|
|
998
1033
|
renders the same.
|
|
999
1034
|
|
|
@@ -1226,17 +1261,23 @@ buys. The cost is not evenly spread:
|
|
|
1226
1261
|
| Plus the PDF or PNG face | The header window of §4.3 or the chunk patching of §5.3, plus a second wrapper choice for the embedded payload |
|
|
1227
1262
|
|
|
1228
1263
|
The second row already needs a rebuild when the rung changes; the third adds the rest of
|
|
1229
|
-
the retry loops
|
|
1230
|
-
|
|
1264
|
+
the retry loops and the first value computed only once the archive is final, the
|
|
1265
|
+
recovery payload; the fourth adds the second such value, the PNG chunk length and CRC
|
|
1266
|
+
(§5.4). A writer that only wants durable saved
|
|
1231
1267
|
pages can stop at the first row; the files it produces are accepted by every reader in
|
|
1232
1268
|
§8.1.
|
|
1233
1269
|
|
|
1234
1270
|
### 6.1 Build order
|
|
1235
1271
|
|
|
1236
1272
|
1. **PNG head.** With the PNG face, copy the signature and `IHDR` from the source
|
|
1237
|
-
image unchanged.
|
|
1238
|
-
`tEXt "
|
|
1239
|
-
|
|
1273
|
+
image unchanged. With the HTML face, choose the wrapper for the pixel-data payload
|
|
1274
|
+
(§5.1) and emit the `tEXt "PNG"` chunk: its 12 header bytes, the head of the
|
|
1275
|
+
prologue through `<body hidden>` as built in steps 2 and 3, the wrapper start tag,
|
|
1276
|
+
and the chunk CRC. Without the HTML face but with the PDF face, emit the
|
|
1277
|
+
`tEXt "PDF"` chunk holding the PDF document here instead, so its header falls
|
|
1278
|
+
inside the PDF scan window (§4.3). Then copy every source chunk between `IHDR` and
|
|
1279
|
+
`IEND`, write the `tEXt "ZIP"` chunk header with a zero length that step 12 patches,
|
|
1280
|
+
and, with the HTML face, the pixel-data wrapper end tag.
|
|
1240
1281
|
2. **HTML prologue.** With the HTML face, emit the doctype (omitted under the PNG
|
|
1241
1282
|
face, which owns the start of the file), the root element start tag, the
|
|
1242
1283
|
`<meta charset>` required by §2.1, any comment the implementation adds — after the
|
|
@@ -1266,12 +1307,13 @@ pages can stop at the first row; the files it produces are accepted by every rea
|
|
|
1266
1307
|
section requires nothing about the doctype there. A writer MAY drop it, as the
|
|
1267
1308
|
reference writer does, or keep it — but a kept one is content like any other, and
|
|
1268
1309
|
both windows are measured from the start of the *file*, which under this face begins
|
|
1269
|
-
45 bytes before the HTML does: the signature, `IHDR`, and the chunk length, type
|
|
1270
|
-
keyword. That is 45 bytes less room than the arithmetic above
|
|
1310
|
+
45 bytes before the HTML does: the signature, `IHDR`, and the chunk length, type,
|
|
1311
|
+
keyword and NUL separator. That is 45 bytes less room than the arithmetic above
|
|
1312
|
+
suggests.
|
|
1271
1313
|
|
|
1272
1314
|
Then the head elements (the
|
|
1273
1315
|
`<title>` and the canonical link among them), the CSS and `<body hidden>`,
|
|
1274
|
-
the wait and error messages, the optional
|
|
1316
|
+
the wait and error messages, the optional text body, and the
|
|
1275
1317
|
bootstrap script. With a password, five of those are left out: the comment, the
|
|
1276
1318
|
title, the canonical link, the text body and the entry comments of step 6 (§5.6).
|
|
1277
1319
|
With the PNG face the head of this region,
|
|
@@ -1289,10 +1331,14 @@ pages can stop at the first row; the files it produces are accepted by every rea
|
|
|
1289
1331
|
or lower (§4.3). Only what a parser needs first precedes it — the doctype, the
|
|
1290
1332
|
root element and the charset declaration — and everything
|
|
1291
1333
|
else in the head (title, link and meta elements, the stylesheet, `<body hidden>`,
|
|
1292
|
-
the messages, the optional
|
|
1334
|
+
the messages, the optional text body) follows it. Emit the
|
|
1293
1335
|
wrapper start tag chosen for the PDF payload (§5.1), the hand-built `page.pdf`
|
|
1294
1336
|
local file header, the PDF document, the wrapper end tag, and record the local
|
|
1295
|
-
header's absolute position; then resume the prologue.
|
|
1337
|
+
header's absolute position; then resume the prologue. The reference writer's
|
|
1338
|
+
header declares version 2.0, the language encoding flag alone, method STORE, the
|
|
1339
|
+
build's modification date in DOS form, the precomputed CRC-32, the document's
|
|
1340
|
+
length as both sizes, and no extra field; its central record adds a Unix
|
|
1341
|
+
"made by" version and external attributes of a regular file, mode 0644.
|
|
1296
1342
|
|
|
1297
1343
|
The window is reachable but not structurally guaranteed, and it is the one place
|
|
1298
1344
|
where the format depends on the writer rather than on its own layout. The
|
|
@@ -1328,7 +1374,9 @@ pages can stop at the first row; the files it produces are accepted by every rea
|
|
|
1328
1374
|
4. **Reserved extra-data.** In universal mode, when a previous pass determined that
|
|
1329
1375
|
the payload must be relocated (§5.2), emit an empty `<sfz-extra-data>` element
|
|
1330
1376
|
followed by enough spaces to fill the reservation. The padding sits **outside** the
|
|
1331
|
-
element, so the element's text stays exactly the payload.
|
|
1377
|
+
element, so the element's text stays exactly the payload. With
|
|
1378
|
+
`preventAppendedData` set from the start there is still no reservation on the first
|
|
1379
|
+
pass: that pass measures the payload, and the second reserves (§6.2).
|
|
1332
1380
|
5. **Wrapper start tag** for the ZIP region, carrying the identifier (§5.1).
|
|
1333
1381
|
6. **The archive.** Create the ZIP writer, telling it the number of bytes already
|
|
1334
1382
|
written so that its offsets are absolute (§5.3). Add `index.html` first, then
|
|
@@ -1342,12 +1390,13 @@ pages can stop at the first row; the files it produces are accepted by every rea
|
|
|
1342
1390
|
8. **Close and patch.** Close the archive, then correct the end of central directory
|
|
1343
1391
|
record for the injected record: entry counts, directory size, and the zip64
|
|
1344
1392
|
record and locator when present (§5.7).
|
|
1345
|
-
9. **
|
|
1346
|
-
self-extracting file needs it — read back the ZIP region
|
|
1347
|
-
current wrapper (§5.1); on a collision, restart (§6.2).
|
|
1348
|
-
only, compute the region's CRC-32 and its newline codes,
|
|
1349
|
-
and decide its placement against the budget
|
|
1350
|
-
|
|
1393
|
+
9. **Wrapper check, then the universal payload.** With the HTML face — not only in
|
|
1394
|
+
universal mode, since any self-extracting file needs it — read back the ZIP region
|
|
1395
|
+
and check it against the current wrapper (§5.1); on a collision, restart (§6.2).
|
|
1396
|
+
Then, in universal mode only, compute the region's CRC-32 and its newline codes,
|
|
1397
|
+
build and compress the payload, and decide its placement against the budget
|
|
1398
|
+
(§5.2), restarting when an appended payload turns out not to fit. Relocation is
|
|
1399
|
+
never undone (§6.2).
|
|
1351
1400
|
10. **Appended run.** Unless appended data is prevented or the payload is relocated
|
|
1352
1401
|
(§5.2), emit the wrapper end tag,
|
|
1353
1402
|
the extra-data element when it is appended, and `</body></html>` — the end tags
|
|
@@ -1355,7 +1404,8 @@ pages can stop at the first row; the files it produces are accepted by every rea
|
|
|
1355
1404
|
11. **Fill the reservation.** In the relocated placement, write the payload into the
|
|
1356
1405
|
space reserved in step 4; if it no longer fits, restart (§6.2). Under
|
|
1357
1406
|
`declareAppendedData` (§4.2), the EOCD's comment-length field is patched here too,
|
|
1358
|
-
the appended run's length now being final
|
|
1407
|
+
the appended run's length now being final, unless the value would complete a
|
|
1408
|
+
pattern of the current wrapper, in which case the raw form stays (§5.1).
|
|
1359
1409
|
12. **PNG tail.** With the PNG face, patch the `tEXt "ZIP"` chunk's length field, now
|
|
1360
1410
|
that the total size is known, compute that chunk's CRC over everything from its
|
|
1361
1411
|
type to the last byte written, and append the CRC and the `IEND` chunk.
|
|
@@ -1407,9 +1457,10 @@ deterministic: the same page produces the same bytes, retries included. `manifes
|
|
|
1407
1457
|
records when the archive was made (§7.1), so two builds of one page at two moments
|
|
1408
1458
|
differ in that entry and in the entry sizes around it. A writer that retries MUST pin
|
|
1409
1459
|
the archive time across the passes of one build rather than read the clock again on
|
|
1410
|
-
each
|
|
1411
|
-
entries, which runs once per
|
|
1412
|
-
|
|
1460
|
+
each. The reference writer reads the clock once, inside the callback that emits the
|
|
1461
|
+
entries, which runs once per build; every retry reuses the entries that callback
|
|
1462
|
+
produced. Two builds of one page still read the clock twice, so its own determinism
|
|
1463
|
+
test freezes it. A consumer MUST NOT
|
|
1413
1464
|
treat the byte identity of two archives of the same page as meaningful.
|
|
1414
1465
|
|
|
1415
1466
|
## 7. Consuming SingleFile archives safely
|
|
@@ -1443,15 +1494,18 @@ handles every variant of §2 without knowing which one it has.
|
|
|
1443
1494
|
- **Resolve the page entry in this order.** The archive's internal layout is
|
|
1444
1495
|
implementation-defined, and two properties of the reference layout matter to a
|
|
1445
1496
|
reader. Every entry MAY sit under a single root directory, which the reference
|
|
1446
|
-
writer names
|
|
1447
|
-
`<root>/index.html`. And a page's nested
|
|
1448
|
-
their own under `frames/<n>/`, recursively,
|
|
1449
|
-
`index.html`
|
|
1497
|
+
writer names `<milliseconds since the epoch>_<tab id>/` when asked to create one
|
|
1498
|
+
(`createRootDirectory`); the page is then `<root>/index.html`. And a page's nested
|
|
1499
|
+
frames are stored as complete pages of their own under `frames/<n>/`, recursively,
|
|
1500
|
+
each with its own `index.html` and `manifest.json`, so an archive normally holds
|
|
1501
|
+
several of both and only the outermost pair is the page. The recognition test above
|
|
1502
|
+
therefore matches every frame directory too. Since neither property
|
|
1450
1503
|
is guaranteed, a reader resolves the entry point in three steps, stopping at the
|
|
1451
1504
|
first that succeeds:
|
|
1452
1505
|
|
|
1453
|
-
1. `manifest.json`
|
|
1454
|
-
|
|
1506
|
+
1. The `indexFilename` of the `manifest.json` at the smallest directory depth,
|
|
1507
|
+
resolved against that manifest's directory, when the entry exists. This is the
|
|
1508
|
+
only authoritative answer, so a writer that departs from
|
|
1455
1509
|
the reference layout SHOULD emit the manifest even though a reader MUST NOT
|
|
1456
1510
|
require it.
|
|
1457
1511
|
2. Otherwise the `index.html` entry at the smallest directory depth.
|
|
@@ -1466,14 +1520,18 @@ handles every variant of §2 without knowing which one it has.
|
|
|
1466
1520
|
URL as `originalUrl`, the title as `title`, the save time as `archiveTime` (an ISO
|
|
1467
1521
|
8601 string), the entry name of the page as `indexFilename` and the resource-to-URL
|
|
1468
1522
|
map as `resources`. The page displays without any of it, and a reader MUST NOT require
|
|
1469
|
-
the entry or any field of it. `indexFilename` names the page relative to the
|
|
1470
|
-
directory, not as a full entry name.
|
|
1523
|
+
the entry or any field of it. `indexFilename` names the page relative to the
|
|
1524
|
+
manifest's own directory, not as a full entry name. A frame's manifest carries the
|
|
1525
|
+
same `archiveTime` as the page's. The set of fields is not closed: a reader
|
|
1471
1526
|
MUST ignore what it does not recognize.
|
|
1472
1527
|
- **Expect a `page.pdf` entry whose data lies outside the archive proper** (§4.2). It
|
|
1473
1528
|
is an ordinary STORE entry at an ordinary offset, so nothing special is needed to
|
|
1474
1529
|
read it, but a reader that assumes the entries are contiguous will reject or
|
|
1475
1530
|
mislocate it: `page.pdf`'s local header is the first in the file, and the whole
|
|
1476
|
-
bootstrap lies between its data and the next one.
|
|
1531
|
+
bootstrap lies between its data and the next one. It is never placed under the root
|
|
1532
|
+
directory: the PDF face is one document per file, so an archive holds at most one
|
|
1533
|
+
`page.pdf`, and it sits at the top level whatever `createRootDirectory` does to the
|
|
1534
|
+
other entries.
|
|
1477
1535
|
|
|
1478
1536
|
### 7.2 Modifying
|
|
1479
1537
|
|
|
@@ -1667,7 +1725,10 @@ results of §8.1 were measured on the 1.5.107 build of the same specimen set, wh
|
|
|
1667
1725
|
differs only inside the prologue and so falls in the same classes: those are grouped
|
|
1668
1726
|
by whether bytes precede the archive and follow the EOCD, which no prologue change
|
|
1669
1727
|
alters. The declared-form results are the exception, `declareAppendedData` being later
|
|
1670
|
-
than that build (§8.5); they were measured separately on a build that has it.
|
|
1728
|
+
than that build (§8.5); they were measured separately on a build that has it. The
|
|
1729
|
+
specimen names carry a `.sfz.html` suffix chosen for the harness; the conventions of
|
|
1730
|
+
§2.2 are what the clients produce, not what these files are called.
|
|
1731
|
+
`--compress-content` makes the output an archive; `extract-data-from-page`
|
|
1671
1732
|
defaults to true there, so the plain variant has to switch it off:
|
|
1672
1733
|
|
|
1673
1734
|
| Specimen | Command |
|
|
@@ -1685,7 +1746,7 @@ defaults to true there, so the plain variant has to switch it off:
|
|
|
1685
1746
|
These specimens are deliberately small, and a reader tested only against them is
|
|
1686
1747
|
undertested: they are all flat archives of two or three entries. None
|
|
1687
1748
|
exercises a root directory, `frames/<n>/` nesting, a second `index.html`, a `data:`-URL
|
|
1688
|
-
entry comment, the optional text body
|
|
1749
|
+
entry comment, the optional text body (§4.6), a UTF-8 BOM, zip64
|
|
1689
1750
|
(§5.7), a payload past the 64 KB budget, or a relocated reservation with padding left
|
|
1690
1751
|
in it. Two omissions matter more than the rest, because they are the parts of §5.1 a
|
|
1691
1752
|
writer is most likely to get wrong: no specimen defeats a rung by its **start**
|
|
@@ -1748,7 +1809,7 @@ predicts.
|
|
|
1748
1809
|
| August 2026 | Core 1.5.108: the ZIP region carries the identifier `sfz-data` and the extractor addresses it with that instead of deducing it from its position beside `<sfz-extra-data>` (§4.5). This fixes universal extraction on the `<style type=sfz-data>` rung, where the reference extractor's own relocation of `style` elements into the head moved the region out from under the positional rule |
|
|
1749
1810
|
| August 2026 | Core 1.5.108: the recovery payload stops two bytes short of the End Of Central Directory record, excluding its comment-length field (§1.3), which lets universal-mode archives declare their appended data as the archive comment — a writer option, for `java.util.zip` and the readers that reject undeclared trailing bytes (§4.2) |
|
|
1750
1811
|
| August 2026 | Core 1.5.108: the PDF and PNG faces test a wrapper rung's start pattern as well as its end pattern, closing the same script-data escape hole the ZIP region was already guarded against — a face payload holding `<!--` and then `<script` took the `<script type=sfz-data>` rung and swallowed the rest of the document (§5.1) |
|
|
1751
|
-
| August 2026 | Core 1.5.108: the retry loop discards a relocation reservation
|
|
1812
|
+
| August 2026 | Core 1.5.108: the retry loop never discards a relocation reservation, so a payload sitting on the appended-data boundary cannot oscillate between the two placements forever (§6.2) |
|
|
1752
1813
|
| August 2026 | Core 1.5.110: a PDF or PNG face whose payload names every rung is dropped instead of written bare (§5.1). Found by nesting an archive inside itself as both faces: the fifth level exhausts the ladder, and readers then extracted the fourth level's archive — checksums intact, no way to tell (§7.4) |
|
|
1753
1814
|
| August 2026 | Core 1.5.110: a PNG face leaving the comment rung on its checksum resumes the rung search instead of taking the next rung untested (§5.1). Taking it put a payload holding `</script>` on the script rung, where its own bytes closed the wrapper 93 bytes in and left the image data, the chunk framing and the whole ZIP region to the parser |
|
|
1754
1815
|
| August 2026 | Core 1.5.110: `<svg><![CDATA[` joins the ladder above `<plaintext>` (§5.1) — the one rung whose terminator, `]]>`, real payloads rarely carry. It gives a payload naming every element rung somewhere to go that does not cost the appended-data placement, and moves the self-nesting limit from the fifth level to the sixth |
|
|
@@ -1791,3 +1852,13 @@ BOM, a user override and a transport-layer charset all outranking it — narrow
|
|
|
1791
1852
|
practice, since the raw read comes first and no encoding applies to it (§2.1) — and that
|
|
1792
1853
|
the recovery payload's 32-bit length field caps the region below 2^32 bytes, with
|
|
1793
1854
|
engine string limits binding well before that (§5.5).
|
|
1855
|
+
|
|
1856
|
+
A pass in September 2026, against core 1.5.120, read the text alone first and then
|
|
1857
|
+
checked each open question against the writer. It corrected two statements about the
|
|
1858
|
+
reference writer that the code contradicted: the retry loop never discards a
|
|
1859
|
+
reservation, and the archive time is read once per build, not once per pass (§6.2).
|
|
1860
|
+
It added what only the code could say: the PNG build steps that §6.1 had skipped, the
|
|
1861
|
+
fields patched after the wrapper check and the size below which they are harmless
|
|
1862
|
+
(§5.1), the chunks the PNG face copies (§3.1), the per-frame manifests and the root
|
|
1863
|
+
directory's name (§7.1), the `page.pdf` header fields (§6.1), and the range-reading
|
|
1864
|
+
failure path (§4.1).
|
package/package.json
CHANGED
|
@@ -93,6 +93,9 @@ const MAX_APPENDED_DATA_LENGTH = 65535;
|
|
|
93
93
|
const PDF_ENTRY_FILENAME = "page.pdf";
|
|
94
94
|
const PRESCAN_WINDOW_LENGTH = 1024;
|
|
95
95
|
const PNG_TEXT_CHUNK_HEADER_LENGTH = 12;
|
|
96
|
+
const PNG_ZIP_CHUNK_TYPE_KEYWORD = new Uint8Array([0x74, 0x45, 0x58, 0x74, 0x5a, 0x49, 0x50, 0]);
|
|
97
|
+
const MAX_HIDDEN_PNG_CHUNK_LENGTH = 0x2D000000;
|
|
98
|
+
const WRAPPER_PATTERN_WINDOW_LENGTH = 12;
|
|
96
99
|
const MINIMAL_DOCTYPE = "<!DOCTYPE html>";
|
|
97
100
|
const UNHIDDEN_FACE_WARNING_MESSAGE = "SingleFile: the page data contains every HTML tag that could hide an embedded file, the archive was written without its";
|
|
98
101
|
const EMBEDDED_IMAGE_LABEL = "PNG image";
|
|
@@ -194,7 +197,7 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
|
|
|
194
197
|
await writeData(zipDataWriter.writable, embeddedImageData);
|
|
195
198
|
await writeData(zipDataWriter.writable, new Uint8Array(4));
|
|
196
199
|
embeddedImageDataOffset = zipDataWriter.offset;
|
|
197
|
-
await writeData(zipDataWriter.writable,
|
|
200
|
+
await writeData(zipDataWriter.writable, PNG_ZIP_CHUNK_TYPE_KEYWORD);
|
|
198
201
|
if (options.selfExtractingArchive) {
|
|
199
202
|
await writeData(zipDataWriter.writable, new TextEncoder().encode(endTag));
|
|
200
203
|
}
|
|
@@ -301,12 +304,16 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
|
|
|
301
304
|
if (options.declareAppendedData) {
|
|
302
305
|
const appendedDataLength = pageContent.length - data.length +
|
|
303
306
|
(options.embeddedImage ? PNG_CHUNK_CRC_LENGTH + PNG_IEND_LENGTH : 0);
|
|
304
|
-
if (appendedDataLength && appendedDataLength <= MAX_APPENDED_DATA_LENGTH) {
|
|
307
|
+
if (appendedDataLength && appendedDataLength <= MAX_APPENDED_DATA_LENGTH && isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options)) {
|
|
305
308
|
new DataView(pageContent.buffer, pageContent.byteOffset).setUint16(zipDataEnd, appendedDataLength, true);
|
|
306
309
|
}
|
|
307
310
|
}
|
|
308
311
|
if (options.embeddedImage) {
|
|
309
|
-
|
|
312
|
+
const chunkLength = zipDataWriter.offset - embeddedImageDataOffset - 4;
|
|
313
|
+
if (options.selfExtractingArchive && chunkLength >= MAX_HIDDEN_PNG_CHUNK_LENGTH) {
|
|
314
|
+
throw new Error("SingleFile: the embedded PNG chunk is too large to be hidden from the HTML parser");
|
|
315
|
+
}
|
|
316
|
+
pageContent.set(getLength(chunkLength), embeddedImageDataOffset - 4);
|
|
310
317
|
return new Blob([
|
|
311
318
|
pageContent,
|
|
312
319
|
getCRC32(pageContent, embeddedImageDataOffset),
|
|
@@ -317,6 +324,18 @@ async function buildArchive(pageData, options, script, entriesData, zipWriterOpt
|
|
|
317
324
|
}
|
|
318
325
|
}
|
|
319
326
|
|
|
327
|
+
function isDeclaredLengthHidden(pageContent, zipDataEnd, appendedDataLength, options) {
|
|
328
|
+
if (options.extractDataFromPageTags && options.extractDataFromPageTags[0] == "<plaintext>") {
|
|
329
|
+
return true;
|
|
330
|
+
}
|
|
331
|
+
const tail = pageContent.slice(zipDataEnd - WRAPPER_PATTERN_WINDOW_LENGTH, zipDataEnd + COMMENT_LENGTH_FIELD_LENGTH);
|
|
332
|
+
new DataView(tail.buffer).setUint16(WRAPPER_PATTERN_WINDOW_LENGTH, appendedDataLength, true);
|
|
333
|
+
const tagIndex = options.extractDataFromPageTags ? getExtraDataTagIndex(options.extractDataFromPageTags) + 1 : 0;
|
|
334
|
+
const [startRegExp, endRegExp] = EMBEDDED_DATA_REGEXPS[tagIndex];
|
|
335
|
+
const tailText = TEXT_DECODER.decode(tail);
|
|
336
|
+
return !tailText.match(startRegExp) && !tailText.match(endRegExp);
|
|
337
|
+
}
|
|
338
|
+
|
|
320
339
|
function getCRC32(data, indexData = 0) {
|
|
321
340
|
const crcArray = new Uint8Array(4);
|
|
322
341
|
setUint32(crcArray, getCRC32Value(data, indexData));
|
|
@@ -625,7 +644,8 @@ function findEmbeddedDataTagIndex(text, fromIndex = 0) {
|
|
|
625
644
|
}
|
|
626
645
|
|
|
627
646
|
function getImageHTMLChunk(pageData, options, lastModDate) {
|
|
628
|
-
const embeddedImageText = TEXT_DECODER.decode(getEmbeddedImageData(options.embeddedImage))
|
|
647
|
+
const embeddedImageText = TEXT_DECODER.decode(getEmbeddedImageData(options.embeddedImage)) +
|
|
648
|
+
TEXT_DECODER.decode(new Uint8Array(4)) + TEXT_DECODER.decode(PNG_ZIP_CHUNK_TYPE_KEYWORD);
|
|
629
649
|
let tagIndex = findEmbeddedDataTagIndex(embeddedImageText);
|
|
630
650
|
while (tagIndex != -1) {
|
|
631
651
|
const [startTag, endTag] = EMBEDDED_DATA_TAGS[tagIndex];
|
|
@@ -683,6 +703,7 @@ async function addPageResources(zipWriter, pageData, options, prefixName, url) {
|
|
|
683
703
|
Promise.all(Object.keys(pageData.resources).map(async resourceType =>
|
|
684
704
|
Promise.all(pageData.resources[resourceType].map(data => {
|
|
685
705
|
if (resourceType == "frames") {
|
|
706
|
+
data.archiveTime = pageData.archiveTime;
|
|
686
707
|
return addPageResources(zipWriter, data, options, prefixName + data.name, data.url);
|
|
687
708
|
} else {
|
|
688
709
|
return addFile(zipWriter, prefixName, data, options.disableCompression);
|
|
@@ -775,7 +796,13 @@ async function getContent() {
|
|
|
775
796
|
}
|
|
776
797
|
};
|
|
777
798
|
if (aborted) {
|
|
778
|
-
xhr.onload = () =>
|
|
799
|
+
xhr.onload = () => {
|
|
800
|
+
if (xhr.status === 200) {
|
|
801
|
+
resolve(xhr.response);
|
|
802
|
+
} else {
|
|
803
|
+
extractDataFromDocument();
|
|
804
|
+
}
|
|
805
|
+
};
|
|
779
806
|
}
|
|
780
807
|
}
|
|
781
808
|
});
|