@awacloud/pdf 0.0.0-stage → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +609 -0
- package/LICENSE +661 -0
- package/NOTICE +77 -0
- package/README.md +363 -2
- package/dist/build/index.js +21 -0
- package/dist/build/pdf-full-rw.js +10972 -0
- package/dist/build/pdf-full-rw.meta.json +105 -0
- package/dist/build/pdf-full-rw.min.js +53 -0
- package/dist/build/pdf-full.js +6078 -0
- package/dist/build/pdf-full.meta.json +90 -0
- package/dist/build/pdf-full.min.js +32 -0
- package/dist/build/pdf-large-rw.js +10367 -0
- package/dist/build/pdf-large-rw.meta.json +99 -0
- package/dist/build/pdf-large-rw.min.js +53 -0
- package/dist/build/pdf-large.js +5473 -0
- package/dist/build/pdf-large.meta.json +84 -0
- package/dist/build/pdf-large.min.js +32 -0
- package/dist/build/pdf-legacy-rw.js +12402 -0
- package/dist/build/pdf-legacy-rw.meta.json +110 -0
- package/dist/build/pdf-legacy-rw.min.js +53 -0
- package/dist/build/pdf-legacy.js +7508 -0
- package/dist/build/pdf-legacy.meta.json +95 -0
- package/dist/build/pdf-legacy.min.js +32 -0
- package/dist/build/pdf-rw.js +7578 -0
- package/dist/build/pdf-rw.meta.json +77 -0
- package/dist/build/pdf-rw.min.js +53 -0
- package/dist/build/pdf.js +2684 -0
- package/dist/build/pdf.meta.json +62 -0
- package/dist/build/pdf.min.js +32 -0
- package/dist/standalone/pdf-full-rw.js +16798 -0
- package/dist/standalone/pdf-full-rw.meta.json +78 -0
- package/dist/standalone/pdf-full-rw.min.js +56 -0
- package/dist/standalone/pdf-full.js +11904 -0
- package/dist/standalone/pdf-full.meta.json +63 -0
- package/dist/standalone/pdf-full.min.js +35 -0
- package/dist/standalone/pdf-large-rw.js +16193 -0
- package/dist/standalone/pdf-large-rw.meta.json +72 -0
- package/dist/standalone/pdf-large-rw.min.js +56 -0
- package/dist/standalone/pdf-large.js +11299 -0
- package/dist/standalone/pdf-large.meta.json +57 -0
- package/dist/standalone/pdf-large.min.js +35 -0
- package/dist/standalone/pdf-legacy-rw.js +18228 -0
- package/dist/standalone/pdf-legacy-rw.meta.json +83 -0
- package/dist/standalone/pdf-legacy-rw.min.js +56 -0
- package/dist/standalone/pdf-legacy.js +13334 -0
- package/dist/standalone/pdf-legacy.meta.json +68 -0
- package/dist/standalone/pdf-legacy.min.js +35 -0
- package/dist/standalone/pdf-rw.js +13404 -0
- package/dist/standalone/pdf-rw.meta.json +50 -0
- package/dist/standalone/pdf-rw.min.js +56 -0
- package/dist/standalone/pdf.js +8510 -0
- package/dist/standalone/pdf.meta.json +35 -0
- package/dist/standalone/pdf.min.js +35 -0
- package/docs/README.md +53 -0
- package/docs/api/README.md +38 -0
- package/docs/api/_shared/README.md +91 -0
- package/docs/api/action/README.md +29 -0
- package/docs/api/action/action.md +81 -0
- package/docs/api/action/goTo.md +66 -0
- package/docs/api/action/launch.md +58 -0
- package/docs/api/action/named.md +55 -0
- package/docs/api/action/uri.md +54 -0
- package/docs/api/annot/README.md +53 -0
- package/docs/api/annot/annot.md +114 -0
- package/docs/api/annot/fileAttach.md +53 -0
- package/docs/api/annot/freeText.md +68 -0
- package/docs/api/annot/ink.md +69 -0
- package/docs/api/annot/link.md +74 -0
- package/docs/api/annot/markup.md +83 -0
- package/docs/api/annot/popup.md +52 -0
- package/docs/api/annot/projection.md +56 -0
- package/docs/api/annot/redact.md +67 -0
- package/docs/api/annot/square.md +87 -0
- package/docs/api/annot/stamp.md +54 -0
- package/docs/api/annot/text.md +69 -0
- package/docs/api/annot/widget.md +69 -0
- package/docs/api/associatedFiles/README.md +9 -0
- package/docs/api/associatedFiles/associatedFiles.md +78 -0
- package/docs/api/bundles/README.md +68 -0
- package/docs/api/bundles/dist-matrix.md +165 -0
- package/docs/api/bundles/pdf-full.md +148 -0
- package/docs/api/bundles/pdf-large.md +144 -0
- package/docs/api/bundles/pdf-legacy.md +169 -0
- package/docs/api/content/README.md +29 -0
- package/docs/api/content/color.md +99 -0
- package/docs/api/content/graphics.md +114 -0
- package/docs/api/content/images.md +124 -0
- package/docs/api/content/ops.md +100 -0
- package/docs/api/content/stream.md +107 -0
- package/docs/api/content/text.md +98 -0
- package/docs/api/crypto/README.md +29 -0
- package/docs/api/crypto/aesGcm.md +72 -0
- package/docs/api/crypto/permissions.md +79 -0
- package/docs/api/crypto/security.md +98 -0
- package/docs/api/crypto/standardV4.md +104 -0
- package/docs/api/crypto/standardV5.md +84 -0
- package/docs/api/crypto/standardV6.md +93 -0
- package/docs/api/destination/README.md +9 -0
- package/docs/api/destination/destination.md +79 -0
- package/docs/api/document/README.md +29 -0
- package/docs/api/document/builder.md +281 -0
- package/docs/api/document/catalog.md +98 -0
- package/docs/api/document/document.md +187 -0
- package/docs/api/document/encryptedWriter.md +149 -0
- package/docs/api/document/incrementalWriter.md +148 -0
- package/docs/api/document/page.md +99 -0
- package/docs/api/document/pages.md +82 -0
- package/docs/api/document/resources.md +102 -0
- package/docs/api/document/writer.md +157 -0
- package/docs/api/document/xrefStreamWriter.md +122 -0
- package/docs/api/embedded/README.md +13 -0
- package/docs/api/embedded/collection.md +80 -0
- package/docs/api/embedded/embeddedFile.md +86 -0
- package/docs/api/embedded/fileSpec.md +87 -0
- package/docs/api/errors.md +110 -0
- package/docs/api/extra/3d-richmedia.md +76 -0
- package/docs/api/extra/README.md +99 -0
- package/docs/api/extra/annot-extended.md +71 -0
- package/docs/api/extra/associated-files.md +70 -0
- package/docs/api/extra/ccitt-fax-decoder.md +74 -0
- package/docs/api/extra/color-spaces-extended.md +72 -0
- package/docs/api/extra/content-ops-extended.md +82 -0
- package/docs/api/extra/document-parts.md +69 -0
- package/docs/api/extra/embedded-files-portfolio.md +87 -0
- package/docs/api/extra/font-cid-typed.md +77 -0
- package/docs/api/extra/font-color-tagging.md +76 -0
- package/docs/api/extra/form-actions-extended.md +75 -0
- package/docs/api/extra/info-dict-deprecated.md +72 -0
- package/docs/api/extra/jbig2-read.md +80 -0
- package/docs/api/extra/legacy-deprecated-annots.md +89 -0
- package/docs/api/extra/legacy-deprecated-filters.md +78 -0
- package/docs/api/extra/legacy-rc4-read.md +74 -0
- package/docs/api/extra/legacy-xfa-read.md +65 -0
- package/docs/api/extra/linearization-write.md +71 -0
- package/docs/api/extra/misc.md +93 -0
- package/docs/api/extra/optional-content-extended.md +83 -0
- package/docs/api/extra/pdf-a-output-intent.md +65 -0
- package/docs/api/extra/pdf-sandbox.md +76 -0
- package/docs/api/extra/pdf-ua-tagged.md +63 -0
- package/docs/api/extra/pdf-x-prepress.md +65 -0
- package/docs/api/extra/redaction-iso32005.md +65 -0
- package/docs/api/extra/shading-typed.md +73 -0
- package/docs/api/extra/sig-aes-gcm.md +69 -0
- package/docs/api/extra/sig-pades.md +103 -0
- package/docs/api/extra/tagged-pdf-typed.md +78 -0
- package/docs/api/extra/transparency-typed.md +74 -0
- package/docs/api/extra/well-tagged-pdf.md +61 -0
- package/docs/api/extra/xmp-extended.md +65 -0
- package/docs/api/font/README.md +25 -0
- package/docs/api/font/embed.md +157 -0
- package/docs/api/font/encoding.md +95 -0
- package/docs/api/font/font.md +97 -0
- package/docs/api/font/type3.md +89 -0
- package/docs/api/form/README.md +35 -0
- package/docs/api/form/acroform.md +88 -0
- package/docs/api/form/appearance.md +87 -0
- package/docs/api/form/button.md +97 -0
- package/docs/api/form/choice.md +96 -0
- package/docs/api/form/fieldTree.md +93 -0
- package/docs/api/form/signature.md +90 -0
- package/docs/api/form/text.md +88 -0
- package/docs/api/linearization/README.md +11 -0
- package/docs/api/linearization/linearization.md +81 -0
- package/docs/api/main.md +116 -0
- package/docs/api/metadata/README.md +10 -0
- package/docs/api/metadata/info.md +70 -0
- package/docs/api/metadata/xmp.md +62 -0
- package/docs/api/ocg/README.md +23 -0
- package/docs/api/ocg/config.md +95 -0
- package/docs/api/ocg/ocg.md +77 -0
- package/docs/api/outline/README.md +11 -0
- package/docs/api/outline/outline.md +107 -0
- package/docs/api/pdf.md +152 -0
- package/docs/api/prepress/README.md +10 -0
- package/docs/api/prepress/outputIntent.md +79 -0
- package/docs/api/prepress/pageBoundary.md +75 -0
- package/docs/api/sig/README.md +32 -0
- package/docs/api/sig/byteRange.md +120 -0
- package/docs/api/sig/certChain.md +84 -0
- package/docs/api/sig/dss.md +111 -0
- package/docs/api/sig/oids.md +76 -0
- package/docs/api/sig/sha1.md +72 -0
- package/docs/api/sig/sign.md +317 -0
- package/docs/api/sig/signature.md +178 -0
- package/docs/api/sig/timestamp.md +84 -0
- package/docs/api/syntax/README.md +29 -0
- package/docs/api/syntax/crossRefStream.md +115 -0
- package/docs/api/syntax/filters/README.md +50 -0
- package/docs/api/syntax/filters/ascii85.md +76 -0
- package/docs/api/syntax/filters/asciiHex.md +73 -0
- package/docs/api/syntax/filters/dispatch.md +125 -0
- package/docs/api/syntax/filters/flate.md +134 -0
- package/docs/api/syntax/filters/runLength.md +78 -0
- package/docs/api/syntax/objStream.md +88 -0
- package/docs/api/syntax/parser-obj.md +97 -0
- package/docs/api/syntax/parser.md +151 -0
- package/docs/api/syntax/serializer.md +109 -0
- package/docs/api/syntax/tokenizer.md +104 -0
- package/docs/api/syntax/trailer.md +85 -0
- package/docs/api/syntax/xref.md +139 -0
- package/docs/api/tagged/README.md +25 -0
- package/docs/api/tagged/classMap.md +67 -0
- package/docs/api/tagged/markedContent.md +62 -0
- package/docs/api/tagged/parentTree.md +67 -0
- package/docs/api/tagged/roleMap.md +67 -0
- package/docs/api/tagged/structElement.md +76 -0
- package/docs/api/tagged/structTree.md +75 -0
- package/docs/guide/coverage.md +113 -0
- package/docs/guide/crypto.md +121 -0
- package/docs/guide/extending.md +76 -0
- package/docs/guide/getting-started.md +75 -0
- package/docs/guide/legacy-1.7.md +42 -0
- package/docs/guide/pades-integration.md +579 -0
- package/docs/guide/read-pdf.md +89 -0
- package/package.json +97 -4
- package/src/_shared/index.js +179 -0
- package/src/action/action.js +119 -0
- package/src/action/goTo.js +89 -0
- package/src/action/launch.js +61 -0
- package/src/action/named.js +54 -0
- package/src/action/uri.js +51 -0
- package/src/annot/annot.js +212 -0
- package/src/annot/fileAttach.js +55 -0
- package/src/annot/freeText.js +82 -0
- package/src/annot/ink.js +77 -0
- package/src/annot/link.js +77 -0
- package/src/annot/markup.js +91 -0
- package/src/annot/popup.js +53 -0
- package/src/annot/projection.js +52 -0
- package/src/annot/redact.js +87 -0
- package/src/annot/square.js +132 -0
- package/src/annot/stamp.js +48 -0
- package/src/annot/text.js +54 -0
- package/src/annot/widget.js +61 -0
- package/src/associatedFiles/associatedFiles.js +86 -0
- package/src/bundles/pdf-full.js +91 -0
- package/src/bundles/pdf-large.js +81 -0
- package/src/bundles/pdf-legacy.js +107 -0
- package/src/content/color.js +114 -0
- package/src/content/graphics.js +192 -0
- package/src/content/images.js +160 -0
- package/src/content/ops.js +137 -0
- package/src/content/stream.js +154 -0
- package/src/content/text.js +125 -0
- package/src/crypto/aesGcm.js +123 -0
- package/src/crypto/permissions.js +112 -0
- package/src/crypto/security.js +327 -0
- package/src/crypto/standardV4.js +443 -0
- package/src/crypto/standardV5.js +306 -0
- package/src/crypto/standardV6.js +334 -0
- package/src/destination/destination.js +183 -0
- package/src/document/builder.js +618 -0
- package/src/document/catalog.js +100 -0
- package/src/document/document.js +472 -0
- package/src/document/encryptedWriter.js +554 -0
- package/src/document/incrementalWriter.js +514 -0
- package/src/document/page.js +131 -0
- package/src/document/pages.js +103 -0
- package/src/document/resources.js +146 -0
- package/src/document/writer.js +211 -0
- package/src/document/xrefStreamWriter.js +353 -0
- package/src/embedded/collection.js +102 -0
- package/src/embedded/embeddedFile.js +99 -0
- package/src/embedded/fileSpec.js +137 -0
- package/src/errors.js +78 -0
- package/src/extra/3d-richmedia.js +171 -0
- package/src/extra/annot-extended.js +200 -0
- package/src/extra/associated-files.js +131 -0
- package/src/extra/ccitt-fax-decoder.js +776 -0
- package/src/extra/color-spaces-extended.js +196 -0
- package/src/extra/content-ops-extended.js +153 -0
- package/src/extra/document-parts.js +149 -0
- package/src/extra/embedded-files-portfolio.js +234 -0
- package/src/extra/font-cid-typed.js +185 -0
- package/src/extra/font-color-tagging.js +144 -0
- package/src/extra/form-actions-extended.js +196 -0
- package/src/extra/info-dict-deprecated.js +137 -0
- package/src/extra/jbig2-read.js +169 -0
- package/src/extra/legacy-deprecated-annots.js +198 -0
- package/src/extra/legacy-deprecated-filters.js +167 -0
- package/src/extra/legacy-rc4-read.js +235 -0
- package/src/extra/legacy-xfa-read.js +104 -0
- package/src/extra/linearization-write.js +97 -0
- package/src/extra/misc.js +217 -0
- package/src/extra/optional-content-extended.js +142 -0
- package/src/extra/pdf-a-output-intent.js +112 -0
- package/src/extra/pdf-sandbox.js +88 -0
- package/src/extra/pdf-ua-tagged.js +116 -0
- package/src/extra/pdf-x-prepress.js +114 -0
- package/src/extra/redaction-iso32005.js +136 -0
- package/src/extra/shading-typed.js +222 -0
- package/src/extra/sig-aes-gcm.js +135 -0
- package/src/extra/sig-pades.js +242 -0
- package/src/extra/tagged-pdf-typed.js +203 -0
- package/src/extra/transparency-typed.js +135 -0
- package/src/extra/well-tagged-pdf.js +138 -0
- package/src/extra/xmp-extended.js +190 -0
- package/src/font/embed.js +480 -0
- package/src/font/encoding.js +92 -0
- package/src/font/font.js +101 -0
- package/src/font/type3.js +75 -0
- package/src/form/acroform.js +94 -0
- package/src/form/appearance.js +90 -0
- package/src/form/button.js +105 -0
- package/src/form/choice.js +152 -0
- package/src/form/fieldTree.js +120 -0
- package/src/form/signature.js +100 -0
- package/src/form/text.js +101 -0
- package/src/linearization/linearization.js +107 -0
- package/src/main.js +411 -0
- package/src/metadata/info.js +87 -0
- package/src/metadata/xmp.js +62 -0
- package/src/ocg/config.js +156 -0
- package/src/ocg/ocg.js +124 -0
- package/src/outline/outline.js +157 -0
- package/src/pdf.js +133 -0
- package/src/prepress/outputIntent.js +118 -0
- package/src/prepress/pageBoundary.js +108 -0
- package/src/sig/byteRange.js +306 -0
- package/src/sig/certChain.js +247 -0
- package/src/sig/dss.js +317 -0
- package/src/sig/oids.js +157 -0
- package/src/sig/sha1.js +142 -0
- package/src/sig/sign.js +1899 -0
- package/src/sig/signature.js +1441 -0
- package/src/sig/timestamp.js +236 -0
- package/src/syntax/crossRefStream.js +133 -0
- package/src/syntax/filters/ascii85.js +122 -0
- package/src/syntax/filters/asciiHex.js +83 -0
- package/src/syntax/filters/dispatch.js +176 -0
- package/src/syntax/filters/flate.js +316 -0
- package/src/syntax/filters/runLength.js +96 -0
- package/src/syntax/objStream.js +99 -0
- package/src/syntax/parser-obj.js +52 -0
- package/src/syntax/parser.js +321 -0
- package/src/syntax/serializer.js +221 -0
- package/src/syntax/tokenizer.js +290 -0
- package/src/syntax/trailer.js +76 -0
- package/src/syntax/xref.js +341 -0
- package/src/tagged/classMap.js +81 -0
- package/src/tagged/markedContent.js +123 -0
- package/src/tagged/parentTree.js +126 -0
- package/src/tagged/roleMap.js +107 -0
- package/src/tagged/structElement.js +138 -0
- package/src/tagged/structTree.js +94 -0
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfCrossRefStream
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: [pdfErrors, pdfParserObj]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfCrossRefStream
|
|
11
|
+
|
|
12
|
+
> Cross-reference stream parser `/Type /XRef` — ISO 32000-2 §7.5.8.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfCrossRefStream` | **Source** `packages/front/office/pdf/src/syntax/crossRefStream.js` | **Deps** `pdfErrors`, `pdfParserObj` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
An xref stream replaces the classical `xref` + `trailer` pair (PDF 1.5+). Its
|
|
17
|
+
dictionary carries `/Size`, `/W` (3-tuple of field widths in bytes), optionally
|
|
18
|
+
`/Index` (pairs `[first count …]`, default `[0 Size]`) and `/Prev`, plus every
|
|
19
|
+
trailer entry (`/Root`, `/Info`, `/Encrypt`, `/ID`). The payload encodes one
|
|
20
|
+
record per entry: three big-endian fields of the `/W` widths:
|
|
21
|
+
|
|
22
|
+
| type | f2 | f3 |
|
|
23
|
+
|------|----|----|
|
|
24
|
+
| 0 | offset of the next free object | gen |
|
|
25
|
+
| 1 | byte offset | gen |
|
|
26
|
+
| 2 | number of the containing ObjStm | index inside that ObjStm |
|
|
27
|
+
|
|
28
|
+
A zero width means "default value" (default type = 1). The payload must already
|
|
29
|
+
be decoded: [`pdfDocument`](../document/document.md) composes this module and
|
|
30
|
+
runs the stream through [`pdfFilterDispatch`](./filters/dispatch.md) first, so
|
|
31
|
+
reading a 1.5+ file needs no explicit call. Callers driving the parser on its
|
|
32
|
+
own decode `/Filter` themselves.
|
|
33
|
+
|
|
34
|
+
## Resolve
|
|
35
|
+
|
|
36
|
+
```js
|
|
37
|
+
const xrefStm = runtime.resolve('pdfCrossRefStream');
|
|
38
|
+
// Returns: { parseCrossRefStream }
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## API
|
|
42
|
+
|
|
43
|
+
| Method | Signature | Returns |
|
|
44
|
+
|--------|-----------|---------|
|
|
45
|
+
| `parseCrossRefStream` | `(decoded: Uint8Array, dict: PdfDict) => { entries, size, trailer }` | Xref table plus trailer. |
|
|
46
|
+
|
|
47
|
+
### Return shape
|
|
48
|
+
|
|
49
|
+
```js
|
|
50
|
+
{
|
|
51
|
+
entries: {
|
|
52
|
+
[num]: {
|
|
53
|
+
type: 0 | 1 | 2,
|
|
54
|
+
offset: number, // type 1 only — 0 otherwise
|
|
55
|
+
gen: number, // type 0/1
|
|
56
|
+
free: boolean,
|
|
57
|
+
objStm?: number, // type 2
|
|
58
|
+
index?: number // type 2
|
|
59
|
+
}
|
|
60
|
+
},
|
|
61
|
+
size: number, // /Size, or the entry count
|
|
62
|
+
trailer: PdfDict // the dictionary itself (carries /Root /Info …)
|
|
63
|
+
}
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## Examples
|
|
67
|
+
|
|
68
|
+
### Parse an xref-stream section
|
|
69
|
+
|
|
70
|
+
```js
|
|
71
|
+
const xrefStm = runtime.resolve('pdfCrossRefStream');
|
|
72
|
+
const decoded = filterDispatch.decode(streamObj);
|
|
73
|
+
const { entries, size, trailer } = xrefStm.parseCrossRefStream(decoded, streamObj.dict);
|
|
74
|
+
console.log(size, Object.keys(entries).length);
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
### Resolve a compressed object
|
|
78
|
+
|
|
79
|
+
```js
|
|
80
|
+
const e = entries[12];
|
|
81
|
+
if (e.type === 2) {
|
|
82
|
+
const objStmRef = { type: 'ref', num: e.objStm, gen: 0 };
|
|
83
|
+
// → fetch the streamObj at objStmRef, decode, parseObjectStream,
|
|
84
|
+
// then pick the member at e.index.
|
|
85
|
+
}
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
### Non-trivial `/Index`
|
|
89
|
+
|
|
90
|
+
```js
|
|
91
|
+
// dict.entries.Index = array([int(0), int(1), int(10), int(3)])
|
|
92
|
+
// → 4 entries in total: num 0, then num 10..12.
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Errors
|
|
96
|
+
|
|
97
|
+
| Code | Class | When |
|
|
98
|
+
|------|-------|------|
|
|
99
|
+
| `pdf/xrefstm/bad-input` | `ParseError` | `decoded` is not a `Uint8Array`. |
|
|
100
|
+
| `pdf/xrefstm/bad-dict` | `ParseError` | `dict` is not a typed dictionary. |
|
|
101
|
+
| `pdf/xrefstm/wrong-type` | `ParseError` | `/Type` missing or not `/XRef`. |
|
|
102
|
+
| `pdf/xrefstm/bad-W` | `ParseError` | `/W` not an array, or not 3 entries. |
|
|
103
|
+
| `pdf/xrefstm/bad-W-entry` | `ParseError` | A `/W` entry is not a non-negative int. |
|
|
104
|
+
| `pdf/xrefstm/bad-W-zero` | `ParseError` | The `/W` widths sum to 0. |
|
|
105
|
+
| `pdf/xrefstm/bad-Index` | `ParseError` | `/Index` not an array, or odd length. |
|
|
106
|
+
| `pdf/xrefstm/bad-Index-entry` | `ParseError` | A `/Index` entry is not an int. |
|
|
107
|
+
| `pdf/xrefstm/missing-int` | `ParseError` | `/Size` required and missing. |
|
|
108
|
+
| `pdf/xrefstm/truncated` | `ParseError` | Payload too short for the announced entry count. |
|
|
109
|
+
|
|
110
|
+
## See also
|
|
111
|
+
|
|
112
|
+
- [`pdfXref`](./xref.md) — classical table §7.5.4.
|
|
113
|
+
- [`pdfObjStream`](./objStream.md) — resolves type-2 entries.
|
|
114
|
+
- [`pdfTrailer`](./trailer.md) — typing of the trailer dictionary.
|
|
115
|
+
- [`pdfDocument`](../document/document.md) — composes this parser in the section walk.
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# Filters — ISO 32000-2 §7.4
|
|
2
|
+
|
|
3
|
+
The `/Filter` + `/DecodeParms` chain on streams. Every filter exposes
|
|
4
|
+
`{ decode, encode }`.
|
|
5
|
+
|
|
6
|
+
| Module | PDF filter | Abbrev. | Spec | Deps |
|
|
7
|
+
|--------|------------|---------|------|------|
|
|
8
|
+
| [`pdfFlate`](./flate.md) | `FlateDecode` | `Fl` | §7.4.4 | `pdfErrors`, `zlib` (`@awacloud/fw`) |
|
|
9
|
+
| [`pdfAsciiHex`](./asciiHex.md) | `ASCIIHexDecode` | `AHx` | §7.4.2 | `pdfErrors` |
|
|
10
|
+
| [`pdfAscii85`](./ascii85.md) | `ASCII85Decode` | `A85` | §7.4.3 | `pdfErrors` |
|
|
11
|
+
| [`pdfRunLength`](./runLength.md) | `RunLengthDecode` | `RL` | §7.4.5 | `pdfErrors` |
|
|
12
|
+
| [`pdfFilterDispatch`](./dispatch.md) | orchestrator | — | §7.4 | `pdfErrors` + the four above |
|
|
13
|
+
|
|
14
|
+
## Coverage (decode / encode)
|
|
15
|
+
|
|
16
|
+
The table below describes the `decoders` map that `pdfFilterDispatch` ships out
|
|
17
|
+
of the box.
|
|
18
|
+
|
|
19
|
+
| Filter | decode | encode | Notes |
|
|
20
|
+
|--------|--------|--------|-------|
|
|
21
|
+
| `FlateDecode` | yes | yes | `/Predictor` 1, 2 and 10–15 supported in both directions. |
|
|
22
|
+
| `ASCIIHexDecode` | yes | yes | Encodes uppercase, wraps every 32 bytes. |
|
|
23
|
+
| `ASCII85Decode` | yes | yes | Handles `z` and `~>`. |
|
|
24
|
+
| `RunLengthDecode` | yes | yes | Greedy encoder. |
|
|
25
|
+
| `DCTDecode` | passthrough | passthrough | The PDF stores the raw JPEG bytes. |
|
|
26
|
+
| `JPXDecode` | passthrough | passthrough | Raw JPEG 2000. |
|
|
27
|
+
| `Crypt` | passthrough | passthrough | Real handling lives in the [crypto layer](../../crypto/README.md). |
|
|
28
|
+
| `LZWDecode` | not registered | — | An implementation exists on the [`pdfLegacyDeprecatedFilters`](../../extra/legacy-deprecated-filters.md) extra (`lzwDecode`/`lzwEncode`, over `@awacloud/fw/io/compress/lzw`). |
|
|
29
|
+
| `CCITTFaxDecode` | not registered | — | Implementations exist on `pdfCcittFaxDecoder` and on `pdfLegacyDeprecatedFilters` (`ccittFaxDecode`). |
|
|
30
|
+
| `JBIG2Decode` | not registered | — | Header-only support on the [`pdfJbig2Read`](../../extra/jbig2-read.md) extra. |
|
|
31
|
+
|
|
32
|
+
No extra auto-registers itself into the dispatch map: wire the ones you need
|
|
33
|
+
with `dispatch.register(name, impl)` — see [`pdfFilterDispatch`](./dispatch.md).
|
|
34
|
+
|
|
35
|
+
`Fl`, `AHx`, `A85`, `RL`, `LZW`, `DCT`, `JPX` and `CCF` are recognised as
|
|
36
|
+
abbreviations and expanded to the full filter name before dispatch — so an
|
|
37
|
+
unregistered `LZW` raises `pdf/filter/unsupported` under the name `LZWDecode`.
|
|
38
|
+
|
|
39
|
+
## Common pattern
|
|
40
|
+
|
|
41
|
+
```js
|
|
42
|
+
const dispatch = runtime.resolve('pdfFilterDispatch');
|
|
43
|
+
const decoded = dispatch.decode(streamObj);
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## See also
|
|
47
|
+
|
|
48
|
+
- [Syntax layer](../README.md)
|
|
49
|
+
- [`pdfObjStream`](../objStream.md) — decoding is mandatory before parsing.
|
|
50
|
+
- [`pdfCrossRefStream`](../crossRefStream.md) — likewise.
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfAscii85
|
|
3
|
+
category: pdf/syntax/filters
|
|
4
|
+
dependencies: [pdfErrors]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfAscii85
|
|
11
|
+
|
|
12
|
+
> `ASCII85Decode` ISO 32000-2 §7.4.3 — base 85, 5 chars ↔ 4 bytes.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfAscii85` | **Source** `packages/front/office/pdf/src/syntax/filters/ascii85.js` | **Deps** `pdfErrors` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Five ASCII characters in the `!`..`u` range decode to 4 binary bytes,
|
|
17
|
+
interpreted as a big-endian 32-bit integer in base 85. The special character `z`
|
|
18
|
+
stands for 4 zero bytes and may appear only at a group boundary. The stream is
|
|
19
|
+
terminated by `~>`. The final group may be 2..4 characters, decoding to 1..3
|
|
20
|
+
trailing bytes (padded with `u` = 84, keeping `count - 1` bytes). Whitespace is
|
|
21
|
+
ignored.
|
|
22
|
+
|
|
23
|
+
## Resolve
|
|
24
|
+
|
|
25
|
+
```js
|
|
26
|
+
const a85 = runtime.resolve('pdfAscii85');
|
|
27
|
+
// Returns: { decode, encode }
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## API
|
|
31
|
+
|
|
32
|
+
| Method | Signature | Returns |
|
|
33
|
+
|--------|-----------|---------|
|
|
34
|
+
| `decode` | `(bytes: Uint8Array) => Uint8Array` | Binary bytes. |
|
|
35
|
+
| `encode` | `(bytes: Uint8Array) => Uint8Array` | ASCII85 plus the `~>` EOD. |
|
|
36
|
+
|
|
37
|
+
## Examples
|
|
38
|
+
|
|
39
|
+
### Decode
|
|
40
|
+
|
|
41
|
+
```js
|
|
42
|
+
const a85 = runtime.resolve('pdfAscii85');
|
|
43
|
+
const src = new TextEncoder().encode('87cURD]j~>');
|
|
44
|
+
const out = a85.decode(src);
|
|
45
|
+
new TextDecoder().decode(out); // 'Hello'
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### Zero group (`z`)
|
|
49
|
+
|
|
50
|
+
```js
|
|
51
|
+
a85.decode(new TextEncoder().encode('z~>')); // [0,0,0,0]
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### Round trip with a tail
|
|
55
|
+
|
|
56
|
+
```js
|
|
57
|
+
const bin = new Uint8Array([1, 2, 3]); // 3 bytes → 4 chars + ~>
|
|
58
|
+
const enc = a85.encode(bin);
|
|
59
|
+
const back = a85.decode(enc);
|
|
60
|
+
// back equals bin.
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Errors
|
|
64
|
+
|
|
65
|
+
| Code | Class | When |
|
|
66
|
+
|------|-------|------|
|
|
67
|
+
| `pdf/ascii85/bad-input` | `ParseError` | Argument is not a `Uint8Array`. |
|
|
68
|
+
| `pdf/ascii85/bad-eod` | `ParseError` | `~` not followed by `>`. |
|
|
69
|
+
| `pdf/ascii85/bad-z` | `ParseError` | `z` inside a group (`count !== 0`). |
|
|
70
|
+
| `pdf/ascii85/bad-digit` | `ParseError` | Character outside `!`..`u`, not whitespace, not `z`/`~`. |
|
|
71
|
+
|
|
72
|
+
## See also
|
|
73
|
+
|
|
74
|
+
- [`pdfAsciiHex`](./asciiHex.md) — simpler alternative, 2 chars per byte.
|
|
75
|
+
- [`pdfFilterDispatch`](./dispatch.md)
|
|
76
|
+
- [Filters index](./README.md)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfAsciiHex
|
|
3
|
+
category: pdf/syntax/filters
|
|
4
|
+
dependencies: [pdfErrors]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfAsciiHex
|
|
11
|
+
|
|
12
|
+
> `ASCIIHexDecode` ISO 32000-2 §7.4.2 — pairs of hex digits → bytes.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfAsciiHex` | **Source** `packages/front/office/pdf/src/syntax/filters/asciiHex.js` | **Deps** `pdfErrors` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Each ASCII hex pair decodes to one byte. Whitespace (`SP`, `HT`, `LF`, `CR`,
|
|
17
|
+
`FF`, `NUL`) is ignored. The `>` terminator stops the stream. An odd number of
|
|
18
|
+
digits is treated as if a trailing `0` were present (spec §7.4.2). Mixed case is
|
|
19
|
+
tolerated on decode; the encoder emits uppercase hex with a line break every 32
|
|
20
|
+
bytes and a final `>`.
|
|
21
|
+
|
|
22
|
+
## Resolve
|
|
23
|
+
|
|
24
|
+
```js
|
|
25
|
+
const ahx = runtime.resolve('pdfAsciiHex');
|
|
26
|
+
// Returns: { decode, encode }
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## API
|
|
30
|
+
|
|
31
|
+
| Method | Signature | Returns |
|
|
32
|
+
|--------|-----------|---------|
|
|
33
|
+
| `decode` | `(bytes: Uint8Array) => Uint8Array` | Binary bytes. |
|
|
34
|
+
| `encode` | `(bytes: Uint8Array) => Uint8Array` | ASCII hex plus the `>` EOD. |
|
|
35
|
+
|
|
36
|
+
## Examples
|
|
37
|
+
|
|
38
|
+
### Decode
|
|
39
|
+
|
|
40
|
+
```js
|
|
41
|
+
const ahx = runtime.resolve('pdfAsciiHex');
|
|
42
|
+
const src = new TextEncoder().encode('48 65 6C 6C 6F>');
|
|
43
|
+
const out = ahx.decode(src);
|
|
44
|
+
new TextDecoder().decode(out); // 'Hello'
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
### Trailing odd digit (implicit `0`)
|
|
48
|
+
|
|
49
|
+
```js
|
|
50
|
+
ahx.decode(new TextEncoder().encode('A>')); // [0xA0]
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
### Encode round trip
|
|
54
|
+
|
|
55
|
+
```js
|
|
56
|
+
const bin = new Uint8Array([0xDE, 0xAD, 0xBE, 0xEF]);
|
|
57
|
+
const enc = ahx.encode(bin); // 'DEADBEEF>'
|
|
58
|
+
const back = ahx.decode(enc);
|
|
59
|
+
// back equals bin.
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## Errors
|
|
63
|
+
|
|
64
|
+
| Code | Class | When |
|
|
65
|
+
|------|-------|------|
|
|
66
|
+
| `pdf/asciiHex/bad-input` | `ParseError` | Argument is not a `Uint8Array`. |
|
|
67
|
+
| `pdf/asciiHex/bad-digit` | `ParseError` | Byte outside `0-9`, `A-F`, `a-f`, whitespace, or `>`. |
|
|
68
|
+
|
|
69
|
+
## See also
|
|
70
|
+
|
|
71
|
+
- [`pdfAscii85`](./ascii85.md) — denser 5-for-4 encoding.
|
|
72
|
+
- [`pdfFilterDispatch`](./dispatch.md)
|
|
73
|
+
- [Filters index](./README.md)
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfFilterDispatch
|
|
3
|
+
category: pdf/syntax/filters
|
|
4
|
+
dependencies: [pdfErrors, pdfFlate, pdfAsciiHex, pdfAscii85, pdfRunLength]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfFilterDispatch
|
|
11
|
+
|
|
12
|
+
> The `/Filter` + `/DecodeParms` chain — ISO 32000-2 §7.4.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfFilterDispatch` | **Source** `packages/front/office/pdf/src/syntax/filters/dispatch.js` | **Deps** `pdfErrors`, `pdfFlate`, `pdfAsciiHex`, `pdfAscii85`, `pdfRunLength` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Reads `/Filter` (a name or an array of names) and applies the decoders **in
|
|
17
|
+
order** — the first name is applied first, which is the spec convention and the
|
|
18
|
+
reverse of the encoding order. Recognised PDF abbreviations: `Fl`, `AHx`, `A85`,
|
|
19
|
+
`RL`, `LZW`, `DCT`, `JPX`, `CCF`. Opaque filters exposed as passthrough:
|
|
20
|
+
`DCTDecode`, `JPXDecode`, `Crypt`. Any filter absent from `decoders` raises
|
|
21
|
+
`pdf/filter/unsupported`.
|
|
22
|
+
|
|
23
|
+
`decodeStream` marshals `/DecodeParms` from the typed AST form
|
|
24
|
+
(`{type:'dict', entries:{K: {type, value}}}`) to a plain `key → value`
|
|
25
|
+
object (reading `entry.value` per key) before handing it to a decoder —
|
|
26
|
+
decoders such as `pdfFlate` read plain properties (`params.Predictor`,
|
|
27
|
+
`params.Columns`, …), never typed nodes. The three `/DecodeParms` shapes
|
|
28
|
+
are all handled: absent (→ `null` params for every filter), a single dict
|
|
29
|
+
(→ applies to the first filter only, `null` for the rest, per spec), and
|
|
30
|
+
an array parallel to `/Filter` (→ one marshalled dict per filter, `null`
|
|
31
|
+
entries preserved).
|
|
32
|
+
|
|
33
|
+
## Resolve
|
|
34
|
+
|
|
35
|
+
```js
|
|
36
|
+
const dispatch = runtime.resolve('pdfFilterDispatch');
|
|
37
|
+
// Returns: { decode, decodeChain, encodeChain, register, names, decoders,
|
|
38
|
+
// normalizeFilterList, applyDecodeChain, applyEncodeChain,
|
|
39
|
+
// decodeStream }
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## API
|
|
43
|
+
|
|
44
|
+
| Member | Signature | Returns |
|
|
45
|
+
|--------|-----------|---------|
|
|
46
|
+
| `decode` | `(streamObj: PdfStream) => Uint8Array` | End-to-end decode from the typed stream, against the built-in `decoders`. |
|
|
47
|
+
| `decodeChain` | `(bytes, filters: string[], params?: object[]) => Uint8Array` | Explicit decode against the built-in `decoders`. |
|
|
48
|
+
| `encodeChain` | `(bytes, filters: string[], params?: object[]) => Uint8Array` | Explicit encode against the built-in `decoders`. |
|
|
49
|
+
| `register` | `(name: string, impl: { decode, encode? }) => void` | Adds or replaces a decoder in `decoders`. |
|
|
50
|
+
| `names` | `() => string[]` | Lists the currently registered filter names. |
|
|
51
|
+
| `decoders` | `object` | The live map itself (escape hatch — mutating it is what `register` does). |
|
|
52
|
+
| `normalizeFilterList` | `(filterEntry) => string[]` | Expands abbreviations and returns the canonical name list. Useful to inspect a chain without decoding. |
|
|
53
|
+
| `applyDecodeChain` | `(bytes, filters, params, decoders) => Uint8Array` | Lower-level decode against a **caller-supplied** decoder map. |
|
|
54
|
+
| `applyEncodeChain` | `(bytes, filters, params, decoders) => Uint8Array` | Lower-level encode against a caller-supplied decoder map. |
|
|
55
|
+
| `decodeStream` | `(streamObj, decoders) => Uint8Array` | Lower-level `decode`: reads `/Filter` + `/DecodeParms` off the stream dictionary, then delegates to `applyDecodeChain`. |
|
|
56
|
+
|
|
57
|
+
`decode`, `decodeChain` and `encodeChain` are thin bindings of the three
|
|
58
|
+
lower-level helpers over the module's own `decoders` map; the `apply*`/
|
|
59
|
+
`decodeStream` triple is exported so a caller can run a chain against an
|
|
60
|
+
isolated map without mutating the shared one.
|
|
61
|
+
|
|
62
|
+
## Examples
|
|
63
|
+
|
|
64
|
+
### End-to-end decode
|
|
65
|
+
|
|
66
|
+
```js
|
|
67
|
+
const dispatch = runtime.resolve('pdfFilterDispatch');
|
|
68
|
+
const decoded = dispatch.decode(streamObj); // reads Filter + DecodeParms
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
### Explicit chain
|
|
72
|
+
|
|
73
|
+
```js
|
|
74
|
+
const out = dispatch.decodeChain(bytes, ['ASCII85Decode', 'FlateDecode']);
|
|
75
|
+
// Applies ASCII85 then Flate (= array order).
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### Register a custom decoder (LZW)
|
|
79
|
+
|
|
80
|
+
```js
|
|
81
|
+
const legacy = runtime.resolve('pdfLegacyDeprecatedFilters');
|
|
82
|
+
dispatch.register('LZWDecode', {
|
|
83
|
+
decode: (bytes, params) => legacy.lzwDecode(bytes, params),
|
|
84
|
+
encode: (bytes) => legacy.lzwEncode(bytes)
|
|
85
|
+
});
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
### Decode against an isolated map
|
|
89
|
+
|
|
90
|
+
```js
|
|
91
|
+
const isolated = { ...dispatch.decoders, DCTDecode: myJpegDecoder };
|
|
92
|
+
const out = dispatch.decodeStream(streamObj, isolated);
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
### Inspect a chain without decoding
|
|
96
|
+
|
|
97
|
+
```js
|
|
98
|
+
dispatch.normalizeFilterList(streamObj.dict.entries.Filter);
|
|
99
|
+
// ['ASCII85Decode', 'FlateDecode']
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
### List the available filters
|
|
103
|
+
|
|
104
|
+
```js
|
|
105
|
+
dispatch.names();
|
|
106
|
+
// ['FlateDecode', 'ASCIIHexDecode', 'ASCII85Decode', 'RunLengthDecode',
|
|
107
|
+
// 'DCTDecode', 'JPXDecode', 'Crypt']
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## Errors
|
|
111
|
+
|
|
112
|
+
| Code | Class | When |
|
|
113
|
+
|------|-------|------|
|
|
114
|
+
| `pdf/filter/non-name` | `ParseError` | An entry of the `/Filter` array is not a name. |
|
|
115
|
+
| `pdf/filter/bad-type` | `ParseError` | `/Filter` is neither a name nor an array. |
|
|
116
|
+
| `pdf/filter/unsupported` | `ParseError` | Filter absent from `decoders`. |
|
|
117
|
+
| `pdf/filter/no-encoder` | `ParseError` | `encodeChain` invoked for a filter with no `encode`. |
|
|
118
|
+
| `pdf/filter/not-stream` | `ParseError` | `decode()`/`decodeStream()` given something other than a typed stream. |
|
|
119
|
+
| `pdf/filter/bad-decodeparms` | `ParseError` | `/DecodeParms` is neither a dict, an array, nor null. |
|
|
120
|
+
|
|
121
|
+
## See also
|
|
122
|
+
|
|
123
|
+
- [Filters index](./README.md)
|
|
124
|
+
- [`pdfFlate`](./flate.md), [`pdfAsciiHex`](./asciiHex.md), [`pdfAscii85`](./ascii85.md), [`pdfRunLength`](./runLength.md)
|
|
125
|
+
- [`pdfObjStream`](../objStream.md), [`pdfCrossRefStream`](../crossRefStream.md) — consumers.
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfFlate
|
|
3
|
+
category: pdf/syntax/filters
|
|
4
|
+
dependencies: [pdfErrors, zlib]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfFlate
|
|
11
|
+
|
|
12
|
+
> `FlateDecode` ISO 32000-2 §7.4.4 — wrapper over `@awacloud/fw` zlib (RFC 1950), with predictors.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfFlate` | **Source** `packages/front/office/pdf/src/syntax/filters/flate.js` | **Deps** `pdfErrors`, `zlib` (`@awacloud/fw/io/compress/zlib`) | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
PDF `FlateDecode` is full zlib framing (CMF/FLG header + deflate payload +
|
|
17
|
+
Adler32 trailer). The raw compression is delegated to `@awacloud/fw`'s `zlib` module,
|
|
18
|
+
which produces and consumes exactly that format.
|
|
19
|
+
|
|
20
|
+
**`/DecodeParms /Predictor` is handled here**, in both directions:
|
|
21
|
+
|
|
22
|
+
| `Predictor` | Meaning |
|
|
23
|
+
|-------------|---------|
|
|
24
|
+
| `1` | None (default) |
|
|
25
|
+
| `2` | TIFF Predictor 2 — requires `BitsPerComponent = 8` |
|
|
26
|
+
| `10`–`14` | PNG None / Sub / Up / Average / Paeth, per row |
|
|
27
|
+
| `15` | PNG optimum — the encoder picks per row; the decoder reads the tag byte either way |
|
|
28
|
+
|
|
29
|
+
`Columns`, `Colors` and `BitsPerComponent` are read alongside `Predictor`
|
|
30
|
+
(defaults `1`, `1`, `8`) and accepted in both the PDF spelling (`Predictor`,
|
|
31
|
+
`Columns`, …) and the lowercase spelling (`predictor`, `columns`, …).
|
|
32
|
+
|
|
33
|
+
## Resolve
|
|
34
|
+
|
|
35
|
+
```js
|
|
36
|
+
const flate = runtime.resolve('pdfFlate');
|
|
37
|
+
// Returns: { decode, encode }
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## API
|
|
41
|
+
|
|
42
|
+
| Method | Signature | Returns |
|
|
43
|
+
|--------|-----------|---------|
|
|
44
|
+
| `decode` | `(bytes: Uint8Array, params?: object) => Uint8Array` | Inflated payload, un-predicted when `params` requests it; the decoded prefix, flagged `truncated`, for a stream that ends before its final block. |
|
|
45
|
+
| `encode` | `(bytes: Uint8Array, params?: object) => Uint8Array` | Predicted (when requested) then zlib-deflated payload. |
|
|
46
|
+
|
|
47
|
+
`params` is optional: omit it, or leave `Predictor` at `1`, and the byte stream
|
|
48
|
+
is passed through the raw zlib codec untouched.
|
|
49
|
+
|
|
50
|
+
## Truncated streams
|
|
51
|
+
|
|
52
|
+
Some producers write `FlateDecode` streams whose deflate data ends before the
|
|
53
|
+
final block — the input runs out mid-stream, though `/Length` is exact.
|
|
54
|
+
Viewers display what decodes, and so does `decode`: when the strict inflate
|
|
55
|
+
fails only because the input ended before the final block, the payload is
|
|
56
|
+
decoded again through fw's streaming decoder (`UnzlibStream`, without the
|
|
57
|
+
end-of-input signal) and the bytes decoded so far are returned.
|
|
58
|
+
|
|
59
|
+
- The returned `Uint8Array` carries a **non-enumerable** own property
|
|
60
|
+
`truncated: true`. It is invisible to `Object.keys`, spread and deep
|
|
61
|
+
equality, so callers that ignore it see an ordinary byte array; callers
|
|
62
|
+
that care test `out.truncated === true`. A complete stream never carries it.
|
|
63
|
+
- With a `Predictor`, only the **whole** predicted rows of the prefix are
|
|
64
|
+
decoded (a trailing partial row is dropped), so a truncated image or xref
|
|
65
|
+
stream does not fail with `pdf/flate/png-row-mismatch`. A complete stream
|
|
66
|
+
keeps the strict row check.
|
|
67
|
+
- The streaming decoder holds back the last 4 input bytes as a presumed
|
|
68
|
+
Adler-32 trailer, so the recovered prefix can stop a few bytes short of
|
|
69
|
+
everything the input encodes.
|
|
70
|
+
|
|
71
|
+
Everything else still throws `pdf/flate/inflate-failed`: a bad zlib header,
|
|
72
|
+
an invalid block type, length or distance, a stream truncated inside its
|
|
73
|
+
header, and a truncated stream from which nothing could be decoded. The
|
|
74
|
+
error's `cause` is the original strict-inflate error.
|
|
75
|
+
|
|
76
|
+
```js
|
|
77
|
+
const out = flate.decode(streamObj.raw);
|
|
78
|
+
if (out.truncated) {
|
|
79
|
+
// a decoded prefix — the stream ended before its final deflate block
|
|
80
|
+
}
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
## Examples
|
|
84
|
+
|
|
85
|
+
### Decode a Flate stream
|
|
86
|
+
|
|
87
|
+
```js
|
|
88
|
+
const flate = runtime.resolve('pdfFlate');
|
|
89
|
+
const decoded = flate.decode(streamObj.raw);
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
### Decode with a PNG predictor
|
|
93
|
+
|
|
94
|
+
```js
|
|
95
|
+
const decoded = flate.decode(streamObj.raw, {
|
|
96
|
+
Predictor: 12, Columns: 5, Colors: 1, BitsPerComponent: 8
|
|
97
|
+
});
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
### Round trip
|
|
101
|
+
|
|
102
|
+
```js
|
|
103
|
+
const enc = new TextEncoder().encode('hello, pdf');
|
|
104
|
+
const compressed = flate.encode(enc);
|
|
105
|
+
const back = flate.decode(compressed);
|
|
106
|
+
new TextDecoder().decode(back); // 'hello, pdf'
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
### Through the dispatch
|
|
110
|
+
|
|
111
|
+
```js
|
|
112
|
+
const dispatch = runtime.resolve('pdfFilterDispatch');
|
|
113
|
+
const out = dispatch.decode(streamObj); // resolves /Filter and /DecodeParms
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## Errors
|
|
117
|
+
|
|
118
|
+
| Code | Class | When |
|
|
119
|
+
|------|-------|------|
|
|
120
|
+
| `pdf/flate/missing-fw` | `ParseError` | Factory invoked without the `@awacloud/fw` zlib module (no `zlibSync`/`unzlibSync`). |
|
|
121
|
+
| `pdf/flate/bad-input` | `ParseError` | Argument is not a `Uint8Array`. |
|
|
122
|
+
| `pdf/flate/inflate-failed` | `ParseError` | `unzlibSync` threw — corrupt or non-zlib payload, or a truncated stream from which nothing could be decoded (see [Truncated streams](#truncated-streams)). `cause` is the original error. |
|
|
123
|
+
| `pdf/flate/deflate-failed` | `ParseError` | `zlibSync` threw. |
|
|
124
|
+
| `pdf/flate/bad-predictor` | `ParseError` | `Predictor` outside `{1, 2, 10..15}`. |
|
|
125
|
+
| `pdf/flate/bad-predictor-params` | `ParseError` | `Columns`, `Colors` or `BitsPerComponent` ≤ 0. |
|
|
126
|
+
| `pdf/flate/bad-png-filter` | `ParseError` | Row tag byte outside `0..4`. |
|
|
127
|
+
| `pdf/flate/tiff-bpc-unsupported` | `ParseError` | TIFF Predictor 2 with `BitsPerComponent ≠ 8`. |
|
|
128
|
+
| `pdf/flate/png-row-mismatch` | `ParseError` | Payload length of a complete stream not a whole number of predicted rows. |
|
|
129
|
+
|
|
130
|
+
## See also
|
|
131
|
+
|
|
132
|
+
- [`pdfFilterDispatch`](./dispatch.md) — filter orchestrator.
|
|
133
|
+
- [Filters index](./README.md)
|
|
134
|
+
- [`pdfObjStream`](../objStream.md) — typical consumer (ObjStm streams are almost always Flate-encoded).
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfRunLength
|
|
3
|
+
category: pdf/syntax/filters
|
|
4
|
+
dependencies: [pdfErrors]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfRunLength
|
|
11
|
+
|
|
12
|
+
> `RunLengthDecode` ISO 32000-2 §7.4.5 — run-length coding.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfRunLength` | **Source** `packages/front/office/pdf/src/syntax/filters/runLength.js` | **Deps** `pdfErrors` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Each control byte `n` is read as:
|
|
17
|
+
|
|
18
|
+
- `0..127` → copy the next `n + 1` bytes literally (1..128 bytes);
|
|
19
|
+
- `128` → end of data (EOD), decoding stops;
|
|
20
|
+
- `129..255` → repeat the next byte `257 - n` times (2..128 repetitions).
|
|
21
|
+
|
|
22
|
+
The encoder is greedy: it first tries a repeat run when 3 or more identical
|
|
23
|
+
bytes are available, otherwise it accumulates up to 128 literal bytes, stopping
|
|
24
|
+
just before a productive run.
|
|
25
|
+
|
|
26
|
+
## Resolve
|
|
27
|
+
|
|
28
|
+
```js
|
|
29
|
+
const rl = runtime.resolve('pdfRunLength');
|
|
30
|
+
// Returns: { decode, encode }
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## API
|
|
34
|
+
|
|
35
|
+
| Method | Signature | Returns |
|
|
36
|
+
|--------|-----------|---------|
|
|
37
|
+
| `decode` | `(bytes: Uint8Array) => Uint8Array` | Expanded bytes. |
|
|
38
|
+
| `encode` | `(bytes: Uint8Array) => Uint8Array` | Encoded bytes plus the EOD (`128`). |
|
|
39
|
+
|
|
40
|
+
## Examples
|
|
41
|
+
|
|
42
|
+
### Decode a repeat run
|
|
43
|
+
|
|
44
|
+
```js
|
|
45
|
+
const rl = runtime.resolve('pdfRunLength');
|
|
46
|
+
// 0xFB = 251 → repeat the next byte (257-251 = 6) times; 128 = EOD
|
|
47
|
+
const src = new Uint8Array([0xFB, 0x41, 0x80]);
|
|
48
|
+
rl.decode(src); // [0x41, 0x41, 0x41, 0x41, 0x41, 0x41] ('AAAAAA')
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
### Decode a literal run
|
|
52
|
+
|
|
53
|
+
```js
|
|
54
|
+
// 0x02 → copy (2+1=3) literal bytes
|
|
55
|
+
rl.decode(new Uint8Array([0x02, 0x48, 0x69, 0x21, 0x80])); // 'Hi!'
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
### Round trip
|
|
59
|
+
|
|
60
|
+
```js
|
|
61
|
+
const bin = new Uint8Array([1, 1, 1, 1, 1, 2, 3, 4]);
|
|
62
|
+
const enc = rl.encode(bin);
|
|
63
|
+
const back = rl.decode(enc);
|
|
64
|
+
// back equals bin.
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Errors
|
|
68
|
+
|
|
69
|
+
| Code | Class | When |
|
|
70
|
+
|------|-------|------|
|
|
71
|
+
| `pdf/runLength/bad-input` | `ParseError` | Argument is not a `Uint8Array`. |
|
|
72
|
+
| `pdf/runLength/truncated-literal` | `ParseError` | Literal run announced but the payload is truncated. |
|
|
73
|
+
| `pdf/runLength/truncated-repeat` | `ParseError` | Repeat run announced but no value byte follows. |
|
|
74
|
+
|
|
75
|
+
## See also
|
|
76
|
+
|
|
77
|
+
- [`pdfFilterDispatch`](./dispatch.md)
|
|
78
|
+
- [Filters index](./README.md)
|