@awacloud/pdf 0.0.0-stage → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +609 -0
- package/LICENSE +661 -0
- package/NOTICE +77 -0
- package/README.md +363 -2
- package/dist/build/index.js +21 -0
- package/dist/build/pdf-full-rw.js +10972 -0
- package/dist/build/pdf-full-rw.meta.json +105 -0
- package/dist/build/pdf-full-rw.min.js +53 -0
- package/dist/build/pdf-full.js +6078 -0
- package/dist/build/pdf-full.meta.json +90 -0
- package/dist/build/pdf-full.min.js +32 -0
- package/dist/build/pdf-large-rw.js +10367 -0
- package/dist/build/pdf-large-rw.meta.json +99 -0
- package/dist/build/pdf-large-rw.min.js +53 -0
- package/dist/build/pdf-large.js +5473 -0
- package/dist/build/pdf-large.meta.json +84 -0
- package/dist/build/pdf-large.min.js +32 -0
- package/dist/build/pdf-legacy-rw.js +12402 -0
- package/dist/build/pdf-legacy-rw.meta.json +110 -0
- package/dist/build/pdf-legacy-rw.min.js +53 -0
- package/dist/build/pdf-legacy.js +7508 -0
- package/dist/build/pdf-legacy.meta.json +95 -0
- package/dist/build/pdf-legacy.min.js +32 -0
- package/dist/build/pdf-rw.js +7578 -0
- package/dist/build/pdf-rw.meta.json +77 -0
- package/dist/build/pdf-rw.min.js +53 -0
- package/dist/build/pdf.js +2684 -0
- package/dist/build/pdf.meta.json +62 -0
- package/dist/build/pdf.min.js +32 -0
- package/dist/standalone/pdf-full-rw.js +16798 -0
- package/dist/standalone/pdf-full-rw.meta.json +78 -0
- package/dist/standalone/pdf-full-rw.min.js +56 -0
- package/dist/standalone/pdf-full.js +11904 -0
- package/dist/standalone/pdf-full.meta.json +63 -0
- package/dist/standalone/pdf-full.min.js +35 -0
- package/dist/standalone/pdf-large-rw.js +16193 -0
- package/dist/standalone/pdf-large-rw.meta.json +72 -0
- package/dist/standalone/pdf-large-rw.min.js +56 -0
- package/dist/standalone/pdf-large.js +11299 -0
- package/dist/standalone/pdf-large.meta.json +57 -0
- package/dist/standalone/pdf-large.min.js +35 -0
- package/dist/standalone/pdf-legacy-rw.js +18228 -0
- package/dist/standalone/pdf-legacy-rw.meta.json +83 -0
- package/dist/standalone/pdf-legacy-rw.min.js +56 -0
- package/dist/standalone/pdf-legacy.js +13334 -0
- package/dist/standalone/pdf-legacy.meta.json +68 -0
- package/dist/standalone/pdf-legacy.min.js +35 -0
- package/dist/standalone/pdf-rw.js +13404 -0
- package/dist/standalone/pdf-rw.meta.json +50 -0
- package/dist/standalone/pdf-rw.min.js +56 -0
- package/dist/standalone/pdf.js +8510 -0
- package/dist/standalone/pdf.meta.json +35 -0
- package/dist/standalone/pdf.min.js +35 -0
- package/docs/README.md +53 -0
- package/docs/api/README.md +38 -0
- package/docs/api/_shared/README.md +91 -0
- package/docs/api/action/README.md +29 -0
- package/docs/api/action/action.md +81 -0
- package/docs/api/action/goTo.md +66 -0
- package/docs/api/action/launch.md +58 -0
- package/docs/api/action/named.md +55 -0
- package/docs/api/action/uri.md +54 -0
- package/docs/api/annot/README.md +53 -0
- package/docs/api/annot/annot.md +114 -0
- package/docs/api/annot/fileAttach.md +53 -0
- package/docs/api/annot/freeText.md +68 -0
- package/docs/api/annot/ink.md +69 -0
- package/docs/api/annot/link.md +74 -0
- package/docs/api/annot/markup.md +83 -0
- package/docs/api/annot/popup.md +52 -0
- package/docs/api/annot/projection.md +56 -0
- package/docs/api/annot/redact.md +67 -0
- package/docs/api/annot/square.md +87 -0
- package/docs/api/annot/stamp.md +54 -0
- package/docs/api/annot/text.md +69 -0
- package/docs/api/annot/widget.md +69 -0
- package/docs/api/associatedFiles/README.md +9 -0
- package/docs/api/associatedFiles/associatedFiles.md +78 -0
- package/docs/api/bundles/README.md +68 -0
- package/docs/api/bundles/dist-matrix.md +165 -0
- package/docs/api/bundles/pdf-full.md +148 -0
- package/docs/api/bundles/pdf-large.md +144 -0
- package/docs/api/bundles/pdf-legacy.md +169 -0
- package/docs/api/content/README.md +29 -0
- package/docs/api/content/color.md +99 -0
- package/docs/api/content/graphics.md +114 -0
- package/docs/api/content/images.md +124 -0
- package/docs/api/content/ops.md +100 -0
- package/docs/api/content/stream.md +107 -0
- package/docs/api/content/text.md +98 -0
- package/docs/api/crypto/README.md +29 -0
- package/docs/api/crypto/aesGcm.md +72 -0
- package/docs/api/crypto/permissions.md +79 -0
- package/docs/api/crypto/security.md +98 -0
- package/docs/api/crypto/standardV4.md +104 -0
- package/docs/api/crypto/standardV5.md +84 -0
- package/docs/api/crypto/standardV6.md +93 -0
- package/docs/api/destination/README.md +9 -0
- package/docs/api/destination/destination.md +79 -0
- package/docs/api/document/README.md +29 -0
- package/docs/api/document/builder.md +281 -0
- package/docs/api/document/catalog.md +98 -0
- package/docs/api/document/document.md +187 -0
- package/docs/api/document/encryptedWriter.md +149 -0
- package/docs/api/document/incrementalWriter.md +148 -0
- package/docs/api/document/page.md +99 -0
- package/docs/api/document/pages.md +82 -0
- package/docs/api/document/resources.md +102 -0
- package/docs/api/document/writer.md +157 -0
- package/docs/api/document/xrefStreamWriter.md +122 -0
- package/docs/api/embedded/README.md +13 -0
- package/docs/api/embedded/collection.md +80 -0
- package/docs/api/embedded/embeddedFile.md +86 -0
- package/docs/api/embedded/fileSpec.md +87 -0
- package/docs/api/errors.md +110 -0
- package/docs/api/extra/3d-richmedia.md +76 -0
- package/docs/api/extra/README.md +99 -0
- package/docs/api/extra/annot-extended.md +71 -0
- package/docs/api/extra/associated-files.md +70 -0
- package/docs/api/extra/ccitt-fax-decoder.md +74 -0
- package/docs/api/extra/color-spaces-extended.md +72 -0
- package/docs/api/extra/content-ops-extended.md +82 -0
- package/docs/api/extra/document-parts.md +69 -0
- package/docs/api/extra/embedded-files-portfolio.md +87 -0
- package/docs/api/extra/font-cid-typed.md +77 -0
- package/docs/api/extra/font-color-tagging.md +76 -0
- package/docs/api/extra/form-actions-extended.md +75 -0
- package/docs/api/extra/info-dict-deprecated.md +72 -0
- package/docs/api/extra/jbig2-read.md +80 -0
- package/docs/api/extra/legacy-deprecated-annots.md +89 -0
- package/docs/api/extra/legacy-deprecated-filters.md +78 -0
- package/docs/api/extra/legacy-rc4-read.md +74 -0
- package/docs/api/extra/legacy-xfa-read.md +65 -0
- package/docs/api/extra/linearization-write.md +71 -0
- package/docs/api/extra/misc.md +93 -0
- package/docs/api/extra/optional-content-extended.md +83 -0
- package/docs/api/extra/pdf-a-output-intent.md +65 -0
- package/docs/api/extra/pdf-sandbox.md +76 -0
- package/docs/api/extra/pdf-ua-tagged.md +63 -0
- package/docs/api/extra/pdf-x-prepress.md +65 -0
- package/docs/api/extra/redaction-iso32005.md +65 -0
- package/docs/api/extra/shading-typed.md +73 -0
- package/docs/api/extra/sig-aes-gcm.md +69 -0
- package/docs/api/extra/sig-pades.md +103 -0
- package/docs/api/extra/tagged-pdf-typed.md +78 -0
- package/docs/api/extra/transparency-typed.md +74 -0
- package/docs/api/extra/well-tagged-pdf.md +61 -0
- package/docs/api/extra/xmp-extended.md +65 -0
- package/docs/api/font/README.md +25 -0
- package/docs/api/font/embed.md +157 -0
- package/docs/api/font/encoding.md +95 -0
- package/docs/api/font/font.md +97 -0
- package/docs/api/font/type3.md +89 -0
- package/docs/api/form/README.md +35 -0
- package/docs/api/form/acroform.md +88 -0
- package/docs/api/form/appearance.md +87 -0
- package/docs/api/form/button.md +97 -0
- package/docs/api/form/choice.md +96 -0
- package/docs/api/form/fieldTree.md +93 -0
- package/docs/api/form/signature.md +90 -0
- package/docs/api/form/text.md +88 -0
- package/docs/api/linearization/README.md +11 -0
- package/docs/api/linearization/linearization.md +81 -0
- package/docs/api/main.md +116 -0
- package/docs/api/metadata/README.md +10 -0
- package/docs/api/metadata/info.md +70 -0
- package/docs/api/metadata/xmp.md +62 -0
- package/docs/api/ocg/README.md +23 -0
- package/docs/api/ocg/config.md +95 -0
- package/docs/api/ocg/ocg.md +77 -0
- package/docs/api/outline/README.md +11 -0
- package/docs/api/outline/outline.md +107 -0
- package/docs/api/pdf.md +152 -0
- package/docs/api/prepress/README.md +10 -0
- package/docs/api/prepress/outputIntent.md +79 -0
- package/docs/api/prepress/pageBoundary.md +75 -0
- package/docs/api/sig/README.md +32 -0
- package/docs/api/sig/byteRange.md +120 -0
- package/docs/api/sig/certChain.md +84 -0
- package/docs/api/sig/dss.md +111 -0
- package/docs/api/sig/oids.md +76 -0
- package/docs/api/sig/sha1.md +72 -0
- package/docs/api/sig/sign.md +317 -0
- package/docs/api/sig/signature.md +178 -0
- package/docs/api/sig/timestamp.md +84 -0
- package/docs/api/syntax/README.md +29 -0
- package/docs/api/syntax/crossRefStream.md +115 -0
- package/docs/api/syntax/filters/README.md +50 -0
- package/docs/api/syntax/filters/ascii85.md +76 -0
- package/docs/api/syntax/filters/asciiHex.md +73 -0
- package/docs/api/syntax/filters/dispatch.md +125 -0
- package/docs/api/syntax/filters/flate.md +134 -0
- package/docs/api/syntax/filters/runLength.md +78 -0
- package/docs/api/syntax/objStream.md +88 -0
- package/docs/api/syntax/parser-obj.md +97 -0
- package/docs/api/syntax/parser.md +151 -0
- package/docs/api/syntax/serializer.md +109 -0
- package/docs/api/syntax/tokenizer.md +104 -0
- package/docs/api/syntax/trailer.md +85 -0
- package/docs/api/syntax/xref.md +139 -0
- package/docs/api/tagged/README.md +25 -0
- package/docs/api/tagged/classMap.md +67 -0
- package/docs/api/tagged/markedContent.md +62 -0
- package/docs/api/tagged/parentTree.md +67 -0
- package/docs/api/tagged/roleMap.md +67 -0
- package/docs/api/tagged/structElement.md +76 -0
- package/docs/api/tagged/structTree.md +75 -0
- package/docs/guide/coverage.md +113 -0
- package/docs/guide/crypto.md +121 -0
- package/docs/guide/extending.md +76 -0
- package/docs/guide/getting-started.md +75 -0
- package/docs/guide/legacy-1.7.md +42 -0
- package/docs/guide/pades-integration.md +579 -0
- package/docs/guide/read-pdf.md +89 -0
- package/package.json +97 -4
- package/src/_shared/index.js +179 -0
- package/src/action/action.js +119 -0
- package/src/action/goTo.js +89 -0
- package/src/action/launch.js +61 -0
- package/src/action/named.js +54 -0
- package/src/action/uri.js +51 -0
- package/src/annot/annot.js +212 -0
- package/src/annot/fileAttach.js +55 -0
- package/src/annot/freeText.js +82 -0
- package/src/annot/ink.js +77 -0
- package/src/annot/link.js +77 -0
- package/src/annot/markup.js +91 -0
- package/src/annot/popup.js +53 -0
- package/src/annot/projection.js +52 -0
- package/src/annot/redact.js +87 -0
- package/src/annot/square.js +132 -0
- package/src/annot/stamp.js +48 -0
- package/src/annot/text.js +54 -0
- package/src/annot/widget.js +61 -0
- package/src/associatedFiles/associatedFiles.js +86 -0
- package/src/bundles/pdf-full.js +91 -0
- package/src/bundles/pdf-large.js +81 -0
- package/src/bundles/pdf-legacy.js +107 -0
- package/src/content/color.js +114 -0
- package/src/content/graphics.js +192 -0
- package/src/content/images.js +160 -0
- package/src/content/ops.js +137 -0
- package/src/content/stream.js +154 -0
- package/src/content/text.js +125 -0
- package/src/crypto/aesGcm.js +123 -0
- package/src/crypto/permissions.js +112 -0
- package/src/crypto/security.js +327 -0
- package/src/crypto/standardV4.js +443 -0
- package/src/crypto/standardV5.js +306 -0
- package/src/crypto/standardV6.js +334 -0
- package/src/destination/destination.js +183 -0
- package/src/document/builder.js +618 -0
- package/src/document/catalog.js +100 -0
- package/src/document/document.js +472 -0
- package/src/document/encryptedWriter.js +554 -0
- package/src/document/incrementalWriter.js +514 -0
- package/src/document/page.js +131 -0
- package/src/document/pages.js +103 -0
- package/src/document/resources.js +146 -0
- package/src/document/writer.js +211 -0
- package/src/document/xrefStreamWriter.js +353 -0
- package/src/embedded/collection.js +102 -0
- package/src/embedded/embeddedFile.js +99 -0
- package/src/embedded/fileSpec.js +137 -0
- package/src/errors.js +78 -0
- package/src/extra/3d-richmedia.js +171 -0
- package/src/extra/annot-extended.js +200 -0
- package/src/extra/associated-files.js +131 -0
- package/src/extra/ccitt-fax-decoder.js +776 -0
- package/src/extra/color-spaces-extended.js +196 -0
- package/src/extra/content-ops-extended.js +153 -0
- package/src/extra/document-parts.js +149 -0
- package/src/extra/embedded-files-portfolio.js +234 -0
- package/src/extra/font-cid-typed.js +185 -0
- package/src/extra/font-color-tagging.js +144 -0
- package/src/extra/form-actions-extended.js +196 -0
- package/src/extra/info-dict-deprecated.js +137 -0
- package/src/extra/jbig2-read.js +169 -0
- package/src/extra/legacy-deprecated-annots.js +198 -0
- package/src/extra/legacy-deprecated-filters.js +167 -0
- package/src/extra/legacy-rc4-read.js +235 -0
- package/src/extra/legacy-xfa-read.js +104 -0
- package/src/extra/linearization-write.js +97 -0
- package/src/extra/misc.js +217 -0
- package/src/extra/optional-content-extended.js +142 -0
- package/src/extra/pdf-a-output-intent.js +112 -0
- package/src/extra/pdf-sandbox.js +88 -0
- package/src/extra/pdf-ua-tagged.js +116 -0
- package/src/extra/pdf-x-prepress.js +114 -0
- package/src/extra/redaction-iso32005.js +136 -0
- package/src/extra/shading-typed.js +222 -0
- package/src/extra/sig-aes-gcm.js +135 -0
- package/src/extra/sig-pades.js +242 -0
- package/src/extra/tagged-pdf-typed.js +203 -0
- package/src/extra/transparency-typed.js +135 -0
- package/src/extra/well-tagged-pdf.js +138 -0
- package/src/extra/xmp-extended.js +190 -0
- package/src/font/embed.js +480 -0
- package/src/font/encoding.js +92 -0
- package/src/font/font.js +101 -0
- package/src/font/type3.js +75 -0
- package/src/form/acroform.js +94 -0
- package/src/form/appearance.js +90 -0
- package/src/form/button.js +105 -0
- package/src/form/choice.js +152 -0
- package/src/form/fieldTree.js +120 -0
- package/src/form/signature.js +100 -0
- package/src/form/text.js +101 -0
- package/src/linearization/linearization.js +107 -0
- package/src/main.js +411 -0
- package/src/metadata/info.js +87 -0
- package/src/metadata/xmp.js +62 -0
- package/src/ocg/config.js +156 -0
- package/src/ocg/ocg.js +124 -0
- package/src/outline/outline.js +157 -0
- package/src/pdf.js +133 -0
- package/src/prepress/outputIntent.js +118 -0
- package/src/prepress/pageBoundary.js +108 -0
- package/src/sig/byteRange.js +306 -0
- package/src/sig/certChain.js +247 -0
- package/src/sig/dss.js +317 -0
- package/src/sig/oids.js +157 -0
- package/src/sig/sha1.js +142 -0
- package/src/sig/sign.js +1899 -0
- package/src/sig/signature.js +1441 -0
- package/src/sig/timestamp.js +236 -0
- package/src/syntax/crossRefStream.js +133 -0
- package/src/syntax/filters/ascii85.js +122 -0
- package/src/syntax/filters/asciiHex.js +83 -0
- package/src/syntax/filters/dispatch.js +176 -0
- package/src/syntax/filters/flate.js +316 -0
- package/src/syntax/filters/runLength.js +96 -0
- package/src/syntax/objStream.js +99 -0
- package/src/syntax/parser-obj.js +52 -0
- package/src/syntax/parser.js +321 -0
- package/src/syntax/serializer.js +221 -0
- package/src/syntax/tokenizer.js +290 -0
- package/src/syntax/trailer.js +76 -0
- package/src/syntax/xref.js +341 -0
- package/src/tagged/classMap.js +81 -0
- package/src/tagged/markedContent.js +123 -0
- package/src/tagged/parentTree.js +126 -0
- package/src/tagged/roleMap.js +107 -0
- package/src/tagged/structElement.js +138 -0
- package/src/tagged/structTree.js +94 -0
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfObjStream
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: [pdfErrors, pdfParserObj, pdfTokenizer, pdfParser]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfObjStream
|
|
11
|
+
|
|
12
|
+
> Object-stream parser `/Type /ObjStm` — ISO 32000-2 §7.5.7.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfObjStream` | **Source** `packages/front/office/pdf/src/syntax/objStream.js` | **Deps** `pdfErrors`, `pdfParserObj`, `pdfTokenizer`, `pdfParser` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
An object stream is a compressed container holding several non-stream indirect
|
|
17
|
+
objects. Its dictionary carries `/N` (object count), `/First` (offset of the
|
|
18
|
+
first body inside the decoded payload) and, optionally, `/Extends`. The payload
|
|
19
|
+
starts with `N` pairs `<num> <byteOffset>` (offsets relative to `/First`),
|
|
20
|
+
followed by the bodies concatenated without `obj`/`endobj`. The generation is
|
|
21
|
+
always `0` by spec §7.5.8.4 (compressed objects). The payload must already be
|
|
22
|
+
decoded: [`pdfDocument`](../document/document.md) composes this module and runs
|
|
23
|
+
the container through [`pdfFilterDispatch`](./filters/dispatch.md) — once per
|
|
24
|
+
container per document — so `doc._raw.resolve(ref)` returns a compressed object
|
|
25
|
+
like any other indirect. Callers driving the parser on its own apply `/Filter`
|
|
26
|
+
(typically Flate) themselves.
|
|
27
|
+
|
|
28
|
+
## Resolve
|
|
29
|
+
|
|
30
|
+
```js
|
|
31
|
+
const objStm = runtime.resolve('pdfObjStream');
|
|
32
|
+
// Returns: { parseObjectStream }
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## API
|
|
36
|
+
|
|
37
|
+
| Method | Signature | Returns |
|
|
38
|
+
|--------|-----------|---------|
|
|
39
|
+
| `parseObjectStream` | `(decoded: Uint8Array, dict: PdfDict) => Array<{num, gen:0, value}>` | Every member object. |
|
|
40
|
+
|
|
41
|
+
### `parseObjectStream`
|
|
42
|
+
|
|
43
|
+
1. Validates `/Type /ObjStm` when present.
|
|
44
|
+
2. Reads `/N` and `/First` (required, integers ≥ 0).
|
|
45
|
+
3. Tokenises `[0, First[` to recover the `(num, offset)` pairs.
|
|
46
|
+
4. Slices each body `[First+off[i], First+off[i+1][` (or to the end for the
|
|
47
|
+
last one) and feeds it to the standard `parseObject`.
|
|
48
|
+
|
|
49
|
+
Out-of-order or out-of-payload offsets raise `pdf/objstm/bad-offset`.
|
|
50
|
+
|
|
51
|
+
## Examples
|
|
52
|
+
|
|
53
|
+
### Parse a freshly inflated ObjStm
|
|
54
|
+
|
|
55
|
+
```js
|
|
56
|
+
const objStm = runtime.resolve('pdfObjStream');
|
|
57
|
+
const dict = streamObj.dict; // { type: 'dict', entries: { Type, N, First, … } }
|
|
58
|
+
const decoded = filterDispatch.decode(streamObj);
|
|
59
|
+
const members = objStm.parseObjectStream(decoded, dict);
|
|
60
|
+
// → [{ num: 5, gen: 0, value: { type: 'dict', entries: { Type: …, Count: … } } }, …]
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
### Feed the members back into a resolver
|
|
64
|
+
|
|
65
|
+
```js
|
|
66
|
+
for (const { num, value } of members) {
|
|
67
|
+
indirects.set(`${num}:0`, { value, offset: -1 }); // -1 → "compressed"
|
|
68
|
+
}
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Errors
|
|
72
|
+
|
|
73
|
+
| Code | Class | When |
|
|
74
|
+
|------|-------|------|
|
|
75
|
+
| `pdf/objstm/bad-input` | `ParseError` | `decoded` is not a `Uint8Array`. |
|
|
76
|
+
| `pdf/objstm/bad-dict` | `ParseError` | `dict` is not a typed dictionary. |
|
|
77
|
+
| `pdf/objstm/wrong-type` | `ParseError` | `/Type` present but not `/ObjStm`. |
|
|
78
|
+
| `pdf/objstm/missing-int` | `ParseError` | `/N` or `/First` missing or not an int. |
|
|
79
|
+
| `pdf/objstm/bad-N` | `ParseError` | `/N` negative. |
|
|
80
|
+
| `pdf/objstm/bad-First` | `ParseError` | `/First` outside the payload range. |
|
|
81
|
+
| `pdf/objstm/bad-pair` | `ParseError` | Malformed `(num, off)` header pair. |
|
|
82
|
+
| `pdf/objstm/bad-offset` | `ParseError` | Member offset out of range. |
|
|
83
|
+
|
|
84
|
+
## See also
|
|
85
|
+
|
|
86
|
+
- [`pdfCrossRefStream`](./crossRefStream.md) — points at ObjStm members (type 2).
|
|
87
|
+
- [`pdfParser`](./parser.md) — consumed for each body.
|
|
88
|
+
- [`pdfFilterDispatch`](./filters/dispatch.md) — mandatory `/Filter` decode beforehand.
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfParserObj
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: []
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfParserObj
|
|
11
|
+
|
|
12
|
+
> Typed constructors and reflection helpers for PDF objects — split out of `parser.js`.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfParserObj` | **Source** `packages/front/office/pdf/src/syntax/parser-obj.js` | **Deps** none | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Companion file to [`parser.js`](./parser.md), extracted to keep the orchestrator
|
|
17
|
+
under 300 LOC. The factory declares no dependencies and returns three members:
|
|
18
|
+
|
|
19
|
+
- `obj` — the constructor factory (`obj.nul`, `obj.bool`, `obj.int`, `obj.real`,
|
|
20
|
+
`obj.name`, `obj.string`, `obj.array`, `obj.dict`, `obj.ref`, `obj.stream`).
|
|
21
|
+
- `getEntry(dict, key)` — safe `dict.entries[key]` access, `undefined` when
|
|
22
|
+
`dict` is not a typed dictionary.
|
|
23
|
+
- `isType(v, kind)` — discriminant test `v && v.type === kind`.
|
|
24
|
+
|
|
25
|
+
Every builder is **pure** and always returns a fresh object — no cache, no pool.
|
|
26
|
+
`pdfParser` re-exposes all three members for convenience.
|
|
27
|
+
|
|
28
|
+
## Resolve
|
|
29
|
+
|
|
30
|
+
```js
|
|
31
|
+
const po = runtime.resolve('pdfParserObj');
|
|
32
|
+
// Returns: { obj, getEntry, isType }
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
`@awacloud/pdf` re-exports module *descriptors* only (`pdfParserObj`), never resolved
|
|
36
|
+
instances — always go through `runtime.resolve`.
|
|
37
|
+
|
|
38
|
+
## API
|
|
39
|
+
|
|
40
|
+
| Member | Signature | Returns |
|
|
41
|
+
|--------|-----------|---------|
|
|
42
|
+
| `obj.nul` | `() => object` | `{ type: 'null' }` |
|
|
43
|
+
| `obj.bool` | `(v: any) => object` | `{ type: 'bool', value: !!v }` |
|
|
44
|
+
| `obj.int` | `(v: number) => object` | `{ type: 'int', value: v\|0 }` |
|
|
45
|
+
| `obj.real` | `(v: number) => object` | `{ type: 'real', value: +v }` |
|
|
46
|
+
| `obj.name` | `(s: string) => object` | `{ type: 'name', value: String(s) }` |
|
|
47
|
+
| `obj.string` | `(bytes: Uint8Array, syntax?: 'lit'\|'hex') => object` | `{ type: 'string', value, syntax }` |
|
|
48
|
+
| `obj.array` | `(items?: object[]) => object` | `{ type: 'array', items }` |
|
|
49
|
+
| `obj.dict` | `(entries?: object) => object` | `{ type: 'dict', entries }` |
|
|
50
|
+
| `obj.ref` | `(num: number, gen?: number) => object` | `{ type: 'ref', num, gen }` |
|
|
51
|
+
| `obj.stream` | `(dict: object, raw: Uint8Array) => object` | `{ type: 'stream', dict, raw }` |
|
|
52
|
+
| `getEntry` | `(dict, key: string) => object\|undefined` | value or `undefined` |
|
|
53
|
+
| `isType` | `(v, kind: string) => boolean` | `v?.type === kind` |
|
|
54
|
+
|
|
55
|
+
## Examples
|
|
56
|
+
|
|
57
|
+
### Build a literal Catalog
|
|
58
|
+
|
|
59
|
+
```js
|
|
60
|
+
const { obj } = runtime.resolve('pdfParserObj');
|
|
61
|
+
|
|
62
|
+
const catalog = obj.dict({
|
|
63
|
+
Type: obj.name('Catalog'),
|
|
64
|
+
Pages: obj.ref(2, 0),
|
|
65
|
+
Version: obj.name('2.0')
|
|
66
|
+
});
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### Build a stream
|
|
70
|
+
|
|
71
|
+
```js
|
|
72
|
+
const { obj } = runtime.resolve('pdfParserObj');
|
|
73
|
+
const data = new TextEncoder().encode('BT /F1 12 Tf (Hi) Tj ET');
|
|
74
|
+
const stream = obj.stream(
|
|
75
|
+
obj.dict({ Length: obj.int(data.length) }),
|
|
76
|
+
data
|
|
77
|
+
);
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### Typed lookup
|
|
81
|
+
|
|
82
|
+
```js
|
|
83
|
+
const { getEntry, isType } = runtime.resolve('pdfParserObj');
|
|
84
|
+
|
|
85
|
+
if (isType(catalog, 'dict')) {
|
|
86
|
+
const pages = getEntry(catalog, 'Pages');
|
|
87
|
+
if (isType(pages, 'ref')) console.log(pages.num);
|
|
88
|
+
}
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
## Errors
|
|
92
|
+
|
|
93
|
+
Pure builders and predicates — this module raises no `PdfError`.
|
|
94
|
+
|
|
95
|
+
## See also
|
|
96
|
+
|
|
97
|
+
- [`pdfParser`](./parser.md) — orchestrator that consumes these helpers
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfParser
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: [pdfErrors, pdfParserObj, pdfTokenizer]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfParser
|
|
11
|
+
|
|
12
|
+
> Typed PDF objects, ISO 32000-2 §7.3 — token stream → `{ type, value/items/entries }` tree.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfParser` | **Source** `packages/front/office/pdf/src/syntax/parser.js` | **Deps** `pdfErrors`, `pdfParserObj`, `pdfTokenizer` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Produces the typed objects `null`, `bool`, `int`, `real`, `name`, `string`,
|
|
17
|
+
`array`, `dict`, `ref`, `stream`. The `<int> <int> R` form is recognised
|
|
18
|
+
speculatively as a `ref`. Indirect definitions `<num> <gen> obj … endobj` are
|
|
19
|
+
read by `parseIndirect`, which promotes `dict + stream` into
|
|
20
|
+
`{ type: 'stream', dict, raw }`. The parser validates neither xref integrity nor
|
|
21
|
+
`/Length` — that is the document layer's job.
|
|
22
|
+
|
|
23
|
+
The module re-exposes the injected `pdfTokenizer`'s `tokenize` and the whole
|
|
24
|
+
`pdfParserObj` surface (`obj`, `getEntry`, `isType`) so a consumer needs a
|
|
25
|
+
single resolve to go from bytes to a typed tree.
|
|
26
|
+
|
|
27
|
+
## Resolve
|
|
28
|
+
|
|
29
|
+
```js
|
|
30
|
+
const parser = runtime.resolve('pdfParser');
|
|
31
|
+
// Returns: { tokenize, parseObject, parseIndirect, parseFromBytes,
|
|
32
|
+
// parseIndirectFromBytes, parserLimits, setParserLimits,
|
|
33
|
+
// obj, getEntry, isType }
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## API
|
|
37
|
+
|
|
38
|
+
| Member | Signature | Returns |
|
|
39
|
+
|--------|-----------|---------|
|
|
40
|
+
| `tokenize` | `(bytes: Uint8Array, opts?) => Tokenizer` | Re-export of [`pdfTokenizer.tokenize`](./tokenizer.md). |
|
|
41
|
+
| `parseObject` | `(tok: Tokenizer) => PdfObject` | One typed object. |
|
|
42
|
+
| `parseIndirect` | `(tok: Tokenizer, resolveRef?) => { num, gen, value }` | Full indirect definition. |
|
|
43
|
+
| `parseFromBytes` | `(bytes: Uint8Array) => PdfObject` | Helper — tokenize then `parseObject`. |
|
|
44
|
+
| `parseIndirectFromBytes` | `(bytes: Uint8Array, at: number, resolveRef?) => { num, gen, value }` | Offset helper. |
|
|
45
|
+
| `parserLimits` | `{ maxDepth, maxArrayLen, maxStreamBytes }` | Live, mutable per-instance guard-rails. |
|
|
46
|
+
| `setParserLimits` | `(partial) => parserLimits` | Merges positive integer overrides and returns the updated object. |
|
|
47
|
+
| `obj` | builders | Primitive constructors (see below). |
|
|
48
|
+
| `getEntry` | `(dict, key: string) => PdfObject \| undefined` | Lookup on `{type:'dict'}`. |
|
|
49
|
+
| `isType` | `(v, kind: string) => boolean` | Test `v.type === kind`. |
|
|
50
|
+
|
|
51
|
+
### `parserLimits` defaults
|
|
52
|
+
|
|
53
|
+
| Key | Default | Guards |
|
|
54
|
+
|-----|---------|--------|
|
|
55
|
+
| `maxDepth` | `200` | Nesting depth — `pdf/parser/depth-exceeded`. |
|
|
56
|
+
| `maxArrayLen` | `1_000_000` | Array element count — `pdf/parser/array-too-long`. |
|
|
57
|
+
| `maxStreamBytes` | `268_435_456` (256 MiB) | Stream payload quota during `parseIndirect`. |
|
|
58
|
+
|
|
59
|
+
Only positive integers are accepted; anything else is ignored, so
|
|
60
|
+
`setParserLimits({})` is a no-op that simply returns the current limits.
|
|
61
|
+
|
|
62
|
+
### `obj` builders
|
|
63
|
+
|
|
64
|
+
| Helper | Signature |
|
|
65
|
+
|--------|-----------|
|
|
66
|
+
| `obj.nul()` | `→ { type: 'null' }` |
|
|
67
|
+
| `obj.bool(v)` | `→ { type: 'bool', value }` |
|
|
68
|
+
| `obj.int(v)` | `→ { type: 'int', value }` |
|
|
69
|
+
| `obj.real(v)` | `→ { type: 'real', value }` |
|
|
70
|
+
| `obj.name(s)` | `→ { type: 'name', value }` |
|
|
71
|
+
| `obj.string(bytes, syntax?)` | `syntax ∈ 'lit'\|'hex'` |
|
|
72
|
+
| `obj.array(items?)` | |
|
|
73
|
+
| `obj.dict(entries?)` | |
|
|
74
|
+
| `obj.ref(num, gen?)` | |
|
|
75
|
+
| `obj.stream(dict, raw)` | |
|
|
76
|
+
|
|
77
|
+
## Token → type mapping
|
|
78
|
+
|
|
79
|
+
| Token | Output type |
|
|
80
|
+
|-------|-------------|
|
|
81
|
+
| `kw 'null'/'true'/'false'` | `null` / `bool` |
|
|
82
|
+
| `name` | `name` |
|
|
83
|
+
| `string` | `string` (`syntax:'lit'`) |
|
|
84
|
+
| `hex` | `string` (`syntax:'hex'`) |
|
|
85
|
+
| `int` + `int` + `kw 'R'` | `ref` |
|
|
86
|
+
| `int` | `int` |
|
|
87
|
+
| `real` | `real` |
|
|
88
|
+
| `open_arr … close_arr` | `array` |
|
|
89
|
+
| `open_dict … close_dict` | `dict` |
|
|
90
|
+
| `dict` + `kw 'stream'` (via `parseIndirect`) | `stream` |
|
|
91
|
+
|
|
92
|
+
## Examples
|
|
93
|
+
|
|
94
|
+
### Parse a standalone object
|
|
95
|
+
|
|
96
|
+
```js
|
|
97
|
+
const parser = runtime.resolve('pdfParser');
|
|
98
|
+
const o = parser.parseFromBytes(new TextEncoder().encode('<< /Size 6 /Root 1 0 R >>'));
|
|
99
|
+
o.type; // 'dict'
|
|
100
|
+
parser.getEntry(o, 'Size').value; // 6
|
|
101
|
+
parser.getEntry(o, 'Root'); // { type: 'ref', num: 1, gen: 0 }
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
### Read an indirect definition carrying a stream
|
|
105
|
+
|
|
106
|
+
```js
|
|
107
|
+
const tok = parser.tokenize(bytes, { start: offset });
|
|
108
|
+
const def = parser.parseIndirect(tok, ref => doc._raw.resolve(ref));
|
|
109
|
+
def.value.type; // 'stream'
|
|
110
|
+
def.value.raw; // Uint8Array
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Tighten the limits for untrusted input
|
|
114
|
+
|
|
115
|
+
```js
|
|
116
|
+
parser.setParserLimits({ maxDepth: 32, maxStreamBytes: 8 * 1024 * 1024 });
|
|
117
|
+
parser.parserLimits.maxDepth; // 32
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
### Build with `obj`
|
|
121
|
+
|
|
122
|
+
```js
|
|
123
|
+
const o = parser.obj.dict({
|
|
124
|
+
Type: parser.obj.name('Catalog'),
|
|
125
|
+
Pages: parser.obj.ref(2, 0)
|
|
126
|
+
});
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## Errors
|
|
130
|
+
|
|
131
|
+
| Code | Class | When |
|
|
132
|
+
|------|-------|------|
|
|
133
|
+
| `pdf/parser/eof` | `ParseError` | End of stream during `parseObject`. |
|
|
134
|
+
| `pdf/parser/unexpected-keyword` | `ParseError` | Unrecognised keyword in object position. |
|
|
135
|
+
| `pdf/parser/unexpected-token` | `ParseError` | Foreign token (e.g. a lone `n`). |
|
|
136
|
+
| `pdf/parser/unbalanced` | `ParseError` | `]` or `>>` without an opener. |
|
|
137
|
+
| `pdf/parser/unterminated-array` | `ParseError` | `[` never closed. |
|
|
138
|
+
| `pdf/parser/unterminated-dict` | `ParseError` | `<<` never closed. |
|
|
139
|
+
| `pdf/parser/dict-key-not-name` | `ParseError` | Dictionary key is not a name. |
|
|
140
|
+
| `pdf/parser/depth-exceeded` | `ParseError` | Nesting deeper than `parserLimits.maxDepth`. |
|
|
141
|
+
| `pdf/parser/array-too-long` | `ParseError` | Array longer than `parserLimits.maxArrayLen`. |
|
|
142
|
+
| `pdf/parser/indirect-bad-num` / `-bad-gen` / `-missing-obj` / `-missing-endobj` | `ParseError` | Malformed `<n> <g> obj … endobj`. |
|
|
143
|
+
| `pdf/parser/stream/no-endstream` | `ParseError` | `endstream` not found. |
|
|
144
|
+
| `pdf/parser/stream/expected-endstream` | `ParseError` | Something else found instead. |
|
|
145
|
+
|
|
146
|
+
## See also
|
|
147
|
+
|
|
148
|
+
- [`pdfTokenizer`](./tokenizer.md)
|
|
149
|
+
- [`pdfParserObj`](./parser-obj.md)
|
|
150
|
+
- [`pdfXref`](./xref.md) — uses `parseObject` for the trailer.
|
|
151
|
+
- [`pdfErrors`](../errors.md)
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfSerializer
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: [pdfErrors]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfSerializer
|
|
11
|
+
|
|
12
|
+
> Typed PDF objects → `Uint8Array` — canonical emission, ISO 32000-2 §7.3.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfSerializer` | **Source** `packages/front/office/pdf/src/syntax/serializer.js` | **Deps** `pdfErrors` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Strict inverse of [`pdfParser`](./parser.md). Emits any typed object (`null`,
|
|
17
|
+
`bool`, `int`, `real`, `name`, `string`, `array`, `dict`, `ref`) as bytes.
|
|
18
|
+
`stream` objects are not emitted directly — go through `serializeIndirect`,
|
|
19
|
+
which rewrites `/Length` to the exact raw byte count. `real` values follow
|
|
20
|
+
§7.3.3 (fixed point, no exponent notation, trailing zeros stripped). `name`
|
|
21
|
+
values are `#xx`-escaped for every delimiter, space, or byte outside `!`..`~`.
|
|
22
|
+
`string` values automatically pick the literal `(…)` or hex `<…>` form based on
|
|
23
|
+
the density of non-printable bytes, unless `syntax: 'hex'` is explicit.
|
|
24
|
+
|
|
25
|
+
## Resolve
|
|
26
|
+
|
|
27
|
+
```js
|
|
28
|
+
const ser = runtime.resolve('pdfSerializer');
|
|
29
|
+
// Returns: { serializeObject, serializeIndirect, formatReal }
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
## API
|
|
33
|
+
|
|
34
|
+
| Method | Signature | Returns |
|
|
35
|
+
|--------|-----------|---------|
|
|
36
|
+
| `serializeObject` | `(obj: PdfObject) => Uint8Array` | Bytes of a standalone object. |
|
|
37
|
+
| `serializeIndirect` | `(num: number, gen: number, body: PdfObject) => Uint8Array` | `<n> <g> obj … endobj\n`. |
|
|
38
|
+
| `formatReal` | `(n: number) => string` | Canonical real per §7.3.3. |
|
|
39
|
+
|
|
40
|
+
### `serializeIndirect`
|
|
41
|
+
|
|
42
|
+
For a `stream` body, the dictionary is cloned with `/Length` forced to
|
|
43
|
+
`body.raw.length`. The result follows
|
|
44
|
+
`<n> <g> obj\n<<…>>\nstream\n<raw>\nendstream\nendobj\n`. Any other type is
|
|
45
|
+
serialised inline between `obj` and `endobj`.
|
|
46
|
+
|
|
47
|
+
## Examples
|
|
48
|
+
|
|
49
|
+
### Emit a dictionary
|
|
50
|
+
|
|
51
|
+
```js
|
|
52
|
+
const ser = runtime.resolve('pdfSerializer');
|
|
53
|
+
const dict = {
|
|
54
|
+
type: 'dict',
|
|
55
|
+
entries: {
|
|
56
|
+
Type: { type: 'name', value: 'Catalog' },
|
|
57
|
+
Pages: { type: 'ref', num: 2, gen: 0 }
|
|
58
|
+
}
|
|
59
|
+
};
|
|
60
|
+
const bytes = ser.serializeObject(dict);
|
|
61
|
+
// → "<< /Type /Catalog /Pages 2 0 R >>"
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### Emit an indirect definition
|
|
65
|
+
|
|
66
|
+
```js
|
|
67
|
+
const def = ser.serializeIndirect(1, 0, dict);
|
|
68
|
+
// → "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n"
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
### Emit a stream — `/Length` recomputed
|
|
72
|
+
|
|
73
|
+
```js
|
|
74
|
+
const raw = new TextEncoder().encode('BT /F1 12 Tf (Hi) Tj ET');
|
|
75
|
+
const stream = {
|
|
76
|
+
type: 'stream',
|
|
77
|
+
dict: { type: 'dict', entries: {} },
|
|
78
|
+
raw
|
|
79
|
+
};
|
|
80
|
+
const bytes = ser.serializeIndirect(4, 0, stream);
|
|
81
|
+
// `/Length` is raw.length, whatever the dict said.
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
### Canonical reals
|
|
85
|
+
|
|
86
|
+
```js
|
|
87
|
+
ser.formatReal(1.5); // '1.5'
|
|
88
|
+
ser.formatReal(2); // '2'
|
|
89
|
+
ser.formatReal(0.0500); // '0.05'
|
|
90
|
+
ser.formatReal(-0); // '0'
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Errors
|
|
94
|
+
|
|
95
|
+
| Code | Class | When |
|
|
96
|
+
|------|-------|------|
|
|
97
|
+
| `pdf/serializer/bad-input` | `RenderError` | `serializeObject` given something other than a typed object. |
|
|
98
|
+
| `pdf/serializer/bad-num` | `RenderError` | `num` not a non-negative integer. |
|
|
99
|
+
| `pdf/serializer/bad-gen` | `RenderError` | `gen` not a non-negative integer. |
|
|
100
|
+
| `pdf/serializer/bad-real` | `RenderError` | Non-finite real (`NaN`, `Infinity`). |
|
|
101
|
+
| `pdf/serializer/bad-string` | `RenderError` | `string.value` is not a `Uint8Array`. |
|
|
102
|
+
| `pdf/serializer/inline-stream` | `RenderError` | Attempt to serialise a `stream` through `serializeObject`. |
|
|
103
|
+
| `pdf/serializer/unknown-type` | `RenderError` | Unrecognised `obj.type`. |
|
|
104
|
+
|
|
105
|
+
## See also
|
|
106
|
+
|
|
107
|
+
- [`pdfParser`](./parser.md) — inverse operation.
|
|
108
|
+
- [`pdfWriter`](../document/writer.md) — main consumer.
|
|
109
|
+
- [`pdfErrors`](../errors.md)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfTokenizer
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: [pdfErrors, pdfShared]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfTokenizer
|
|
11
|
+
|
|
12
|
+
> Binary lexer, ISO 32000-2 §7.2 — bytes → token stream.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfTokenizer` | **Source** `packages/front/office/pdf/src/syntax/tokenizer.js` | **Deps** `pdfErrors`, `pdfShared` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Emits `{ kind, value?, offset, end }` tokens for the PDF lexical conventions.
|
|
17
|
+
The tokenizer is **stream-aware at the keyword level**: when it emits
|
|
18
|
+
`{ kind: 'kw', value: 'stream' }`, the consumer skips the binary payload itself
|
|
19
|
+
(using the preceding dictionary's `/Length`) and then resumes at `endstream`.
|
|
20
|
+
|
|
21
|
+
## Resolve
|
|
22
|
+
|
|
23
|
+
```js
|
|
24
|
+
const tokMod = runtime.resolve('pdfTokenizer');
|
|
25
|
+
// Returns: { tokenize, lastIndexOfBytes }
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## API
|
|
29
|
+
|
|
30
|
+
| Method | Signature | Returns |
|
|
31
|
+
|--------|-----------|---------|
|
|
32
|
+
| `tokenize` | `(bytes: Uint8Array, opts?: { keepWhitespace?, start?, end? }) => Tokenizer` | Iterator (`next`, `peek`, `pos`, `seek`, `bytes`). |
|
|
33
|
+
| `lastIndexOfBytes` | `(bytes: Uint8Array, needle: Uint8Array, from?: number) => number` | Offset (or `-1`) of the last match. |
|
|
34
|
+
|
|
35
|
+
### Token kinds
|
|
36
|
+
|
|
37
|
+
| `kind` | Payload |
|
|
38
|
+
|--------|---------|
|
|
39
|
+
| `ws` | whitespace run (only when `keepWhitespace`) |
|
|
40
|
+
| `comment` | comment body (bytes between `%` and EOL) |
|
|
41
|
+
| `name` | `value: string` (`#xx` escapes decoded) |
|
|
42
|
+
| `int` / `real` | `value: number`, `raw: string` |
|
|
43
|
+
| `string` | `value: Uint8Array` — literal `(...)` form only |
|
|
44
|
+
| `hex` | `value: Uint8Array` — hex `<...>` form |
|
|
45
|
+
| `open_arr` / `close_arr` | `[` / `]` |
|
|
46
|
+
| `open_dict` / `close_dict` | `<<` / `>>` |
|
|
47
|
+
| `kw` | `value: string` (`obj`, `endobj`, `R`, `stream`, `endstream`, `xref`, `trailer`, `startxref`, `true`, `false`, `null`, `n`, `f`) |
|
|
48
|
+
| `eof_marker` | `%%EOF` |
|
|
49
|
+
|
|
50
|
+
### Tokenizer interface
|
|
51
|
+
|
|
52
|
+
| Member | Description |
|
|
53
|
+
|--------|-------------|
|
|
54
|
+
| `next()` | Consumes and returns the next token (or `null` at EOF). |
|
|
55
|
+
| `peek()` | Looks ahead without consuming. |
|
|
56
|
+
| `pos()` | Current offset. |
|
|
57
|
+
| `seek(n)` | Forces the offset (used after skipping a stream body). |
|
|
58
|
+
| `bytes` | Reference to the source `Uint8Array`. |
|
|
59
|
+
|
|
60
|
+
## Examples
|
|
61
|
+
|
|
62
|
+
### Iterating a stream
|
|
63
|
+
|
|
64
|
+
```js
|
|
65
|
+
const tok = tokMod.tokenize(bytes);
|
|
66
|
+
let t;
|
|
67
|
+
while ((t = tok.next())) {
|
|
68
|
+
if (t.kind === 'kw' && t.value === 'xref') break;
|
|
69
|
+
}
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
### Locating `startxref` by backward scan
|
|
73
|
+
|
|
74
|
+
```js
|
|
75
|
+
const needle = new Uint8Array([0x73,0x74,0x61,0x72,0x74,0x78,0x72,0x65,0x66]);
|
|
76
|
+
const at = tokMod.lastIndexOfBytes(bytes, needle);
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
### Starting from a known offset
|
|
80
|
+
|
|
81
|
+
```js
|
|
82
|
+
const tok = tokMod.tokenize(bytes, { start: 12345 });
|
|
83
|
+
tok.next(); // first token at or after 12345
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## Errors
|
|
87
|
+
|
|
88
|
+
| Code | Class | When |
|
|
89
|
+
|------|-------|------|
|
|
90
|
+
| `pdf/tokenizer/bad-input` | `ParseError` | `bytes` is not a `Uint8Array`. |
|
|
91
|
+
| `pdf/tokenizer/bad-seek` | `ParseError` | `seek(n)` out of range. |
|
|
92
|
+
| `pdf/tokenizer/bad-name-escape` | `ParseError` | `#xx` truncated or non-hex. |
|
|
93
|
+
| `pdf/tokenizer/bad-number` | `ParseError` | Empty or unparsable numeric token. |
|
|
94
|
+
| `pdf/tokenizer/bad-string` | `ParseError` | Trailing backslash. |
|
|
95
|
+
| `pdf/tokenizer/unterminated-string` | `ParseError` | `(...)` never closed. |
|
|
96
|
+
| `pdf/tokenizer/bad-hex` | `ParseError` | Non-hex character inside `<...>`. |
|
|
97
|
+
| `pdf/tokenizer/unexpected-rangle` | `ParseError` | Lone `>` outside a hex string. |
|
|
98
|
+
| `pdf/tokenizer/empty-keyword` | `ParseError` | Empty keyword. |
|
|
99
|
+
|
|
100
|
+
## See also
|
|
101
|
+
|
|
102
|
+
- [`pdfParser`](./parser.md) — main consumer.
|
|
103
|
+
- [`pdfXref`](./xref.md) — uses `lastIndexOfBytes` for `startxref`.
|
|
104
|
+
- [`pdfErrors`](../errors.md)
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfTrailer
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: [pdfErrors, pdfParserObj]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfTrailer
|
|
11
|
+
|
|
12
|
+
> Typing of the trailer dictionary, ISO 32000-2 §7.5.5 — `{ size, root, info?, prev?, id?, encrypt? }`.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfTrailer` | **Source** `packages/front/office/pdf/src/syntax/trailer.js` | **Deps** `pdfErrors`, `pdfParserObj` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
A thin layer over the raw trailer dictionary. It checks the **required** entries
|
|
17
|
+
(`/Size`, `/Root`) and extracts the well-defined optional ones. `/Encrypt` is
|
|
18
|
+
left raw (a reference is narrowed to `{num, gen}`, an inline dictionary is passed
|
|
19
|
+
through). Entries this typer does not model — `/XRefStm` for hybrid-reference
|
|
20
|
+
files among them — are reachable only through `raw`.
|
|
21
|
+
|
|
22
|
+
## Resolve
|
|
23
|
+
|
|
24
|
+
```js
|
|
25
|
+
const trailer = runtime.resolve('pdfTrailer');
|
|
26
|
+
// Returns: { typeTrailer }
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## API
|
|
30
|
+
|
|
31
|
+
| Method | Signature | Returns |
|
|
32
|
+
|--------|-----------|---------|
|
|
33
|
+
| `typeTrailer` | `(dict: PdfDict) => TypedTrailer` | Typed record (see shape). |
|
|
34
|
+
|
|
35
|
+
### `TypedTrailer` shape
|
|
36
|
+
|
|
37
|
+
```js
|
|
38
|
+
{
|
|
39
|
+
size: number, // /Size — required
|
|
40
|
+
root: { num: number, gen: number }, // /Root — required
|
|
41
|
+
info?: { num: number, gen: number }, // /Info — only when a ref
|
|
42
|
+
id?: [Uint8Array, Uint8Array], // /ID — pair of strings
|
|
43
|
+
prev?: number, // /Prev — previous xref offset
|
|
44
|
+
encrypt?: { num, gen } | PdfObject, // /Encrypt — ref or inline dict
|
|
45
|
+
raw: PdfDict // the original dictionary
|
|
46
|
+
}
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Examples
|
|
50
|
+
|
|
51
|
+
### Plain typing
|
|
52
|
+
|
|
53
|
+
```js
|
|
54
|
+
const xref = runtime.resolve('pdfXref');
|
|
55
|
+
const trailer = runtime.resolve('pdfTrailer');
|
|
56
|
+
|
|
57
|
+
const { dict } = xref.parseTrailerDict(bytes, end);
|
|
58
|
+
const t = trailer.typeTrailer(dict);
|
|
59
|
+
t.size; // 42
|
|
60
|
+
t.root; // { num: 1, gen: 0 }
|
|
61
|
+
t.prev; // 12345 (on an incremental update)
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### Detecting an encrypted file
|
|
65
|
+
|
|
66
|
+
```js
|
|
67
|
+
if (t.encrypt) {
|
|
68
|
+
const sec = runtime.resolve('pdfSecurity');
|
|
69
|
+
// → see the crypto layer
|
|
70
|
+
}
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## Errors
|
|
74
|
+
|
|
75
|
+
| Code | Class | When |
|
|
76
|
+
|------|-------|------|
|
|
77
|
+
| `pdf/trailer/not-dict` | `ParseError` | Argument is not `{type:'dict'}`. |
|
|
78
|
+
| `pdf/trailer/missing-size` | `ParseError` | `/Size` missing or not a non-negative int. |
|
|
79
|
+
| `pdf/trailer/missing-root` | `ParseError` | `/Root` missing or not a reference. |
|
|
80
|
+
|
|
81
|
+
## See also
|
|
82
|
+
|
|
83
|
+
- [`pdfXref`](./xref.md) — produces the raw dictionary consumed here.
|
|
84
|
+
- [`pdfDocument`](../document/document.md) — chains trailers through `/Prev`.
|
|
85
|
+
- [`pdfErrors`](../errors.md)
|