@awacloud/pdf 0.0.0-stage → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +609 -0
- package/LICENSE +661 -0
- package/NOTICE +77 -0
- package/README.md +363 -2
- package/dist/build/index.js +21 -0
- package/dist/build/pdf-full-rw.js +10972 -0
- package/dist/build/pdf-full-rw.meta.json +105 -0
- package/dist/build/pdf-full-rw.min.js +53 -0
- package/dist/build/pdf-full.js +6078 -0
- package/dist/build/pdf-full.meta.json +90 -0
- package/dist/build/pdf-full.min.js +32 -0
- package/dist/build/pdf-large-rw.js +10367 -0
- package/dist/build/pdf-large-rw.meta.json +99 -0
- package/dist/build/pdf-large-rw.min.js +53 -0
- package/dist/build/pdf-large.js +5473 -0
- package/dist/build/pdf-large.meta.json +84 -0
- package/dist/build/pdf-large.min.js +32 -0
- package/dist/build/pdf-legacy-rw.js +12402 -0
- package/dist/build/pdf-legacy-rw.meta.json +110 -0
- package/dist/build/pdf-legacy-rw.min.js +53 -0
- package/dist/build/pdf-legacy.js +7508 -0
- package/dist/build/pdf-legacy.meta.json +95 -0
- package/dist/build/pdf-legacy.min.js +32 -0
- package/dist/build/pdf-rw.js +7578 -0
- package/dist/build/pdf-rw.meta.json +77 -0
- package/dist/build/pdf-rw.min.js +53 -0
- package/dist/build/pdf.js +2684 -0
- package/dist/build/pdf.meta.json +62 -0
- package/dist/build/pdf.min.js +32 -0
- package/dist/standalone/pdf-full-rw.js +16798 -0
- package/dist/standalone/pdf-full-rw.meta.json +78 -0
- package/dist/standalone/pdf-full-rw.min.js +56 -0
- package/dist/standalone/pdf-full.js +11904 -0
- package/dist/standalone/pdf-full.meta.json +63 -0
- package/dist/standalone/pdf-full.min.js +35 -0
- package/dist/standalone/pdf-large-rw.js +16193 -0
- package/dist/standalone/pdf-large-rw.meta.json +72 -0
- package/dist/standalone/pdf-large-rw.min.js +56 -0
- package/dist/standalone/pdf-large.js +11299 -0
- package/dist/standalone/pdf-large.meta.json +57 -0
- package/dist/standalone/pdf-large.min.js +35 -0
- package/dist/standalone/pdf-legacy-rw.js +18228 -0
- package/dist/standalone/pdf-legacy-rw.meta.json +83 -0
- package/dist/standalone/pdf-legacy-rw.min.js +56 -0
- package/dist/standalone/pdf-legacy.js +13334 -0
- package/dist/standalone/pdf-legacy.meta.json +68 -0
- package/dist/standalone/pdf-legacy.min.js +35 -0
- package/dist/standalone/pdf-rw.js +13404 -0
- package/dist/standalone/pdf-rw.meta.json +50 -0
- package/dist/standalone/pdf-rw.min.js +56 -0
- package/dist/standalone/pdf.js +8510 -0
- package/dist/standalone/pdf.meta.json +35 -0
- package/dist/standalone/pdf.min.js +35 -0
- package/docs/README.md +53 -0
- package/docs/api/README.md +38 -0
- package/docs/api/_shared/README.md +91 -0
- package/docs/api/action/README.md +29 -0
- package/docs/api/action/action.md +81 -0
- package/docs/api/action/goTo.md +66 -0
- package/docs/api/action/launch.md +58 -0
- package/docs/api/action/named.md +55 -0
- package/docs/api/action/uri.md +54 -0
- package/docs/api/annot/README.md +53 -0
- package/docs/api/annot/annot.md +114 -0
- package/docs/api/annot/fileAttach.md +53 -0
- package/docs/api/annot/freeText.md +68 -0
- package/docs/api/annot/ink.md +69 -0
- package/docs/api/annot/link.md +74 -0
- package/docs/api/annot/markup.md +83 -0
- package/docs/api/annot/popup.md +52 -0
- package/docs/api/annot/projection.md +56 -0
- package/docs/api/annot/redact.md +67 -0
- package/docs/api/annot/square.md +87 -0
- package/docs/api/annot/stamp.md +54 -0
- package/docs/api/annot/text.md +69 -0
- package/docs/api/annot/widget.md +69 -0
- package/docs/api/associatedFiles/README.md +9 -0
- package/docs/api/associatedFiles/associatedFiles.md +78 -0
- package/docs/api/bundles/README.md +68 -0
- package/docs/api/bundles/dist-matrix.md +165 -0
- package/docs/api/bundles/pdf-full.md +148 -0
- package/docs/api/bundles/pdf-large.md +144 -0
- package/docs/api/bundles/pdf-legacy.md +169 -0
- package/docs/api/content/README.md +29 -0
- package/docs/api/content/color.md +99 -0
- package/docs/api/content/graphics.md +114 -0
- package/docs/api/content/images.md +124 -0
- package/docs/api/content/ops.md +100 -0
- package/docs/api/content/stream.md +107 -0
- package/docs/api/content/text.md +98 -0
- package/docs/api/crypto/README.md +29 -0
- package/docs/api/crypto/aesGcm.md +72 -0
- package/docs/api/crypto/permissions.md +79 -0
- package/docs/api/crypto/security.md +98 -0
- package/docs/api/crypto/standardV4.md +104 -0
- package/docs/api/crypto/standardV5.md +84 -0
- package/docs/api/crypto/standardV6.md +93 -0
- package/docs/api/destination/README.md +9 -0
- package/docs/api/destination/destination.md +79 -0
- package/docs/api/document/README.md +29 -0
- package/docs/api/document/builder.md +281 -0
- package/docs/api/document/catalog.md +98 -0
- package/docs/api/document/document.md +187 -0
- package/docs/api/document/encryptedWriter.md +149 -0
- package/docs/api/document/incrementalWriter.md +148 -0
- package/docs/api/document/page.md +99 -0
- package/docs/api/document/pages.md +82 -0
- package/docs/api/document/resources.md +102 -0
- package/docs/api/document/writer.md +157 -0
- package/docs/api/document/xrefStreamWriter.md +122 -0
- package/docs/api/embedded/README.md +13 -0
- package/docs/api/embedded/collection.md +80 -0
- package/docs/api/embedded/embeddedFile.md +86 -0
- package/docs/api/embedded/fileSpec.md +87 -0
- package/docs/api/errors.md +110 -0
- package/docs/api/extra/3d-richmedia.md +76 -0
- package/docs/api/extra/README.md +99 -0
- package/docs/api/extra/annot-extended.md +71 -0
- package/docs/api/extra/associated-files.md +70 -0
- package/docs/api/extra/ccitt-fax-decoder.md +74 -0
- package/docs/api/extra/color-spaces-extended.md +72 -0
- package/docs/api/extra/content-ops-extended.md +82 -0
- package/docs/api/extra/document-parts.md +69 -0
- package/docs/api/extra/embedded-files-portfolio.md +87 -0
- package/docs/api/extra/font-cid-typed.md +77 -0
- package/docs/api/extra/font-color-tagging.md +76 -0
- package/docs/api/extra/form-actions-extended.md +75 -0
- package/docs/api/extra/info-dict-deprecated.md +72 -0
- package/docs/api/extra/jbig2-read.md +80 -0
- package/docs/api/extra/legacy-deprecated-annots.md +89 -0
- package/docs/api/extra/legacy-deprecated-filters.md +78 -0
- package/docs/api/extra/legacy-rc4-read.md +74 -0
- package/docs/api/extra/legacy-xfa-read.md +65 -0
- package/docs/api/extra/linearization-write.md +71 -0
- package/docs/api/extra/misc.md +93 -0
- package/docs/api/extra/optional-content-extended.md +83 -0
- package/docs/api/extra/pdf-a-output-intent.md +65 -0
- package/docs/api/extra/pdf-sandbox.md +76 -0
- package/docs/api/extra/pdf-ua-tagged.md +63 -0
- package/docs/api/extra/pdf-x-prepress.md +65 -0
- package/docs/api/extra/redaction-iso32005.md +65 -0
- package/docs/api/extra/shading-typed.md +73 -0
- package/docs/api/extra/sig-aes-gcm.md +69 -0
- package/docs/api/extra/sig-pades.md +103 -0
- package/docs/api/extra/tagged-pdf-typed.md +78 -0
- package/docs/api/extra/transparency-typed.md +74 -0
- package/docs/api/extra/well-tagged-pdf.md +61 -0
- package/docs/api/extra/xmp-extended.md +65 -0
- package/docs/api/font/README.md +25 -0
- package/docs/api/font/embed.md +157 -0
- package/docs/api/font/encoding.md +95 -0
- package/docs/api/font/font.md +97 -0
- package/docs/api/font/type3.md +89 -0
- package/docs/api/form/README.md +35 -0
- package/docs/api/form/acroform.md +88 -0
- package/docs/api/form/appearance.md +87 -0
- package/docs/api/form/button.md +97 -0
- package/docs/api/form/choice.md +96 -0
- package/docs/api/form/fieldTree.md +93 -0
- package/docs/api/form/signature.md +90 -0
- package/docs/api/form/text.md +88 -0
- package/docs/api/linearization/README.md +11 -0
- package/docs/api/linearization/linearization.md +81 -0
- package/docs/api/main.md +116 -0
- package/docs/api/metadata/README.md +10 -0
- package/docs/api/metadata/info.md +70 -0
- package/docs/api/metadata/xmp.md +62 -0
- package/docs/api/ocg/README.md +23 -0
- package/docs/api/ocg/config.md +95 -0
- package/docs/api/ocg/ocg.md +77 -0
- package/docs/api/outline/README.md +11 -0
- package/docs/api/outline/outline.md +107 -0
- package/docs/api/pdf.md +152 -0
- package/docs/api/prepress/README.md +10 -0
- package/docs/api/prepress/outputIntent.md +79 -0
- package/docs/api/prepress/pageBoundary.md +75 -0
- package/docs/api/sig/README.md +32 -0
- package/docs/api/sig/byteRange.md +120 -0
- package/docs/api/sig/certChain.md +84 -0
- package/docs/api/sig/dss.md +111 -0
- package/docs/api/sig/oids.md +76 -0
- package/docs/api/sig/sha1.md +72 -0
- package/docs/api/sig/sign.md +317 -0
- package/docs/api/sig/signature.md +178 -0
- package/docs/api/sig/timestamp.md +84 -0
- package/docs/api/syntax/README.md +29 -0
- package/docs/api/syntax/crossRefStream.md +115 -0
- package/docs/api/syntax/filters/README.md +50 -0
- package/docs/api/syntax/filters/ascii85.md +76 -0
- package/docs/api/syntax/filters/asciiHex.md +73 -0
- package/docs/api/syntax/filters/dispatch.md +125 -0
- package/docs/api/syntax/filters/flate.md +134 -0
- package/docs/api/syntax/filters/runLength.md +78 -0
- package/docs/api/syntax/objStream.md +88 -0
- package/docs/api/syntax/parser-obj.md +97 -0
- package/docs/api/syntax/parser.md +151 -0
- package/docs/api/syntax/serializer.md +109 -0
- package/docs/api/syntax/tokenizer.md +104 -0
- package/docs/api/syntax/trailer.md +85 -0
- package/docs/api/syntax/xref.md +139 -0
- package/docs/api/tagged/README.md +25 -0
- package/docs/api/tagged/classMap.md +67 -0
- package/docs/api/tagged/markedContent.md +62 -0
- package/docs/api/tagged/parentTree.md +67 -0
- package/docs/api/tagged/roleMap.md +67 -0
- package/docs/api/tagged/structElement.md +76 -0
- package/docs/api/tagged/structTree.md +75 -0
- package/docs/guide/coverage.md +113 -0
- package/docs/guide/crypto.md +121 -0
- package/docs/guide/extending.md +76 -0
- package/docs/guide/getting-started.md +75 -0
- package/docs/guide/legacy-1.7.md +42 -0
- package/docs/guide/pades-integration.md +579 -0
- package/docs/guide/read-pdf.md +89 -0
- package/package.json +97 -4
- package/src/_shared/index.js +179 -0
- package/src/action/action.js +119 -0
- package/src/action/goTo.js +89 -0
- package/src/action/launch.js +61 -0
- package/src/action/named.js +54 -0
- package/src/action/uri.js +51 -0
- package/src/annot/annot.js +212 -0
- package/src/annot/fileAttach.js +55 -0
- package/src/annot/freeText.js +82 -0
- package/src/annot/ink.js +77 -0
- package/src/annot/link.js +77 -0
- package/src/annot/markup.js +91 -0
- package/src/annot/popup.js +53 -0
- package/src/annot/projection.js +52 -0
- package/src/annot/redact.js +87 -0
- package/src/annot/square.js +132 -0
- package/src/annot/stamp.js +48 -0
- package/src/annot/text.js +54 -0
- package/src/annot/widget.js +61 -0
- package/src/associatedFiles/associatedFiles.js +86 -0
- package/src/bundles/pdf-full.js +91 -0
- package/src/bundles/pdf-large.js +81 -0
- package/src/bundles/pdf-legacy.js +107 -0
- package/src/content/color.js +114 -0
- package/src/content/graphics.js +192 -0
- package/src/content/images.js +160 -0
- package/src/content/ops.js +137 -0
- package/src/content/stream.js +154 -0
- package/src/content/text.js +125 -0
- package/src/crypto/aesGcm.js +123 -0
- package/src/crypto/permissions.js +112 -0
- package/src/crypto/security.js +327 -0
- package/src/crypto/standardV4.js +443 -0
- package/src/crypto/standardV5.js +306 -0
- package/src/crypto/standardV6.js +334 -0
- package/src/destination/destination.js +183 -0
- package/src/document/builder.js +618 -0
- package/src/document/catalog.js +100 -0
- package/src/document/document.js +472 -0
- package/src/document/encryptedWriter.js +554 -0
- package/src/document/incrementalWriter.js +514 -0
- package/src/document/page.js +131 -0
- package/src/document/pages.js +103 -0
- package/src/document/resources.js +146 -0
- package/src/document/writer.js +211 -0
- package/src/document/xrefStreamWriter.js +353 -0
- package/src/embedded/collection.js +102 -0
- package/src/embedded/embeddedFile.js +99 -0
- package/src/embedded/fileSpec.js +137 -0
- package/src/errors.js +78 -0
- package/src/extra/3d-richmedia.js +171 -0
- package/src/extra/annot-extended.js +200 -0
- package/src/extra/associated-files.js +131 -0
- package/src/extra/ccitt-fax-decoder.js +776 -0
- package/src/extra/color-spaces-extended.js +196 -0
- package/src/extra/content-ops-extended.js +153 -0
- package/src/extra/document-parts.js +149 -0
- package/src/extra/embedded-files-portfolio.js +234 -0
- package/src/extra/font-cid-typed.js +185 -0
- package/src/extra/font-color-tagging.js +144 -0
- package/src/extra/form-actions-extended.js +196 -0
- package/src/extra/info-dict-deprecated.js +137 -0
- package/src/extra/jbig2-read.js +169 -0
- package/src/extra/legacy-deprecated-annots.js +198 -0
- package/src/extra/legacy-deprecated-filters.js +167 -0
- package/src/extra/legacy-rc4-read.js +235 -0
- package/src/extra/legacy-xfa-read.js +104 -0
- package/src/extra/linearization-write.js +97 -0
- package/src/extra/misc.js +217 -0
- package/src/extra/optional-content-extended.js +142 -0
- package/src/extra/pdf-a-output-intent.js +112 -0
- package/src/extra/pdf-sandbox.js +88 -0
- package/src/extra/pdf-ua-tagged.js +116 -0
- package/src/extra/pdf-x-prepress.js +114 -0
- package/src/extra/redaction-iso32005.js +136 -0
- package/src/extra/shading-typed.js +222 -0
- package/src/extra/sig-aes-gcm.js +135 -0
- package/src/extra/sig-pades.js +242 -0
- package/src/extra/tagged-pdf-typed.js +203 -0
- package/src/extra/transparency-typed.js +135 -0
- package/src/extra/well-tagged-pdf.js +138 -0
- package/src/extra/xmp-extended.js +190 -0
- package/src/font/embed.js +480 -0
- package/src/font/encoding.js +92 -0
- package/src/font/font.js +101 -0
- package/src/font/type3.js +75 -0
- package/src/form/acroform.js +94 -0
- package/src/form/appearance.js +90 -0
- package/src/form/button.js +105 -0
- package/src/form/choice.js +152 -0
- package/src/form/fieldTree.js +120 -0
- package/src/form/signature.js +100 -0
- package/src/form/text.js +101 -0
- package/src/linearization/linearization.js +107 -0
- package/src/main.js +411 -0
- package/src/metadata/info.js +87 -0
- package/src/metadata/xmp.js +62 -0
- package/src/ocg/config.js +156 -0
- package/src/ocg/ocg.js +124 -0
- package/src/outline/outline.js +157 -0
- package/src/pdf.js +133 -0
- package/src/prepress/outputIntent.js +118 -0
- package/src/prepress/pageBoundary.js +108 -0
- package/src/sig/byteRange.js +306 -0
- package/src/sig/certChain.js +247 -0
- package/src/sig/dss.js +317 -0
- package/src/sig/oids.js +157 -0
- package/src/sig/sha1.js +142 -0
- package/src/sig/sign.js +1899 -0
- package/src/sig/signature.js +1441 -0
- package/src/sig/timestamp.js +236 -0
- package/src/syntax/crossRefStream.js +133 -0
- package/src/syntax/filters/ascii85.js +122 -0
- package/src/syntax/filters/asciiHex.js +83 -0
- package/src/syntax/filters/dispatch.js +176 -0
- package/src/syntax/filters/flate.js +316 -0
- package/src/syntax/filters/runLength.js +96 -0
- package/src/syntax/objStream.js +99 -0
- package/src/syntax/parser-obj.js +52 -0
- package/src/syntax/parser.js +321 -0
- package/src/syntax/serializer.js +221 -0
- package/src/syntax/tokenizer.js +290 -0
- package/src/syntax/trailer.js +76 -0
- package/src/syntax/xref.js +341 -0
- package/src/tagged/classMap.js +81 -0
- package/src/tagged/markedContent.js +123 -0
- package/src/tagged/parentTree.js +126 -0
- package/src/tagged/roleMap.js +107 -0
- package/src/tagged/structElement.js +138 -0
- package/src/tagged/structTree.js +94 -0
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfXref
|
|
3
|
+
category: pdf/syntax
|
|
4
|
+
dependencies: [pdfErrors, pdfTokenizer, pdfParser]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfXref
|
|
11
|
+
|
|
12
|
+
> Classical cross-reference table, ISO 32000-2 §7.5.4 — locate + parse + trailer; plus the xref-stream dict reader and emitter the incremental writer uses.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfXref` | **Source** `packages/front/office/pdf/src/syntax/xref.js` | **Deps** `pdfErrors`, `pdfTokenizer`, `pdfParser` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Reads the **classical table** (`xref` keyword plus 20-byte subsection entries)
|
|
17
|
+
only. Cross-reference streams (`/Type /XRef`, §7.5.8) and object streams
|
|
18
|
+
(`/Type /ObjStm`, §7.5.7) are handled by the dedicated
|
|
19
|
+
[`pdfCrossRefStream`](./crossRefStream.md) and [`pdfObjStream`](./objStream.md)
|
|
20
|
+
modules, which [`pdfDocument`](../document/document.md) composes: the section
|
|
21
|
+
walk calls `parseXrefTable` only when the bytes at the offset actually start
|
|
22
|
+
with `xref`. The backward scan tolerance for `startxref`
|
|
23
|
+
is widened to 8192 bytes (versus the 1024 of the spec) to absorb annotated
|
|
24
|
+
`%%EOF` tails.
|
|
25
|
+
|
|
26
|
+
Two cross-reference-stream helpers serve
|
|
27
|
+
[`pdfIncrementalWriter`](../document/incrementalWriter.md).
|
|
28
|
+
`readXrefStreamDict` reads only the **dictionary** of a `/Type /XRef` stream
|
|
29
|
+
section and never decodes its data, so it needs no filter module.
|
|
30
|
+
`buildXrefStream` emits one uncompressed `/Type /XRef` stream section for an
|
|
31
|
+
incremental update. Decoding a stream's entries stays the job of
|
|
32
|
+
[`pdfCrossRefStream`](./crossRefStream.md).
|
|
33
|
+
|
|
34
|
+
## Resolve
|
|
35
|
+
|
|
36
|
+
```js
|
|
37
|
+
const xref = runtime.resolve('pdfXref');
|
|
38
|
+
// Returns: { locateStartXref, readStartXref, parseXrefTable, parseTrailerDict,
|
|
39
|
+
// readXrefStreamDict, buildXrefStream }
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## API
|
|
43
|
+
|
|
44
|
+
| Method | Signature | Returns |
|
|
45
|
+
|--------|-----------|---------|
|
|
46
|
+
| `locateStartXref` | `(bytes: Uint8Array) => number` | Offset of the last `startxref`, or `-1`. |
|
|
47
|
+
| `readStartXref` | `(bytes: Uint8Array, at: number) => number` | Offset of the xref it points to. |
|
|
48
|
+
| `parseXrefTable` | `(bytes: Uint8Array, at: number) => { entries, end }` | Table plus the offset just after it. |
|
|
49
|
+
| `parseTrailerDict` | `(bytes: Uint8Array, at: number) => { dict, end }` | Raw trailer plus the offset just after it. |
|
|
50
|
+
| `readXrefStreamDict` | `(bytes: Uint8Array, at: number) => { num, gen, dict }` | The raw dict of the `/Type /XRef` stream object at `at`; the data is not decoded. |
|
|
51
|
+
| `buildXrefStream` | `(opts: XrefStreamOpts) => Uint8Array` | One complete `num 0 obj … endobj` xref-stream section. The caller appends `startxref`/`%%EOF`. |
|
|
52
|
+
|
|
53
|
+
### `XrefStreamOpts`
|
|
54
|
+
|
|
55
|
+
```js
|
|
56
|
+
{
|
|
57
|
+
num: number, // the stream's own object number (>= 1)
|
|
58
|
+
offset: number, // byte offset where the object will start
|
|
59
|
+
entries: [ { num, offset, gen? }, … ], // type-1 rows; object 0's free head and
|
|
60
|
+
// the stream's own row are added for you
|
|
61
|
+
prev: number, // /Prev — the previous section's offset
|
|
62
|
+
root: { num, gen }, // /Root
|
|
63
|
+
info?: { num, gen } | null, // /Info
|
|
64
|
+
id?: [ Uint8Array, Uint8Array ] | null, // /ID
|
|
65
|
+
size?: number, // /Size = max(size, num + 1)
|
|
66
|
+
encrypt?: { num, gen } | null // /Encrypt, written right after /ID
|
|
67
|
+
}
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
`encrypt` repeats an encrypted document's `/Encrypt` in the update
|
|
71
|
+
(ISO 32000-2 §7.5.6). Only an indirect reference is accepted. Without it
|
|
72
|
+
the output is unchanged.
|
|
73
|
+
|
|
74
|
+
`/W` is the narrowest `[1 w2 w3]` that fits the rows, and `/Index` holds one
|
|
75
|
+
pair per contiguous run of object numbers. No `/Filter` is written.
|
|
76
|
+
|
|
77
|
+
### `entries` shape
|
|
78
|
+
|
|
79
|
+
```js
|
|
80
|
+
{
|
|
81
|
+
[num: number]: { offset: number, gen: number, free: boolean }
|
|
82
|
+
}
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
`free === true` corresponds to the `f` flag (released object; must not be
|
|
86
|
+
dereferenced).
|
|
87
|
+
|
|
88
|
+
## Examples
|
|
89
|
+
|
|
90
|
+
### Full xref pipeline
|
|
91
|
+
|
|
92
|
+
```js
|
|
93
|
+
const xref = runtime.resolve('pdfXref');
|
|
94
|
+
|
|
95
|
+
const sxAt = xref.locateStartXref(bytes);
|
|
96
|
+
if (sxAt < 0) throw new Error('no startxref');
|
|
97
|
+
const xrefAt = xref.readStartXref(bytes, sxAt);
|
|
98
|
+
|
|
99
|
+
const { entries, end } = xref.parseXrefTable(bytes, xrefAt);
|
|
100
|
+
const { dict: trailer } = xref.parseTrailerDict(bytes, end);
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
### Chaining through `/Prev` (incremental update)
|
|
104
|
+
|
|
105
|
+
```js
|
|
106
|
+
const sections = [];
|
|
107
|
+
let cursor = xrefAt;
|
|
108
|
+
for (let i = 0; i < 32 && cursor >= 0; i++) {
|
|
109
|
+
const s = xref.parseXrefTable(bytes, cursor);
|
|
110
|
+
sections.push(s);
|
|
111
|
+
const { dict } = xref.parseTrailerDict(bytes, s.end);
|
|
112
|
+
const prev = dict.entries.Prev;
|
|
113
|
+
cursor = prev && prev.type === 'int' ? prev.value : -1;
|
|
114
|
+
}
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
## Errors
|
|
118
|
+
|
|
119
|
+
| Code | Class | When |
|
|
120
|
+
|------|-------|------|
|
|
121
|
+
| `pdf/xref/no-startxref` | `ParseError` | No `startxref` keyword at `at`. |
|
|
122
|
+
| `pdf/xref/bad-startxref` | `ParseError` | `startxref` not followed by an int ≥ 0. |
|
|
123
|
+
| `pdf/xref/no-xref-keyword` | `ParseError` | No `xref` at the expected offset. `pdfDocument` never produces it for a cross-reference stream — it inspects the bytes first and takes the stream path. |
|
|
124
|
+
| `pdf/xref/bad-subsection-header` | `ParseError` | Invalid `<first> <count>` header. |
|
|
125
|
+
| `pdf/xref/truncated-entry` | `ParseError` | Subsection cut short. |
|
|
126
|
+
| `pdf/xref/bad-entry-format` | `ParseError` | Missing space at column 10 or 16. |
|
|
127
|
+
| `pdf/xref/bad-entry-flag` | `ParseError` | Final flag is neither `n` nor `f`. |
|
|
128
|
+
| `pdf/xref/bad-digit` | `ParseError` | Non-ASCII-digit character inside `xxxxxxxxxx`/`ggggg`. |
|
|
129
|
+
| `pdf/xref/no-trailer` | `ParseError` | No `trailer` keyword. |
|
|
130
|
+
| `pdf/xref/trailer-not-dict` | `ParseError` | `trailer` not followed by a dictionary. |
|
|
131
|
+
| `pdf/xref/not-xref-stream` | `ParseError` | `readXrefStreamDict`: no indirect object at `at`, or one that is not a `/Type /XRef` stream. |
|
|
132
|
+
| `pdf/xref/bad-stream-section` | `RenderError` | `buildXrefStream`: unusable `num`, `offset`, `prev`, `root`, `encrypt` (anything but an indirect reference) or entry. |
|
|
133
|
+
|
|
134
|
+
## See also
|
|
135
|
+
|
|
136
|
+
- [`pdfTrailer`](./trailer.md) — types the dictionary returned here.
|
|
137
|
+
- [`pdfCrossRefStream`](./crossRefStream.md) — the §7.5.8 alternative.
|
|
138
|
+
- [`pdfDocument`](../document/document.md) — orchestrates xref plus the `/Prev` chain.
|
|
139
|
+
- [`pdfTokenizer`](./tokenizer.md), [`pdfParser`](./parser.md)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# Tagged PDF — ISO 32000-2 §14.6 / §14.7 / §14.8
|
|
2
|
+
|
|
3
|
+
Logical layer: structure tree, marked content, role mapping. Basis of PDF/UA-2 (ISO 14289-2).
|
|
4
|
+
|
|
5
|
+
| Module | Returns | Deps | Description |
|
|
6
|
+
|--------|----------|------|-------------|
|
|
7
|
+
| [`pdfStructTree`](./structTree.md) | `{ typeStructTreeRoot }` | `pdfErrors`, `pdfParser` | Root, §14.7.2. |
|
|
8
|
+
| [`pdfStructElement`](./structElement.md) | `{ typeStructElement }` | `pdfErrors`, `pdfParser` | Elem, §14.7.3. |
|
|
9
|
+
| [`pdfRoleMap`](./roleMap.md) | `{ typeRoleMap, resolveStandardType, isStandardType, STANDARD_TYPES }` | `pdfErrors`, `pdfParser` | RoleMap, §14.7.4. |
|
|
10
|
+
| [`pdfParentTree`](./parentTree.md) | `{ lookupParent }` | `pdfErrors`, `pdfParser` | ParentTree, §14.7.5. |
|
|
11
|
+
| [`pdfClassMap`](./classMap.md) | `{ typeClassMap, getClassAttributes }` | `pdfErrors`, `pdfParser` | ClassMap, §14.7.6. |
|
|
12
|
+
| [`pdfMarkedContent`](./markedContent.md) | `{ extractMcids, resolveMcidToStruct }` | `pdfErrors`, `pdfParser`, `pdfParentTree` | BMC/BDC/EMC, §14.6. |
|
|
13
|
+
|
|
14
|
+
## PDF/UA-2 pattern
|
|
15
|
+
|
|
16
|
+
```js
|
|
17
|
+
const st = runtime.resolve('pdfStructTree');
|
|
18
|
+
const root = st.typeStructTreeRoot(doc._raw.resolve(doc.catalog.structTreeRoot));
|
|
19
|
+
// walk root.kids → pdfStructElement.typeStructElement
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## See also
|
|
23
|
+
|
|
24
|
+
- [Content streams](../content/README.md)
|
|
25
|
+
- [Annot — Widget](../annot/widget.md)
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfClassMap
|
|
3
|
+
category: pdf/tagged
|
|
4
|
+
dependencies: [pdfErrors, pdfParser]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfClassMap
|
|
11
|
+
|
|
12
|
+
> ClassMap — ISO 32000-2 §14.7.6 — dict of reusable named attribute sets.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfClassMap` | **Source** `packages/front/office/pdf/src/tagged/classMap.js` | **Deps** `pdfErrors`, `pdfParser` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
`/ClassMap` (on StructTreeRoot) is a dict `name → dict | array<dict>`: each key is an attribute-class name, the value is one or more attribute dicts (§14.7.7). A `StructElement` may reference one or more classes via `/C`. `getClassAttributes(typed, name, owner?)` always returns an **array** — filtered to entries whose `/O` equals `owner` when supplied, or the full list otherwise (never a bare dict, never `null`).
|
|
17
|
+
|
|
18
|
+
## Resolve
|
|
19
|
+
|
|
20
|
+
```js
|
|
21
|
+
const cm = runtime.resolve('pdfClassMap');
|
|
22
|
+
// Returns: { typeClassMap, getClassAttributes }
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## API
|
|
26
|
+
|
|
27
|
+
| Method | Signature | Returns |
|
|
28
|
+
|--------|-----------|---------|
|
|
29
|
+
| `typeClassMap` | `(dict) => { classes: Record<string, dict[]>, raw }` | Typing — `classes` is a **plain object**, not a `Map`. |
|
|
30
|
+
| `getClassAttributes` | `(typed, name, owner?: string) => dict[]` | Lookup; `[]` when `name` is unknown. |
|
|
31
|
+
|
|
32
|
+
## Examples
|
|
33
|
+
|
|
34
|
+
### Reading the classes
|
|
35
|
+
|
|
36
|
+
```js
|
|
37
|
+
const cm = runtime.resolve('pdfClassMap').typeClassMap(root.classMap);
|
|
38
|
+
cm.classes.Heading1; // [{type:'dict', entries:{O:{value:'Layout'}, …}}, …]
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
### Lookup by owner
|
|
42
|
+
|
|
43
|
+
```js
|
|
44
|
+
const layout = cm.getClassAttributes(cm, 'Heading1', 'Layout');
|
|
45
|
+
if (layout.length) layout[0].entries.SpaceBefore;
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### Applying to a StructElement
|
|
49
|
+
|
|
50
|
+
```js
|
|
51
|
+
const el = runtime.resolve('pdfStructElement').typeStructElement(dict);
|
|
52
|
+
for (const className of (el.c && el.c.items) || []) {
|
|
53
|
+
const attrs = cm.getClassAttributes(cm, className.value);
|
|
54
|
+
}
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Errors
|
|
58
|
+
|
|
59
|
+
| Code | Class | When |
|
|
60
|
+
|------|--------|------|
|
|
61
|
+
| `pdf/tagged/class-map/not-dict` | `ParseError` | Argument is not a dict. |
|
|
62
|
+
| `pdf/tagged/class-map/bad-value` | `ParseError` | Value is neither a dict nor an array. |
|
|
63
|
+
| `pdf/tagged/class-map/bad-item` | `ParseError` | An array element is not a dict. |
|
|
64
|
+
|
|
65
|
+
## See also
|
|
66
|
+
|
|
67
|
+
- [`pdfStructElement`](./structElement.md) · [`pdfStructTree`](./structTree.md)
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfMarkedContent
|
|
3
|
+
category: pdf/tagged
|
|
4
|
+
dependencies: [pdfErrors, pdfParser, pdfParentTree]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfMarkedContent
|
|
11
|
+
|
|
12
|
+
> Marked-content sequences — ISO 32000-2 §14.6 — bridge between content streams and the struct tree.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfMarkedContent` | **Source** `packages/front/office/pdf/src/tagged/markedContent.js` | **Deps** `pdfErrors`, `pdfParser`, `pdfParentTree` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
The `BMC` / `BDC` / `EMC` operators (§14.6.1) delimit marked sequences in a content stream. `BDC` may carry an `MCID` in its properties dict, used as the key into the `ParentTree` (§14.7.5) to link content back to a `StructElement`. This module provides `extractMcids` (walks an ops list already typed by [`pdfContentStream`](../content/stream.md)) and `resolveMcidToStruct` (full resolution via `pdfParentTree.lookupParent`).
|
|
17
|
+
|
|
18
|
+
## Resolve
|
|
19
|
+
|
|
20
|
+
```js
|
|
21
|
+
const mc = runtime.resolve('pdfMarkedContent');
|
|
22
|
+
// Returns: { extractMcids, resolveMcidToStruct }
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## API
|
|
26
|
+
|
|
27
|
+
| Method | Signature | Returns |
|
|
28
|
+
|--------|-----------|---------|
|
|
29
|
+
| `extractMcids` | `(opsList) => Array<{ tag, start, end, mcid, properties }>` | Linear walk, tracks the BMC/BDC/EMC stack; `start`/`end` are inclusive op-index positions, sorted by `start`. |
|
|
30
|
+
| `resolveMcidToStruct` | `(pageDict, mcid, parentTree, resolveRef) => PdfObject \| null` | Lookup via `pdfParentTree.lookupParent`, indexed by the page's `/StructParents` then `mcid`. |
|
|
31
|
+
|
|
32
|
+
## Examples
|
|
33
|
+
|
|
34
|
+
### Extraction from a content stream
|
|
35
|
+
|
|
36
|
+
```js
|
|
37
|
+
const ops = runtime.resolve('pdfContentStream').parseContentStream(streamBytes);
|
|
38
|
+
const mc = runtime.resolve('pdfMarkedContent');
|
|
39
|
+
const seqs = mc.extractMcids(ops);
|
|
40
|
+
// [{ tag: 'P', start: 3, end: 9, mcid: 0, properties: {…} }, …]
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
### Resolution to the struct elem
|
|
44
|
+
|
|
45
|
+
```js
|
|
46
|
+
const elemRef = mc.resolveMcidToStruct(pageDict, 0, parentTree, resolveRef);
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Errors
|
|
50
|
+
|
|
51
|
+
| Code | Class | When |
|
|
52
|
+
|------|--------|------|
|
|
53
|
+
| `pdf/tagged/marked-content/bad-ops` | `ParseError` | Argument is not an array. |
|
|
54
|
+
| `pdf/tagged/marked-content/unmatched-emc` | `ParseError` | `EMC` without a matching `BMC`/`BDC`. |
|
|
55
|
+
| `pdf/tagged/marked-content/unterminated` | `ParseError` | Non-empty stack at end of stream. |
|
|
56
|
+
| `pdf/tagged/marked-content/bad-tag` | `ParseError` | `BMC`/`BDC` tag is not a name. |
|
|
57
|
+
| `pdf/tagged/marked-content/bad-page` | `ParseError` | Page argument is not a dict. |
|
|
58
|
+
| `pdf/tagged/marked-content/bad-entry` | `ParseError` | ParentTree entry for the page is not an array. |
|
|
59
|
+
|
|
60
|
+
## See also
|
|
61
|
+
|
|
62
|
+
- [`pdfContentStream`](../content/stream.md) · [`pdfStructTree`](./structTree.md) · [`pdfParentTree`](./parentTree.md)
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfParentTree
|
|
3
|
+
category: pdf/tagged
|
|
4
|
+
dependencies: [pdfErrors, pdfParser]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfParentTree
|
|
11
|
+
|
|
12
|
+
> ParentTree — ISO 32000-2 §14.7.5 — number tree, MCID → struct element / OBJR.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfParentTree` | **Source** `packages/front/office/pdf/src/tagged/parentTree.js` | **Deps** `pdfErrors`, `pdfParser` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
The StructTreeRoot's `/ParentTree` is a *number tree* (§7.9.7) mapping `StructParents`/`StructParent` keys (on pages, annotations, content-stream objects) to their parent structure elements. The module provides `lookupParent(tree, key, resolveRef, opts?)`, which descends `/Kids` nodes (checking `/Limits`) then scans `/Nums` leaves until `key` is found. Returns either a single ref (annotations/form XObjects) or an array indexed by MCID (pages).
|
|
17
|
+
|
|
18
|
+
## Resolve
|
|
19
|
+
|
|
20
|
+
```js
|
|
21
|
+
const pt = runtime.resolve('pdfParentTree');
|
|
22
|
+
// Returns: { lookupParent }
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## API
|
|
26
|
+
|
|
27
|
+
| Method | Signature | Returns |
|
|
28
|
+
|--------|-----------|---------|
|
|
29
|
+
| `lookupParent` | `(tree, key: number, resolveRef, opts?: { maxDepth? }) => PdfObject \| null` | Value of the number tree for `key`. |
|
|
30
|
+
|
|
31
|
+
`opts.maxDepth` defaults to `32`. Returns `null` when `key` is not found.
|
|
32
|
+
|
|
33
|
+
## Examples
|
|
34
|
+
|
|
35
|
+
### Finding the struct elem of an MCID on a page
|
|
36
|
+
|
|
37
|
+
```js
|
|
38
|
+
const pt = runtime.resolve('pdfParentTree');
|
|
39
|
+
const page = doc.pages[0];
|
|
40
|
+
const parents = pt.lookupParent(structRoot.parentTree, page.structParents, resolveRef);
|
|
41
|
+
// parents is an array PdfObject: parents.items[mcid] -> ref to the StructElement
|
|
42
|
+
const elemRef = parents.items[42];
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
### Parent of an annotation
|
|
46
|
+
|
|
47
|
+
```js
|
|
48
|
+
const annotParent = pt.lookupParent(parentTree, annot.structParent, resolveRef);
|
|
49
|
+
// direct ref to the StructElement
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Errors
|
|
53
|
+
|
|
54
|
+
| Code | Class | When |
|
|
55
|
+
|------|--------|------|
|
|
56
|
+
| `pdf/tagged/parent-tree/max-depth` | `ParseError` | Depth exceeds `opts.maxDepth`. |
|
|
57
|
+
| `pdf/tagged/parent-tree/not-dict` | `ParseError` | Node is not a dict. |
|
|
58
|
+
| `pdf/tagged/parent-tree/non-ref-kid` | `ParseError` | `/Kids` contains a non-ref entry. |
|
|
59
|
+
| `pdf/tagged/parent-tree/bad-child` | `ParseError` | Resolved child is not a dict. |
|
|
60
|
+
| `pdf/tagged/parent-tree/empty-node` | `ParseError` | Node has neither `/Kids` nor `/Nums`. |
|
|
61
|
+
| `pdf/tagged/parent-tree/bad-nums` | `ParseError` | `/Nums` is not an even-length array. |
|
|
62
|
+
| `pdf/tagged/parent-tree/bad-key` | `ParseError` | `/Nums` key is not a number. |
|
|
63
|
+
| `pdf/tagged/parent-tree/bad-limits` | `ParseError` | `/Limits` malformed. |
|
|
64
|
+
|
|
65
|
+
## See also
|
|
66
|
+
|
|
67
|
+
- [`pdfStructTree`](./structTree.md) · [`pdfStructElement`](./structElement.md) · [`pdfMarkedContent`](./markedContent.md)
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfRoleMap
|
|
3
|
+
category: pdf/tagged
|
|
4
|
+
dependencies: [pdfErrors, pdfParser]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfRoleMap
|
|
11
|
+
|
|
12
|
+
> RoleMap — ISO 32000-2 §14.7.4 — maps custom roles onto standard structure types.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfRoleMap` | **Source** `packages/front/office/pdf/src/tagged/roleMap.js` | **Deps** `pdfErrors`, `pdfParser` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
`/RoleMap` (on the `StructTreeRoot`) is a dict `name → name`. It lets a custom role (e.g. `MyTitle`) be interpreted as a standard structure type (e.g. `H1`). Resolution is iterative with cycle detection. The module knows the PDF 2.0 standard types (§14.8.4) — `Document`, `DocumentFragment`, `Part`, `Art`, `Sect`, `Div`, `BlockQuote`, `Caption`, `TOC`, `TOCI`, `Index`, `NonStruct`, `Private`, `Aside`, `Title`, `FENote`, `P`, `H`, `H1`–`H7`, `L`, `LI`, `Lbl`, `LBody`, `Table`, `TR`, `TH`, `TD`, `THead`, `TBody`, `TFoot`, `Span`, `Quote`, `Note`, `Reference`, `BibEntry`, `Code`, `Link`, `Annot`, `Em`, `Strong`, `Ruby`, `RB`, `RT`, `RP`, `Warichu`, `WT`, `WP`, `Figure`, `Formula`, `Form`, `Artifact`. It also supports the PDF 2.0 namespace model (`/Namespaces`, per-element `/NS`, and a namespace's own `/RoleMapNS` taking precedence via `resolveStandardType`'s `namespaceMap` argument).
|
|
17
|
+
|
|
18
|
+
## Resolve
|
|
19
|
+
|
|
20
|
+
```js
|
|
21
|
+
const rm = runtime.resolve('pdfRoleMap');
|
|
22
|
+
// Returns: { typeRoleMap, resolveStandardType, isStandardType, STANDARD_TYPES }
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## API
|
|
26
|
+
|
|
27
|
+
| Method | Signature | Returns |
|
|
28
|
+
|--------|-----------|---------|
|
|
29
|
+
| `typeRoleMap` | `(dict) => { map: Record<string,string>, raw }` | Typing — `map` is a **plain object**, not a `Map`. |
|
|
30
|
+
| `resolveStandardType` | `(name, map, namespaceMap?) => { standard: string \| null, chain: string[] }` | Recursively resolves toward a standard type; cycle-safe. |
|
|
31
|
+
| `isStandardType` | `(name) => boolean` | Tests against the §14.8.4 table. |
|
|
32
|
+
| `STANDARD_TYPES` | `Set<string>` | The full standard-type set used by `isStandardType`/`resolveStandardType`. |
|
|
33
|
+
|
|
34
|
+
## Examples
|
|
35
|
+
|
|
36
|
+
### Resolution
|
|
37
|
+
|
|
38
|
+
```js
|
|
39
|
+
const rm = runtime.resolve('pdfRoleMap');
|
|
40
|
+
const r = rm.typeRoleMap(root.roleMap);
|
|
41
|
+
const out = rm.resolveStandardType('MyTitle', r.map);
|
|
42
|
+
out.standard; // 'H1'
|
|
43
|
+
out.chain; // ['MyTitle', 'Heading', 'H1']
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
### PDF/UA validation
|
|
47
|
+
|
|
48
|
+
```js
|
|
49
|
+
for (const [custom, target] of Object.entries(r.map)) {
|
|
50
|
+
void target;
|
|
51
|
+
if (!rm.isStandardType(rm.resolveStandardType(custom, r.map).standard)) {
|
|
52
|
+
/* warning: custom role unresolved */
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Errors
|
|
58
|
+
|
|
59
|
+
| Code | Class | When |
|
|
60
|
+
|------|--------|------|
|
|
61
|
+
| `pdf/tagged/role-map/not-dict` | `ParseError` | Argument is not a dict. |
|
|
62
|
+
| `pdf/tagged/role-map/bad-value` | `ParseError` | A value is not a name. |
|
|
63
|
+
| `pdf/tagged/role-map/cycle` | `ParseError` | Cycle detected while resolving. |
|
|
64
|
+
|
|
65
|
+
## See also
|
|
66
|
+
|
|
67
|
+
- [`pdfStructTree`](./structTree.md) · [`pdfStructElement`](./structElement.md)
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfStructElement
|
|
3
|
+
category: pdf/tagged
|
|
4
|
+
dependencies: [pdfErrors, pdfParser]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfStructElement
|
|
11
|
+
|
|
12
|
+
> Structure element — ISO 32000-2 §14.7.3, the granularity of Tagged PDF §14.8.
|
|
13
|
+
|
|
14
|
+
**Module** `pdfStructElement` | **Source** `packages/front/office/pdf/src/tagged/structElement.js` | **Deps** `pdfErrors`, `pdfParser` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Types a `/Type /StructElem` dict. Fields: `/S` (structure type — required), `/P` (parent), `/ID`, `/Pg` (owning page), `/K` (kids — `int`/`ref`/`dict` or array), `/A`/`/C` (attributes/classes), `/R` (revision), `/T`/`/Lang`/`/Alt`/`/E`/`/ActualText` (accessibility), `/AF` (PDF 2.0 associated files), `/NS` (namespace), `/PhoneticAlphabet`/`/Phoneme`. Kids are normalized to records `{ kind: 'mcid'|'elem'|'mcr'|'objr', … }`.
|
|
17
|
+
|
|
18
|
+
## Resolve
|
|
19
|
+
|
|
20
|
+
```js
|
|
21
|
+
const se = runtime.resolve('pdfStructElement');
|
|
22
|
+
// Returns: { typeStructElement }
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## API
|
|
26
|
+
|
|
27
|
+
| Method | Signature | Returns |
|
|
28
|
+
|--------|-----------|---------|
|
|
29
|
+
| `typeStructElement` | `(dict) => StructElement` | Typing (kids are not recursively typed — each `kind: 'elem'` kid still carries a raw `ref`, resolved and re-typed by the caller). |
|
|
30
|
+
|
|
31
|
+
### Shape of a kid
|
|
32
|
+
|
|
33
|
+
```js
|
|
34
|
+
{ kind: 'mcid', mcid: number }
|
|
35
|
+
{ kind: 'elem', ref: { type:'ref', num, gen } }
|
|
36
|
+
{ kind: 'mcr', pg?, stm?, stmOwn?, mcid: number|null, raw } // /Type /MCR
|
|
37
|
+
{ kind: 'objr', pg?, obj?, raw } // /Type /OBJR
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Examples
|
|
41
|
+
|
|
42
|
+
### Recursive walk
|
|
43
|
+
|
|
44
|
+
```js
|
|
45
|
+
function walk(elDict, resolveRef, depth = 0) {
|
|
46
|
+
const e = runtime.resolve('pdfStructElement').typeStructElement(elDict);
|
|
47
|
+
console.log(' '.repeat(depth) + e.s + (e.alt ? ` [${e.alt}]` : ''));
|
|
48
|
+
for (const k of e.k) {
|
|
49
|
+
if (k.kind === 'elem') walk(resolveRef(k.ref), resolveRef, depth + 1);
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### Accessibility
|
|
55
|
+
|
|
56
|
+
```js
|
|
57
|
+
const el = runtime.resolve('pdfStructElement').typeStructElement(dict);
|
|
58
|
+
el.s; // 'Figure'
|
|
59
|
+
el.alt; // alt-text (required for PDF/UA)
|
|
60
|
+
el.actualText;
|
|
61
|
+
el.lang; // 'fr-FR'
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Errors
|
|
65
|
+
|
|
66
|
+
| Code | Class | When |
|
|
67
|
+
|------|--------|------|
|
|
68
|
+
| `pdf/tagged/struct-elem/not-dict` | `ParseError` | Argument is not a dict. |
|
|
69
|
+
| `pdf/tagged/struct-elem/bad-type` | `ParseError` | `/Type` present and ≠ `/StructElem`. |
|
|
70
|
+
| `pdf/tagged/struct-elem/missing-s` | `ParseError` | `/S` missing or not a name. |
|
|
71
|
+
| `pdf/tagged/struct-elem/bad-kid` | `ParseError` | Kid of an unhandled shape. |
|
|
72
|
+
|
|
73
|
+
## See also
|
|
74
|
+
|
|
75
|
+
- [`pdfStructTree`](./structTree.md) · [`pdfMarkedContent`](./markedContent.md)
|
|
76
|
+
- [`pdfRoleMap`](./roleMap.md) — resolution of custom `/S` values.
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
---
|
|
2
|
+
module: pdfStructTree
|
|
3
|
+
category: pdf/tagged
|
|
4
|
+
dependencies: [pdfErrors, pdfParser]
|
|
5
|
+
returns: object
|
|
6
|
+
worker-safe: true
|
|
7
|
+
status: complete
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# pdfStructTree
|
|
11
|
+
|
|
12
|
+
> Structure Tree Root — ISO 32000-2 §14.7.2, basis of PDF/UA-2 (ISO 14289-2).
|
|
13
|
+
|
|
14
|
+
**Module** `pdfStructTree` | **Source** `packages/front/office/pdf/src/tagged/structTree.js` | **Deps** `pdfErrors`, `pdfParser` | **Worker-safe** yes
|
|
15
|
+
|
|
16
|
+
Types the `/Type /StructTreeRoot` dict referenced by the Catalog. Carries `/K` (root kid(s)), `/IDTree` (ID name tree), `/ParentTree` (number tree mapping MCIDs and OBJRs), `/ParentTreeNextKey`, `/RoleMap`, `/ClassMap`, `/Namespaces` (PDF 2.0), `/AF`, `/PronunciationLexicon`. `/K` is normalized to `Array<ref|dict>`.
|
|
17
|
+
|
|
18
|
+
## Resolve
|
|
19
|
+
|
|
20
|
+
```js
|
|
21
|
+
const st = runtime.resolve('pdfStructTree');
|
|
22
|
+
// Returns: { typeStructTreeRoot }
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## API
|
|
26
|
+
|
|
27
|
+
| Method | Signature | Returns |
|
|
28
|
+
|--------|-----------|---------|
|
|
29
|
+
| `typeStructTreeRoot` | `(dict) => StructTreeRoot` | Typing of the root. |
|
|
30
|
+
|
|
31
|
+
### Shape
|
|
32
|
+
|
|
33
|
+
```js
|
|
34
|
+
{
|
|
35
|
+
kids: Array<ref|dict>,
|
|
36
|
+
idTree, parentTree,
|
|
37
|
+
parentTreeNextKey: number,
|
|
38
|
+
roleMap, classMap, namespaces, af, pronunciationLexicon,
|
|
39
|
+
raw, _extras
|
|
40
|
+
}
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Examples
|
|
44
|
+
|
|
45
|
+
### Walking from the Catalog
|
|
46
|
+
|
|
47
|
+
```js
|
|
48
|
+
const st = runtime.resolve('pdfStructTree');
|
|
49
|
+
const root = st.typeStructTreeRoot(doc._raw.resolve(doc.catalog.structTreeRoot));
|
|
50
|
+
for (const kid of root.kids) {
|
|
51
|
+
const el = runtime.resolve('pdfStructElement').typeStructElement(
|
|
52
|
+
kid.type === 'ref' ? doc._raw.resolve(kid) : kid
|
|
53
|
+
);
|
|
54
|
+
}
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
### Mapping custom roles
|
|
58
|
+
|
|
59
|
+
```js
|
|
60
|
+
const rm = runtime.resolve('pdfRoleMap').typeRoleMap(root.roleMap);
|
|
61
|
+
rm.map.MyCustomP; // -> 'P'
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Errors
|
|
65
|
+
|
|
66
|
+
| Code | Class | When |
|
|
67
|
+
|------|--------|------|
|
|
68
|
+
| `pdf/tagged/struct-tree/not-dict` | `ParseError` | Argument is not a dict. |
|
|
69
|
+
| `pdf/tagged/struct-tree/bad-type` | `ParseError` | `/Type` ≠ `/StructTreeRoot`. |
|
|
70
|
+
| `pdf/tagged/struct-tree/bad-kid` | `ParseError` | `/K` is neither ref/dict, or an array with an invalid entry. |
|
|
71
|
+
|
|
72
|
+
## See also
|
|
73
|
+
|
|
74
|
+
- [`pdfStructElement`](./structElement.md) · [`pdfRoleMap`](./roleMap.md) · [`pdfParentTree`](./parentTree.md) · [`pdfClassMap`](./classMap.md)
|
|
75
|
+
- [`pdfMarkedContent`](./markedContent.md)
|