@awacloud/pdf 0.0.0-stage → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +609 -0
- package/LICENSE +661 -0
- package/NOTICE +77 -0
- package/README.md +363 -2
- package/dist/build/index.js +21 -0
- package/dist/build/pdf-full-rw.js +10972 -0
- package/dist/build/pdf-full-rw.meta.json +105 -0
- package/dist/build/pdf-full-rw.min.js +53 -0
- package/dist/build/pdf-full.js +6078 -0
- package/dist/build/pdf-full.meta.json +90 -0
- package/dist/build/pdf-full.min.js +32 -0
- package/dist/build/pdf-large-rw.js +10367 -0
- package/dist/build/pdf-large-rw.meta.json +99 -0
- package/dist/build/pdf-large-rw.min.js +53 -0
- package/dist/build/pdf-large.js +5473 -0
- package/dist/build/pdf-large.meta.json +84 -0
- package/dist/build/pdf-large.min.js +32 -0
- package/dist/build/pdf-legacy-rw.js +12402 -0
- package/dist/build/pdf-legacy-rw.meta.json +110 -0
- package/dist/build/pdf-legacy-rw.min.js +53 -0
- package/dist/build/pdf-legacy.js +7508 -0
- package/dist/build/pdf-legacy.meta.json +95 -0
- package/dist/build/pdf-legacy.min.js +32 -0
- package/dist/build/pdf-rw.js +7578 -0
- package/dist/build/pdf-rw.meta.json +77 -0
- package/dist/build/pdf-rw.min.js +53 -0
- package/dist/build/pdf.js +2684 -0
- package/dist/build/pdf.meta.json +62 -0
- package/dist/build/pdf.min.js +32 -0
- package/dist/standalone/pdf-full-rw.js +16798 -0
- package/dist/standalone/pdf-full-rw.meta.json +78 -0
- package/dist/standalone/pdf-full-rw.min.js +56 -0
- package/dist/standalone/pdf-full.js +11904 -0
- package/dist/standalone/pdf-full.meta.json +63 -0
- package/dist/standalone/pdf-full.min.js +35 -0
- package/dist/standalone/pdf-large-rw.js +16193 -0
- package/dist/standalone/pdf-large-rw.meta.json +72 -0
- package/dist/standalone/pdf-large-rw.min.js +56 -0
- package/dist/standalone/pdf-large.js +11299 -0
- package/dist/standalone/pdf-large.meta.json +57 -0
- package/dist/standalone/pdf-large.min.js +35 -0
- package/dist/standalone/pdf-legacy-rw.js +18228 -0
- package/dist/standalone/pdf-legacy-rw.meta.json +83 -0
- package/dist/standalone/pdf-legacy-rw.min.js +56 -0
- package/dist/standalone/pdf-legacy.js +13334 -0
- package/dist/standalone/pdf-legacy.meta.json +68 -0
- package/dist/standalone/pdf-legacy.min.js +35 -0
- package/dist/standalone/pdf-rw.js +13404 -0
- package/dist/standalone/pdf-rw.meta.json +50 -0
- package/dist/standalone/pdf-rw.min.js +56 -0
- package/dist/standalone/pdf.js +8510 -0
- package/dist/standalone/pdf.meta.json +35 -0
- package/dist/standalone/pdf.min.js +35 -0
- package/docs/README.md +53 -0
- package/docs/api/README.md +38 -0
- package/docs/api/_shared/README.md +91 -0
- package/docs/api/action/README.md +29 -0
- package/docs/api/action/action.md +81 -0
- package/docs/api/action/goTo.md +66 -0
- package/docs/api/action/launch.md +58 -0
- package/docs/api/action/named.md +55 -0
- package/docs/api/action/uri.md +54 -0
- package/docs/api/annot/README.md +53 -0
- package/docs/api/annot/annot.md +114 -0
- package/docs/api/annot/fileAttach.md +53 -0
- package/docs/api/annot/freeText.md +68 -0
- package/docs/api/annot/ink.md +69 -0
- package/docs/api/annot/link.md +74 -0
- package/docs/api/annot/markup.md +83 -0
- package/docs/api/annot/popup.md +52 -0
- package/docs/api/annot/projection.md +56 -0
- package/docs/api/annot/redact.md +67 -0
- package/docs/api/annot/square.md +87 -0
- package/docs/api/annot/stamp.md +54 -0
- package/docs/api/annot/text.md +69 -0
- package/docs/api/annot/widget.md +69 -0
- package/docs/api/associatedFiles/README.md +9 -0
- package/docs/api/associatedFiles/associatedFiles.md +78 -0
- package/docs/api/bundles/README.md +68 -0
- package/docs/api/bundles/dist-matrix.md +165 -0
- package/docs/api/bundles/pdf-full.md +148 -0
- package/docs/api/bundles/pdf-large.md +144 -0
- package/docs/api/bundles/pdf-legacy.md +169 -0
- package/docs/api/content/README.md +29 -0
- package/docs/api/content/color.md +99 -0
- package/docs/api/content/graphics.md +114 -0
- package/docs/api/content/images.md +124 -0
- package/docs/api/content/ops.md +100 -0
- package/docs/api/content/stream.md +107 -0
- package/docs/api/content/text.md +98 -0
- package/docs/api/crypto/README.md +29 -0
- package/docs/api/crypto/aesGcm.md +72 -0
- package/docs/api/crypto/permissions.md +79 -0
- package/docs/api/crypto/security.md +98 -0
- package/docs/api/crypto/standardV4.md +104 -0
- package/docs/api/crypto/standardV5.md +84 -0
- package/docs/api/crypto/standardV6.md +93 -0
- package/docs/api/destination/README.md +9 -0
- package/docs/api/destination/destination.md +79 -0
- package/docs/api/document/README.md +29 -0
- package/docs/api/document/builder.md +281 -0
- package/docs/api/document/catalog.md +98 -0
- package/docs/api/document/document.md +187 -0
- package/docs/api/document/encryptedWriter.md +149 -0
- package/docs/api/document/incrementalWriter.md +148 -0
- package/docs/api/document/page.md +99 -0
- package/docs/api/document/pages.md +82 -0
- package/docs/api/document/resources.md +102 -0
- package/docs/api/document/writer.md +157 -0
- package/docs/api/document/xrefStreamWriter.md +122 -0
- package/docs/api/embedded/README.md +13 -0
- package/docs/api/embedded/collection.md +80 -0
- package/docs/api/embedded/embeddedFile.md +86 -0
- package/docs/api/embedded/fileSpec.md +87 -0
- package/docs/api/errors.md +110 -0
- package/docs/api/extra/3d-richmedia.md +76 -0
- package/docs/api/extra/README.md +99 -0
- package/docs/api/extra/annot-extended.md +71 -0
- package/docs/api/extra/associated-files.md +70 -0
- package/docs/api/extra/ccitt-fax-decoder.md +74 -0
- package/docs/api/extra/color-spaces-extended.md +72 -0
- package/docs/api/extra/content-ops-extended.md +82 -0
- package/docs/api/extra/document-parts.md +69 -0
- package/docs/api/extra/embedded-files-portfolio.md +87 -0
- package/docs/api/extra/font-cid-typed.md +77 -0
- package/docs/api/extra/font-color-tagging.md +76 -0
- package/docs/api/extra/form-actions-extended.md +75 -0
- package/docs/api/extra/info-dict-deprecated.md +72 -0
- package/docs/api/extra/jbig2-read.md +80 -0
- package/docs/api/extra/legacy-deprecated-annots.md +89 -0
- package/docs/api/extra/legacy-deprecated-filters.md +78 -0
- package/docs/api/extra/legacy-rc4-read.md +74 -0
- package/docs/api/extra/legacy-xfa-read.md +65 -0
- package/docs/api/extra/linearization-write.md +71 -0
- package/docs/api/extra/misc.md +93 -0
- package/docs/api/extra/optional-content-extended.md +83 -0
- package/docs/api/extra/pdf-a-output-intent.md +65 -0
- package/docs/api/extra/pdf-sandbox.md +76 -0
- package/docs/api/extra/pdf-ua-tagged.md +63 -0
- package/docs/api/extra/pdf-x-prepress.md +65 -0
- package/docs/api/extra/redaction-iso32005.md +65 -0
- package/docs/api/extra/shading-typed.md +73 -0
- package/docs/api/extra/sig-aes-gcm.md +69 -0
- package/docs/api/extra/sig-pades.md +103 -0
- package/docs/api/extra/tagged-pdf-typed.md +78 -0
- package/docs/api/extra/transparency-typed.md +74 -0
- package/docs/api/extra/well-tagged-pdf.md +61 -0
- package/docs/api/extra/xmp-extended.md +65 -0
- package/docs/api/font/README.md +25 -0
- package/docs/api/font/embed.md +157 -0
- package/docs/api/font/encoding.md +95 -0
- package/docs/api/font/font.md +97 -0
- package/docs/api/font/type3.md +89 -0
- package/docs/api/form/README.md +35 -0
- package/docs/api/form/acroform.md +88 -0
- package/docs/api/form/appearance.md +87 -0
- package/docs/api/form/button.md +97 -0
- package/docs/api/form/choice.md +96 -0
- package/docs/api/form/fieldTree.md +93 -0
- package/docs/api/form/signature.md +90 -0
- package/docs/api/form/text.md +88 -0
- package/docs/api/linearization/README.md +11 -0
- package/docs/api/linearization/linearization.md +81 -0
- package/docs/api/main.md +116 -0
- package/docs/api/metadata/README.md +10 -0
- package/docs/api/metadata/info.md +70 -0
- package/docs/api/metadata/xmp.md +62 -0
- package/docs/api/ocg/README.md +23 -0
- package/docs/api/ocg/config.md +95 -0
- package/docs/api/ocg/ocg.md +77 -0
- package/docs/api/outline/README.md +11 -0
- package/docs/api/outline/outline.md +107 -0
- package/docs/api/pdf.md +152 -0
- package/docs/api/prepress/README.md +10 -0
- package/docs/api/prepress/outputIntent.md +79 -0
- package/docs/api/prepress/pageBoundary.md +75 -0
- package/docs/api/sig/README.md +32 -0
- package/docs/api/sig/byteRange.md +120 -0
- package/docs/api/sig/certChain.md +84 -0
- package/docs/api/sig/dss.md +111 -0
- package/docs/api/sig/oids.md +76 -0
- package/docs/api/sig/sha1.md +72 -0
- package/docs/api/sig/sign.md +317 -0
- package/docs/api/sig/signature.md +178 -0
- package/docs/api/sig/timestamp.md +84 -0
- package/docs/api/syntax/README.md +29 -0
- package/docs/api/syntax/crossRefStream.md +115 -0
- package/docs/api/syntax/filters/README.md +50 -0
- package/docs/api/syntax/filters/ascii85.md +76 -0
- package/docs/api/syntax/filters/asciiHex.md +73 -0
- package/docs/api/syntax/filters/dispatch.md +125 -0
- package/docs/api/syntax/filters/flate.md +134 -0
- package/docs/api/syntax/filters/runLength.md +78 -0
- package/docs/api/syntax/objStream.md +88 -0
- package/docs/api/syntax/parser-obj.md +97 -0
- package/docs/api/syntax/parser.md +151 -0
- package/docs/api/syntax/serializer.md +109 -0
- package/docs/api/syntax/tokenizer.md +104 -0
- package/docs/api/syntax/trailer.md +85 -0
- package/docs/api/syntax/xref.md +139 -0
- package/docs/api/tagged/README.md +25 -0
- package/docs/api/tagged/classMap.md +67 -0
- package/docs/api/tagged/markedContent.md +62 -0
- package/docs/api/tagged/parentTree.md +67 -0
- package/docs/api/tagged/roleMap.md +67 -0
- package/docs/api/tagged/structElement.md +76 -0
- package/docs/api/tagged/structTree.md +75 -0
- package/docs/guide/coverage.md +113 -0
- package/docs/guide/crypto.md +121 -0
- package/docs/guide/extending.md +76 -0
- package/docs/guide/getting-started.md +75 -0
- package/docs/guide/legacy-1.7.md +42 -0
- package/docs/guide/pades-integration.md +579 -0
- package/docs/guide/read-pdf.md +89 -0
- package/package.json +97 -4
- package/src/_shared/index.js +179 -0
- package/src/action/action.js +119 -0
- package/src/action/goTo.js +89 -0
- package/src/action/launch.js +61 -0
- package/src/action/named.js +54 -0
- package/src/action/uri.js +51 -0
- package/src/annot/annot.js +212 -0
- package/src/annot/fileAttach.js +55 -0
- package/src/annot/freeText.js +82 -0
- package/src/annot/ink.js +77 -0
- package/src/annot/link.js +77 -0
- package/src/annot/markup.js +91 -0
- package/src/annot/popup.js +53 -0
- package/src/annot/projection.js +52 -0
- package/src/annot/redact.js +87 -0
- package/src/annot/square.js +132 -0
- package/src/annot/stamp.js +48 -0
- package/src/annot/text.js +54 -0
- package/src/annot/widget.js +61 -0
- package/src/associatedFiles/associatedFiles.js +86 -0
- package/src/bundles/pdf-full.js +91 -0
- package/src/bundles/pdf-large.js +81 -0
- package/src/bundles/pdf-legacy.js +107 -0
- package/src/content/color.js +114 -0
- package/src/content/graphics.js +192 -0
- package/src/content/images.js +160 -0
- package/src/content/ops.js +137 -0
- package/src/content/stream.js +154 -0
- package/src/content/text.js +125 -0
- package/src/crypto/aesGcm.js +123 -0
- package/src/crypto/permissions.js +112 -0
- package/src/crypto/security.js +327 -0
- package/src/crypto/standardV4.js +443 -0
- package/src/crypto/standardV5.js +306 -0
- package/src/crypto/standardV6.js +334 -0
- package/src/destination/destination.js +183 -0
- package/src/document/builder.js +618 -0
- package/src/document/catalog.js +100 -0
- package/src/document/document.js +472 -0
- package/src/document/encryptedWriter.js +554 -0
- package/src/document/incrementalWriter.js +514 -0
- package/src/document/page.js +131 -0
- package/src/document/pages.js +103 -0
- package/src/document/resources.js +146 -0
- package/src/document/writer.js +211 -0
- package/src/document/xrefStreamWriter.js +353 -0
- package/src/embedded/collection.js +102 -0
- package/src/embedded/embeddedFile.js +99 -0
- package/src/embedded/fileSpec.js +137 -0
- package/src/errors.js +78 -0
- package/src/extra/3d-richmedia.js +171 -0
- package/src/extra/annot-extended.js +200 -0
- package/src/extra/associated-files.js +131 -0
- package/src/extra/ccitt-fax-decoder.js +776 -0
- package/src/extra/color-spaces-extended.js +196 -0
- package/src/extra/content-ops-extended.js +153 -0
- package/src/extra/document-parts.js +149 -0
- package/src/extra/embedded-files-portfolio.js +234 -0
- package/src/extra/font-cid-typed.js +185 -0
- package/src/extra/font-color-tagging.js +144 -0
- package/src/extra/form-actions-extended.js +196 -0
- package/src/extra/info-dict-deprecated.js +137 -0
- package/src/extra/jbig2-read.js +169 -0
- package/src/extra/legacy-deprecated-annots.js +198 -0
- package/src/extra/legacy-deprecated-filters.js +167 -0
- package/src/extra/legacy-rc4-read.js +235 -0
- package/src/extra/legacy-xfa-read.js +104 -0
- package/src/extra/linearization-write.js +97 -0
- package/src/extra/misc.js +217 -0
- package/src/extra/optional-content-extended.js +142 -0
- package/src/extra/pdf-a-output-intent.js +112 -0
- package/src/extra/pdf-sandbox.js +88 -0
- package/src/extra/pdf-ua-tagged.js +116 -0
- package/src/extra/pdf-x-prepress.js +114 -0
- package/src/extra/redaction-iso32005.js +136 -0
- package/src/extra/shading-typed.js +222 -0
- package/src/extra/sig-aes-gcm.js +135 -0
- package/src/extra/sig-pades.js +242 -0
- package/src/extra/tagged-pdf-typed.js +203 -0
- package/src/extra/transparency-typed.js +135 -0
- package/src/extra/well-tagged-pdf.js +138 -0
- package/src/extra/xmp-extended.js +190 -0
- package/src/font/embed.js +480 -0
- package/src/font/encoding.js +92 -0
- package/src/font/font.js +101 -0
- package/src/font/type3.js +75 -0
- package/src/form/acroform.js +94 -0
- package/src/form/appearance.js +90 -0
- package/src/form/button.js +105 -0
- package/src/form/choice.js +152 -0
- package/src/form/fieldTree.js +120 -0
- package/src/form/signature.js +100 -0
- package/src/form/text.js +101 -0
- package/src/linearization/linearization.js +107 -0
- package/src/main.js +411 -0
- package/src/metadata/info.js +87 -0
- package/src/metadata/xmp.js +62 -0
- package/src/ocg/config.js +156 -0
- package/src/ocg/ocg.js +124 -0
- package/src/outline/outline.js +157 -0
- package/src/pdf.js +133 -0
- package/src/prepress/outputIntent.js +118 -0
- package/src/prepress/pageBoundary.js +108 -0
- package/src/sig/byteRange.js +306 -0
- package/src/sig/certChain.js +247 -0
- package/src/sig/dss.js +317 -0
- package/src/sig/oids.js +157 -0
- package/src/sig/sha1.js +142 -0
- package/src/sig/sign.js +1899 -0
- package/src/sig/signature.js +1441 -0
- package/src/sig/timestamp.js +236 -0
- package/src/syntax/crossRefStream.js +133 -0
- package/src/syntax/filters/ascii85.js +122 -0
- package/src/syntax/filters/asciiHex.js +83 -0
- package/src/syntax/filters/dispatch.js +176 -0
- package/src/syntax/filters/flate.js +316 -0
- package/src/syntax/filters/runLength.js +96 -0
- package/src/syntax/objStream.js +99 -0
- package/src/syntax/parser-obj.js +52 -0
- package/src/syntax/parser.js +321 -0
- package/src/syntax/serializer.js +221 -0
- package/src/syntax/tokenizer.js +290 -0
- package/src/syntax/trailer.js +76 -0
- package/src/syntax/xref.js +341 -0
- package/src/tagged/classMap.js +81 -0
- package/src/tagged/markedContent.js +123 -0
- package/src/tagged/parentTree.js +126 -0
- package/src/tagged/roleMap.js +107 -0
- package/src/tagged/structElement.js +138 -0
- package/src/tagged/structTree.js +94 -0
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
// Copyright (c) 2026 AwaCloud SAS
|
|
2
|
+
// Author: Matthieu Bouilloux
|
|
3
|
+
// SPDX-License-Identifier: AGPL-3.0-only
|
|
4
|
+
// Dual-licensed; see the NOTICE file for licensing and any additional terms.
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* @fileoverview Classical PDF cross-reference table parser per ISO
|
|
8
|
+
* 32000-2:2020 §7.5.4, plus the two cross-reference-stream helpers the
|
|
9
|
+
* incremental writer needs (§7.5.8): `readXrefStreamDict` reads the
|
|
10
|
+
* dictionary of a `/Type /XRef` stream section WITHOUT decoding its data,
|
|
11
|
+
* and `buildXrefStream` emits an uncompressed `/Type /XRef` stream section
|
|
12
|
+
* for an incremental update over a stream base.
|
|
13
|
+
*
|
|
14
|
+
* @module pdf/syntax/xref
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Module factory — worker-safe, self-contained.
|
|
19
|
+
*/
|
|
20
|
+
import { pdfErrors } from '../errors.js';
|
|
21
|
+
import { pdfTokenizer } from './tokenizer.js';
|
|
22
|
+
import { pdfParser } from './parser.js';
|
|
23
|
+
|
|
24
|
+
export const pdfXref = {
|
|
25
|
+
name: 'pdfXref',
|
|
26
|
+
dependencies: ['pdfErrors', 'pdfTokenizer', 'pdfParser'],
|
|
27
|
+
deps: [pdfErrors, pdfTokenizer, pdfParser],
|
|
28
|
+
factory(errors, tokenizerMod, parserMod) {
|
|
29
|
+
const { ParseError, RenderError } = errors;
|
|
30
|
+
const tokenize = tokenizerMod.tokenize;
|
|
31
|
+
const lastIndexOfBytes = tokenizerMod.lastIndexOfBytes;
|
|
32
|
+
const parseObject = parserMod.parseObject;
|
|
33
|
+
const LF = 0x0A, CR = 0x0D, SP = 0x20;
|
|
34
|
+
|
|
35
|
+
function locateStartXref(bytes) {
|
|
36
|
+
const tail = Math.max(0, bytes.length - 8192);
|
|
37
|
+
const idx = lastIndexOfBytes(bytes,
|
|
38
|
+
new Uint8Array([0x73, 0x74, 0x61, 0x72, 0x74, 0x78, 0x72, 0x65, 0x66]),
|
|
39
|
+
bytes.length);
|
|
40
|
+
if (idx < tail) return -1;
|
|
41
|
+
return idx;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function readStartXref(bytes, at) {
|
|
45
|
+
const tok = tokenize(bytes, { start: at });
|
|
46
|
+
const kw = tok.next();
|
|
47
|
+
if (!kw || kw.kind !== 'kw' || kw.value !== 'startxref') {
|
|
48
|
+
throw new ParseError('pdf/xref/no-startxref',
|
|
49
|
+
'startxref keyword not found',
|
|
50
|
+
{ context: { offset: at } });
|
|
51
|
+
}
|
|
52
|
+
const num = tok.next();
|
|
53
|
+
if (!num || num.kind !== 'int' || num.value < 0) {
|
|
54
|
+
throw new ParseError('pdf/xref/bad-startxref',
|
|
55
|
+
'startxref must be followed by a non-negative integer',
|
|
56
|
+
{ context: { offset: at } });
|
|
57
|
+
}
|
|
58
|
+
return num.value;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function parseIntBytes(bytes, off, len) {
|
|
62
|
+
let n = 0;
|
|
63
|
+
for (let i = 0; i < len; i++) {
|
|
64
|
+
const b = bytes[off + i];
|
|
65
|
+
if (b < 0x30 || b > 0x39) {
|
|
66
|
+
throw new ParseError('pdf/xref/bad-digit',
|
|
67
|
+
'expected ASCII digit in xref entry',
|
|
68
|
+
{ context: { offset: off + i, byte: b } });
|
|
69
|
+
}
|
|
70
|
+
n = n * 10 + (b - 0x30);
|
|
71
|
+
}
|
|
72
|
+
return n;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function parseXrefTable(bytes, at) {
|
|
76
|
+
const tok = tokenize(bytes, { start: at });
|
|
77
|
+
const kw = tok.next();
|
|
78
|
+
if (!kw || kw.kind !== 'kw' || kw.value !== 'xref') {
|
|
79
|
+
throw new ParseError('pdf/xref/no-xref-keyword',
|
|
80
|
+
'expected xref keyword',
|
|
81
|
+
{ context: { offset: at } });
|
|
82
|
+
}
|
|
83
|
+
const entries = {};
|
|
84
|
+
for (;;) {
|
|
85
|
+
const first = tok.peek();
|
|
86
|
+
if (!first || first.kind !== 'int') {
|
|
87
|
+
if (first) tok.seek(first.offset);
|
|
88
|
+
break;
|
|
89
|
+
}
|
|
90
|
+
tok.next();
|
|
91
|
+
const cnt = tok.next();
|
|
92
|
+
if (!cnt || cnt.kind !== 'int' || cnt.value < 0) {
|
|
93
|
+
throw new ParseError('pdf/xref/bad-subsection-header',
|
|
94
|
+
'xref subsection header must be `<first> <count>`',
|
|
95
|
+
{ context: { offset: first.offset } });
|
|
96
|
+
}
|
|
97
|
+
let p = tok.pos();
|
|
98
|
+
if (p < bytes.length && bytes[p] === CR) p++;
|
|
99
|
+
if (p < bytes.length && bytes[p] === LF) p++;
|
|
100
|
+
for (let k = 0; k < cnt.value; k++) {
|
|
101
|
+
if (p + 20 > bytes.length) {
|
|
102
|
+
throw new ParseError('pdf/xref/truncated-entry',
|
|
103
|
+
'truncated xref entry',
|
|
104
|
+
{ context: { offset: p, expected: 20 } });
|
|
105
|
+
}
|
|
106
|
+
const off = parseIntBytes(bytes, p, 10);
|
|
107
|
+
if (bytes[p + 10] !== SP) {
|
|
108
|
+
throw new ParseError('pdf/xref/bad-entry-format',
|
|
109
|
+
'xref entry missing space at col 10',
|
|
110
|
+
{ context: { offset: p } });
|
|
111
|
+
}
|
|
112
|
+
const gen = parseIntBytes(bytes, p + 11, 5);
|
|
113
|
+
if (bytes[p + 16] !== SP) {
|
|
114
|
+
throw new ParseError('pdf/xref/bad-entry-format',
|
|
115
|
+
'xref entry missing space at col 16',
|
|
116
|
+
{ context: { offset: p } });
|
|
117
|
+
}
|
|
118
|
+
const tag = bytes[p + 17];
|
|
119
|
+
if (tag !== 0x6E && tag !== 0x66) {
|
|
120
|
+
throw new ParseError('pdf/xref/bad-entry-flag',
|
|
121
|
+
'xref entry flag must be n or f',
|
|
122
|
+
{ context: { offset: p, byte: tag } });
|
|
123
|
+
}
|
|
124
|
+
entries[first.value + k] = {
|
|
125
|
+
offset: off,
|
|
126
|
+
gen,
|
|
127
|
+
free: tag === 0x66
|
|
128
|
+
};
|
|
129
|
+
p += 20;
|
|
130
|
+
}
|
|
131
|
+
tok.seek(p);
|
|
132
|
+
}
|
|
133
|
+
return { entries, end: tok.pos() };
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
function parseTrailerDict(bytes, at) {
|
|
137
|
+
const tok = tokenize(bytes, { start: at });
|
|
138
|
+
const kw = tok.next();
|
|
139
|
+
if (!kw || kw.kind !== 'kw' || kw.value !== 'trailer') {
|
|
140
|
+
throw new ParseError('pdf/xref/no-trailer',
|
|
141
|
+
'expected trailer keyword',
|
|
142
|
+
{ context: { offset: at } });
|
|
143
|
+
}
|
|
144
|
+
const dict = parseObject(tok);
|
|
145
|
+
if (dict.type !== 'dict') {
|
|
146
|
+
throw new ParseError('pdf/xref/trailer-not-dict',
|
|
147
|
+
'trailer must be a dictionary',
|
|
148
|
+
{ context: { offset: at } });
|
|
149
|
+
}
|
|
150
|
+
return { dict, end: tok.pos() };
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Read the dictionary of the cross-reference stream section at `at`
|
|
155
|
+
* (`N G obj << /Type /XRef … >> stream`, §7.5.8). The stream data is
|
|
156
|
+
* NOT decoded — the dict alone carries the section's trailer keys
|
|
157
|
+
* (`/Root`, `/Info`, `/ID`, `/Size`, `/Prev`) — so no filter module
|
|
158
|
+
* is needed and an indirect `/Length` is not an obstacle here.
|
|
159
|
+
*
|
|
160
|
+
* @param {Uint8Array} bytes
|
|
161
|
+
* @param {number} at Offset designated by `startxref` or a `/Prev`.
|
|
162
|
+
* @returns {{num: number, gen: number, dict: object}}
|
|
163
|
+
* @throws ParseError `pdf/xref/not-xref-stream` — no indirect object
|
|
164
|
+
* at `at`, or one that is not a `/Type /XRef` stream.
|
|
165
|
+
*/
|
|
166
|
+
function readXrefStreamDict(bytes, at) {
|
|
167
|
+
const notStream = (why) => new ParseError('pdf/xref/not-xref-stream',
|
|
168
|
+
`no cross-reference stream at offset ${at}: ${why}`,
|
|
169
|
+
{ context: { offset: at } });
|
|
170
|
+
if (!Number.isInteger(at) || at < 0 || at >= bytes.length) {
|
|
171
|
+
throw notStream('offset outside the file');
|
|
172
|
+
}
|
|
173
|
+
let num, gen, kw, dict, after;
|
|
174
|
+
try {
|
|
175
|
+
const tok = tokenize(bytes, { start: at });
|
|
176
|
+
num = tok.next();
|
|
177
|
+
gen = tok.next();
|
|
178
|
+
kw = tok.next();
|
|
179
|
+
if (!num || num.kind !== 'int' || !gen || gen.kind !== 'int'
|
|
180
|
+
|| !kw || kw.kind !== 'kw' || kw.value !== 'obj') {
|
|
181
|
+
throw notStream('no indirect object header');
|
|
182
|
+
}
|
|
183
|
+
dict = parseObject(tok);
|
|
184
|
+
after = tok.peek();
|
|
185
|
+
} catch (e) {
|
|
186
|
+
if (e && e.code === 'pdf/xref/not-xref-stream') throw e;
|
|
187
|
+
throw notStream('the object does not parse');
|
|
188
|
+
}
|
|
189
|
+
const type = dict && dict.type === 'dict' && dict.entries.Type;
|
|
190
|
+
if (!type || type.type !== 'name' || type.value !== 'XRef'
|
|
191
|
+
|| !after || after.kind !== 'kw' || after.value !== 'stream') {
|
|
192
|
+
throw notStream('the object is not a /Type /XRef stream');
|
|
193
|
+
}
|
|
194
|
+
return { num: num.value, gen: gen.value, dict };
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** Smallest number of bytes (>= 1) that holds `n` big-endian. */
|
|
198
|
+
function byteWidth(n) {
|
|
199
|
+
let w = 1;
|
|
200
|
+
while (n > 0xFF) { n = Math.floor(n / 256); w++; }
|
|
201
|
+
return w;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
function hexLit(bytes) {
|
|
205
|
+
const H = '0123456789ABCDEF';
|
|
206
|
+
let s = '<';
|
|
207
|
+
for (let i = 0; i < bytes.length; i++) {
|
|
208
|
+
s += H[bytes[i] >> 4] + H[bytes[i] & 0xF];
|
|
209
|
+
}
|
|
210
|
+
return s + '>';
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
function isRef(r) {
|
|
214
|
+
return !!r && Number.isInteger(r.num) && r.num >= 1;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
/**
|
|
218
|
+
* Emit one uncompressed cross-reference stream section for an
|
|
219
|
+
* incremental update (§7.5.8, §7.5.6), as the complete indirect
|
|
220
|
+
* object `num 0 obj << /Type /XRef … >> stream … endstream endobj`.
|
|
221
|
+
* The caller appends `startxref` / `%%EOF`.
|
|
222
|
+
*
|
|
223
|
+
* Rows: object 0's free-list head, one type-1 row per `entries`
|
|
224
|
+
* item, and the stream's own type-1 row at `offset`. `/Index` lists
|
|
225
|
+
* one pair per contiguous run of object numbers; `/W` is the
|
|
226
|
+
* narrowest `[1 w2 w3]` that fits; `/Size` is `num + 1` or
|
|
227
|
+
* `opts.size`, whichever is larger. No `/Filter`: the data is
|
|
228
|
+
* written uncompressed, so no filter module is involved.
|
|
229
|
+
*
|
|
230
|
+
* `opts.encrypt` (an indirect reference) is written as
|
|
231
|
+
* `/Encrypt n g R` right after `/ID`: an update over an encrypted
|
|
232
|
+
* document repeats the document's `/Encrypt` (ISO 32000-2 §7.5.6).
|
|
233
|
+
* Without it the output is unchanged.
|
|
234
|
+
*
|
|
235
|
+
* @param {{num: number, offset: number,
|
|
236
|
+
* entries: Array<{num: number, offset: number, gen?: number}>,
|
|
237
|
+
* size?: number, prev: number,
|
|
238
|
+
* root: {num: number, gen?: number},
|
|
239
|
+
* info?: {num: number, gen?: number}|null,
|
|
240
|
+
* id?: Array<Uint8Array>|null,
|
|
241
|
+
* encrypt?: {num: number, gen?: number}|null}} opts
|
|
242
|
+
* @returns {Uint8Array}
|
|
243
|
+
* @throws RenderError `pdf/xref/bad-stream-section` on an unusable
|
|
244
|
+
* `num`, `offset`, `prev`, `root`, `encrypt` (anything but an
|
|
245
|
+
* indirect reference — a direct `/Encrypt` dictionary is not
|
|
246
|
+
* written here) or entry.
|
|
247
|
+
*/
|
|
248
|
+
function buildXrefStream(opts) {
|
|
249
|
+
const bad = (why) => new RenderError('pdf/xref/bad-stream-section',
|
|
250
|
+
`cannot build the cross-reference stream: ${why}`);
|
|
251
|
+
if (!opts || !Number.isInteger(opts.num) || opts.num < 1) {
|
|
252
|
+
throw bad('num must be an integer >= 1');
|
|
253
|
+
}
|
|
254
|
+
if (!Number.isInteger(opts.offset) || opts.offset < 0) {
|
|
255
|
+
throw bad('offset must be an integer >= 0');
|
|
256
|
+
}
|
|
257
|
+
if (!Number.isInteger(opts.prev) || opts.prev < 0) {
|
|
258
|
+
throw bad('prev must be an integer >= 0');
|
|
259
|
+
}
|
|
260
|
+
if (!isRef(opts.root)) throw bad('root must be { num >= 1, gen }');
|
|
261
|
+
if (opts.encrypt != null && !isRef(opts.encrypt)) {
|
|
262
|
+
throw bad('encrypt must be an indirect reference { num >= 1, gen }');
|
|
263
|
+
}
|
|
264
|
+
const rows = new Map();
|
|
265
|
+
rows.set(0, [0, 0, 65535]);
|
|
266
|
+
for (const e of opts.entries || []) {
|
|
267
|
+
if (!e || !Number.isInteger(e.num) || e.num < 1
|
|
268
|
+
|| !Number.isInteger(e.offset) || e.offset < 0) {
|
|
269
|
+
throw bad('each entry needs num >= 1 and offset >= 0');
|
|
270
|
+
}
|
|
271
|
+
rows.set(e.num, [1, e.offset, e.gen | 0]);
|
|
272
|
+
}
|
|
273
|
+
rows.set(opts.num, [1, opts.offset, 0]);
|
|
274
|
+
|
|
275
|
+
const nums = Array.from(rows.keys()).sort((a, b) => a - b);
|
|
276
|
+
let max2 = 0, max3 = 0;
|
|
277
|
+
for (const r of rows.values()) {
|
|
278
|
+
if (r[1] > max2) max2 = r[1];
|
|
279
|
+
if (r[2] > max3) max3 = r[2];
|
|
280
|
+
}
|
|
281
|
+
const w = [1, byteWidth(max2), byteWidth(max3)];
|
|
282
|
+
const rec = w[0] + w[1] + w[2];
|
|
283
|
+
const data = new Uint8Array(nums.length * rec);
|
|
284
|
+
let o = 0;
|
|
285
|
+
for (const n of nums) {
|
|
286
|
+
const r = rows.get(n);
|
|
287
|
+
let at = o;
|
|
288
|
+
for (let f = 0; f < 3; f++) {
|
|
289
|
+
let v = r[f];
|
|
290
|
+
for (let k = w[f] - 1; k >= 0; k--) {
|
|
291
|
+
data[at + k] = v & 0xFF;
|
|
292
|
+
v = Math.floor(v / 256);
|
|
293
|
+
}
|
|
294
|
+
at += w[f];
|
|
295
|
+
}
|
|
296
|
+
o += rec;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
const index = [];
|
|
300
|
+
for (let i = 0; i < nums.length;) {
|
|
301
|
+
let j = i;
|
|
302
|
+
while (j + 1 < nums.length && nums[j + 1] === nums[j] + 1) j++;
|
|
303
|
+
index.push(`${nums[i]} ${j - i + 1}`);
|
|
304
|
+
i = j + 1;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
const size = Math.max(opts.num + 1,
|
|
308
|
+
Number.isInteger(opts.size) ? opts.size : 0);
|
|
309
|
+
let d = `<< /Type /XRef /Size ${size} /W [${w.join(' ')}]`
|
|
310
|
+
+ ` /Index [${index.join(' ')}]`
|
|
311
|
+
+ ` /Root ${opts.root.num} ${opts.root.gen | 0} R`;
|
|
312
|
+
if (isRef(opts.info)) d += ` /Info ${opts.info.num} ${opts.info.gen | 0} R`;
|
|
313
|
+
if (opts.id && opts.id[0] instanceof Uint8Array
|
|
314
|
+
&& opts.id[1] instanceof Uint8Array) {
|
|
315
|
+
d += ` /ID [${hexLit(opts.id[0])}${hexLit(opts.id[1])}]`;
|
|
316
|
+
}
|
|
317
|
+
if (isRef(opts.encrypt)) {
|
|
318
|
+
d += ` /Encrypt ${opts.encrypt.num} ${opts.encrypt.gen | 0} R`;
|
|
319
|
+
}
|
|
320
|
+
d += ` /Prev ${opts.prev} /Length ${data.length} >>`;
|
|
321
|
+
|
|
322
|
+
const te = new TextEncoder();
|
|
323
|
+
const head = te.encode(`${opts.num} 0 obj\n${d}\nstream\n`);
|
|
324
|
+
const tail = te.encode('\nendstream\nendobj\n');
|
|
325
|
+
const out = new Uint8Array(head.length + data.length + tail.length);
|
|
326
|
+
out.set(head, 0);
|
|
327
|
+
out.set(data, head.length);
|
|
328
|
+
out.set(tail, head.length + data.length);
|
|
329
|
+
return out;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
return {
|
|
333
|
+
locateStartXref,
|
|
334
|
+
readStartXref,
|
|
335
|
+
parseXrefTable,
|
|
336
|
+
parseTrailerDict,
|
|
337
|
+
readXrefStreamDict,
|
|
338
|
+
buildXrefStream
|
|
339
|
+
};
|
|
340
|
+
}
|
|
341
|
+
};
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
// Copyright (c) 2026 AwaCloud SAS
|
|
2
|
+
// Author: Matthieu Bouilloux
|
|
3
|
+
// SPDX-License-Identifier: AGPL-3.0-only
|
|
4
|
+
// Dual-licensed; see the NOTICE file for licensing and any additional terms.
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* @fileoverview ClassMap typing per ISO 32000-2:2020 §14.7.5.4.
|
|
8
|
+
*
|
|
9
|
+
* The structure-tree `/ClassMap` maps class names (used in StructElem
|
|
10
|
+
* `/C` entries) to attribute objects or arrays of attribute objects.
|
|
11
|
+
* An attribute object is a dict that begins with `/O` naming the owner
|
|
12
|
+
* (e.g. `/Layout`, `/Table`, `/List`).
|
|
13
|
+
*
|
|
14
|
+
* The values are stored verbatim; only top-level shape is validated.
|
|
15
|
+
*
|
|
16
|
+
* @module pdf/tagged/classMap
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Module factory.
|
|
21
|
+
*/
|
|
22
|
+
import { pdfErrors } from '../errors.js';
|
|
23
|
+
import { pdfParser } from '../syntax/parser.js';
|
|
24
|
+
|
|
25
|
+
export const pdfClassMap = {
|
|
26
|
+
name: 'pdfClassMap',
|
|
27
|
+
dependencies: ['pdfErrors', 'pdfParser'],
|
|
28
|
+
deps: [pdfErrors, pdfParser],
|
|
29
|
+
factory(errors, parser) {
|
|
30
|
+
const { ParseError } = errors;
|
|
31
|
+
const { isType } = parser;
|
|
32
|
+
|
|
33
|
+
function typeClassMap(dict) {
|
|
34
|
+
if (!isType(dict, 'dict')) {
|
|
35
|
+
throw new ParseError('pdf/tagged/class-map/not-dict',
|
|
36
|
+
'ClassMap must be a dictionary',
|
|
37
|
+
{ context: { type: dict && dict.type } });
|
|
38
|
+
}
|
|
39
|
+
const classes = {};
|
|
40
|
+
for (const [k, v] of Object.entries(dict.entries)) {
|
|
41
|
+
if (!v) {
|
|
42
|
+
throw new ParseError('pdf/tagged/class-map/bad-value',
|
|
43
|
+
'ClassMap entry has no value',
|
|
44
|
+
{ context: { key: k } });
|
|
45
|
+
}
|
|
46
|
+
if (v.type === 'dict') {
|
|
47
|
+
classes[k] = [v];
|
|
48
|
+
} else if (v.type === 'array') {
|
|
49
|
+
for (const item of v.items) {
|
|
50
|
+
if (!item || item.type !== 'dict') {
|
|
51
|
+
throw new ParseError('pdf/tagged/class-map/bad-item',
|
|
52
|
+
'ClassMap array entries must be attribute dicts',
|
|
53
|
+
{ context: { key: k, kind: item && item.type } });
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
classes[k] = v.items.slice();
|
|
57
|
+
} else {
|
|
58
|
+
throw new ParseError('pdf/tagged/class-map/bad-value',
|
|
59
|
+
'ClassMap values must be a dict or array of dicts',
|
|
60
|
+
{ context: { key: k, kind: v.type } });
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
return { classes, raw: dict };
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function getClassAttributes(typed, name, owner) {
|
|
67
|
+
if (!typed || !typed.classes) return [];
|
|
68
|
+
const list = typed.classes[name];
|
|
69
|
+
if (!list) return [];
|
|
70
|
+
if (!owner) return list.slice();
|
|
71
|
+
const out = [];
|
|
72
|
+
for (const d of list) {
|
|
73
|
+
const o = d.entries && d.entries.O;
|
|
74
|
+
if (o && o.type === 'name' && o.value === owner) out.push(d);
|
|
75
|
+
}
|
|
76
|
+
return out;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
return { typeClassMap, getClassAttributes };
|
|
80
|
+
}
|
|
81
|
+
};
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
// Copyright (c) 2026 AwaCloud SAS
|
|
2
|
+
// Author: Matthieu Bouilloux
|
|
3
|
+
// SPDX-License-Identifier: AGPL-3.0-only
|
|
4
|
+
// Dual-licensed; see the NOTICE file for licensing and any additional terms.
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* @fileoverview Marked-content scanner per ISO 32000-2:2020 §14.6.
|
|
8
|
+
*
|
|
9
|
+
* Walks an operator list produced by `pdfContentStream.parseContentStream`
|
|
10
|
+
* and extracts marked-content sequences delimited by `BMC` / `BDC` /
|
|
11
|
+
* `EMC`. Each entry records its tag, properties (the inline dict or the
|
|
12
|
+
* /Properties name resolved by the caller from the page resources), the
|
|
13
|
+
* `MCID` (if any) and the inclusive op-index range.
|
|
14
|
+
*
|
|
15
|
+
* Also exposes `resolveMcidToStruct(pageRef, mcid, parentTree, resolveRef)`
|
|
16
|
+
* which combines a page's `/StructParents` key with the document
|
|
17
|
+
* `/ParentTree` to find the owning StructElem ref of a given MCID.
|
|
18
|
+
*
|
|
19
|
+
* @module pdf/tagged/markedContent
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Module factory.
|
|
24
|
+
*/
|
|
25
|
+
import { pdfErrors } from '../errors.js';
|
|
26
|
+
import { pdfParser } from '../syntax/parser.js';
|
|
27
|
+
import { pdfParentTree } from './parentTree.js';
|
|
28
|
+
|
|
29
|
+
export const pdfMarkedContent = {
|
|
30
|
+
name: 'pdfMarkedContent',
|
|
31
|
+
dependencies: ['pdfErrors', 'pdfParser', 'pdfParentTree'],
|
|
32
|
+
deps: [pdfErrors, pdfParser, pdfParentTree],
|
|
33
|
+
factory(errors, parser, parentTree) {
|
|
34
|
+
const { ParseError } = errors;
|
|
35
|
+
const { isType } = parser;
|
|
36
|
+
const { lookupParent } = parentTree;
|
|
37
|
+
|
|
38
|
+
function extractMcids(opsList) {
|
|
39
|
+
if (!Array.isArray(opsList)) {
|
|
40
|
+
throw new ParseError('pdf/tagged/marked-content/bad-ops',
|
|
41
|
+
'extractMcids expects an array of ops',
|
|
42
|
+
{ context: { typeof: typeof opsList } });
|
|
43
|
+
}
|
|
44
|
+
const out = [];
|
|
45
|
+
const stack = [];
|
|
46
|
+
for (let i = 0; i < opsList.length; i++) {
|
|
47
|
+
const op = opsList[i];
|
|
48
|
+
if (!op || typeof op.op !== 'string') continue;
|
|
49
|
+
if (op.op === 'BMC') {
|
|
50
|
+
stack.push(openEntry(op, i, null));
|
|
51
|
+
} else if (op.op === 'BDC') {
|
|
52
|
+
stack.push(openEntry(op, i, op.args[1] || null));
|
|
53
|
+
} else if (op.op === 'EMC') {
|
|
54
|
+
if (stack.length === 0) {
|
|
55
|
+
throw new ParseError('pdf/tagged/marked-content/unmatched-emc',
|
|
56
|
+
'EMC without matching BMC/BDC',
|
|
57
|
+
{ context: { index: i } });
|
|
58
|
+
}
|
|
59
|
+
const e = stack.pop();
|
|
60
|
+
e.end = i;
|
|
61
|
+
out.push(e);
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
if (stack.length > 0) {
|
|
65
|
+
throw new ParseError('pdf/tagged/marked-content/unterminated',
|
|
66
|
+
'BMC/BDC without matching EMC',
|
|
67
|
+
{ context: { open: stack.length } });
|
|
68
|
+
}
|
|
69
|
+
out.sort((a, b) => a.start - b.start);
|
|
70
|
+
return out;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function openEntry(op, index, properties) {
|
|
74
|
+
const tag = op.args[0];
|
|
75
|
+
if (!tag || tag.type !== 'name') {
|
|
76
|
+
throw new ParseError('pdf/tagged/marked-content/bad-tag',
|
|
77
|
+
'marked-content tag must be a name',
|
|
78
|
+
{ context: { index, kind: tag && tag.type } });
|
|
79
|
+
}
|
|
80
|
+
return {
|
|
81
|
+
tag: tag.value,
|
|
82
|
+
start: index,
|
|
83
|
+
end: -1,
|
|
84
|
+
mcid: readMcid(properties),
|
|
85
|
+
properties
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function readMcid(properties) {
|
|
90
|
+
if (!properties || properties.type !== 'dict') return null;
|
|
91
|
+
const m = properties.entries.MCID;
|
|
92
|
+
if (!m) return null;
|
|
93
|
+
if (m.type !== 'int' && m.type !== 'real') return null;
|
|
94
|
+
return m.value | 0;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function resolveMcidToStruct(pageDict, mcid, pTree, resolveRef) {
|
|
98
|
+
if (!isType(pageDict, 'dict')) {
|
|
99
|
+
throw new ParseError('pdf/tagged/marked-content/bad-page',
|
|
100
|
+
'page must be a dict',
|
|
101
|
+
{ context: { kind: pageDict && pageDict.type } });
|
|
102
|
+
}
|
|
103
|
+
const sp = pageDict.entries.StructParents;
|
|
104
|
+
if (!sp || (sp.type !== 'int' && sp.type !== 'real')) {
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
const found = lookupParent(pTree, sp.value | 0, resolveRef);
|
|
108
|
+
if (!found) return null;
|
|
109
|
+
let arr = found;
|
|
110
|
+
if (arr.type === 'ref') arr = resolveRef(arr);
|
|
111
|
+
if (!arr || arr.type !== 'array') {
|
|
112
|
+
throw new ParseError('pdf/tagged/marked-content/bad-entry',
|
|
113
|
+
'page /ParentTree entry must be an array indexed by MCID',
|
|
114
|
+
{ context: { kind: arr && arr.type } });
|
|
115
|
+
}
|
|
116
|
+
const item = arr.items[mcid | 0];
|
|
117
|
+
if (!item) return null;
|
|
118
|
+
return item;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
return { extractMcids, resolveMcidToStruct };
|
|
122
|
+
}
|
|
123
|
+
};
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
// Copyright (c) 2026 AwaCloud SAS
|
|
2
|
+
// Author: Matthieu Bouilloux
|
|
3
|
+
// SPDX-License-Identifier: AGPL-3.0-only
|
|
4
|
+
// Dual-licensed; see the NOTICE file for licensing and any additional terms.
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* @fileoverview Number-tree walker for `/ParentTree` per ISO 32000-2:2020
|
|
8
|
+
* §7.9.7 and §14.7.4.
|
|
9
|
+
*
|
|
10
|
+
* The structure tree's `/ParentTree` is a number tree mapping integer
|
|
11
|
+
* keys (the `/StructParents` or `/StructParent` values found on pages
|
|
12
|
+
* and annotations / content-stream objects) to:
|
|
13
|
+
*
|
|
14
|
+
* - a StructElem reference (annotations / form XObjects, where a
|
|
15
|
+
* single MCID/object owns the parent), or
|
|
16
|
+
* - an array of StructElem references indexed by MCID (pages, where
|
|
17
|
+
* each MCID on the page has its own parent).
|
|
18
|
+
*
|
|
19
|
+
* A number-tree node is a dict with either `/Nums` (leaf) or `/Kids`
|
|
20
|
+
* (intermediate). Intermediate nodes carry `/Limits` (2-int array)
|
|
21
|
+
* giving the inclusive key range of the subtree.
|
|
22
|
+
*
|
|
23
|
+
* @module pdf/tagged/parentTree
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Module factory.
|
|
28
|
+
*/
|
|
29
|
+
import { pdfErrors } from '../errors.js';
|
|
30
|
+
import { pdfParser } from '../syntax/parser.js';
|
|
31
|
+
|
|
32
|
+
export const pdfParentTree = {
|
|
33
|
+
name: 'pdfParentTree',
|
|
34
|
+
dependencies: ['pdfErrors', 'pdfParser'],
|
|
35
|
+
deps: [pdfErrors, pdfParser],
|
|
36
|
+
factory(errors, parser) {
|
|
37
|
+
const { ParseError } = errors;
|
|
38
|
+
const { isType } = parser;
|
|
39
|
+
|
|
40
|
+
function lookupParent(tree, key, resolveRef, opts) {
|
|
41
|
+
const maxDepth = (opts && opts.maxDepth) || 32;
|
|
42
|
+
let node = tree;
|
|
43
|
+
if (node && node.type === 'ref') node = resolveRef(node);
|
|
44
|
+
return walk(node, 0);
|
|
45
|
+
|
|
46
|
+
function walk(n, depth) {
|
|
47
|
+
if (depth > maxDepth) {
|
|
48
|
+
throw new ParseError('pdf/tagged/parent-tree/max-depth',
|
|
49
|
+
'number tree depth limit exceeded',
|
|
50
|
+
{ context: { depth, maxDepth } });
|
|
51
|
+
}
|
|
52
|
+
if (!isType(n, 'dict')) {
|
|
53
|
+
throw new ParseError('pdf/tagged/parent-tree/not-dict',
|
|
54
|
+
'number tree node is not a dict',
|
|
55
|
+
{ context: { kind: n && n.type } });
|
|
56
|
+
}
|
|
57
|
+
const nums = n.entries.Nums;
|
|
58
|
+
const kids = n.entries.Kids;
|
|
59
|
+
|
|
60
|
+
if (nums && nums.type === 'array') {
|
|
61
|
+
return findInNums(nums.items, key);
|
|
62
|
+
}
|
|
63
|
+
if (kids && kids.type === 'array') {
|
|
64
|
+
for (const k of kids.items) {
|
|
65
|
+
if (k.type !== 'ref') {
|
|
66
|
+
throw new ParseError('pdf/tagged/parent-tree/non-ref-kid',
|
|
67
|
+
'number-tree /Kids entries must be indirect refs',
|
|
68
|
+
{ context: { kind: k.type } });
|
|
69
|
+
}
|
|
70
|
+
const child = resolveRef(k);
|
|
71
|
+
if (!isType(child, 'dict')) {
|
|
72
|
+
throw new ParseError('pdf/tagged/parent-tree/bad-child',
|
|
73
|
+
'number-tree kid did not resolve to dict',
|
|
74
|
+
{ context: { ref: k } });
|
|
75
|
+
}
|
|
76
|
+
if (inLimits(child.entries.Limits, key)) {
|
|
77
|
+
return walk(child, depth + 1);
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
return null;
|
|
81
|
+
}
|
|
82
|
+
throw new ParseError('pdf/tagged/parent-tree/empty-node',
|
|
83
|
+
'number tree node has neither /Nums nor /Kids',
|
|
84
|
+
{ context: {} });
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function findInNums(items, key) {
|
|
89
|
+
if (items.length % 2 !== 0) {
|
|
90
|
+
throw new ParseError('pdf/tagged/parent-tree/bad-nums',
|
|
91
|
+
'/Nums must have an even number of entries',
|
|
92
|
+
{ context: { len: items.length } });
|
|
93
|
+
}
|
|
94
|
+
for (let i = 0; i < items.length; i += 2) {
|
|
95
|
+
const k = items[i];
|
|
96
|
+
if (!k || (k.type !== 'int' && k.type !== 'real')) {
|
|
97
|
+
throw new ParseError('pdf/tagged/parent-tree/bad-key',
|
|
98
|
+
'/Nums key must be a number',
|
|
99
|
+
{ context: { kind: k && k.type } });
|
|
100
|
+
}
|
|
101
|
+
if ((k.value | 0) === key) return items[i + 1];
|
|
102
|
+
}
|
|
103
|
+
return null;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function inLimits(limits, key) {
|
|
107
|
+
if (!limits) return true;
|
|
108
|
+
if (limits.type !== 'array' || limits.items.length !== 2) {
|
|
109
|
+
throw new ParseError('pdf/tagged/parent-tree/bad-limits',
|
|
110
|
+
'/Limits must be a 2-element array of integers',
|
|
111
|
+
{ context: { kind: limits.type } });
|
|
112
|
+
}
|
|
113
|
+
const lo = limits.items[0];
|
|
114
|
+
const hi = limits.items[1];
|
|
115
|
+
if (!lo || !hi || (lo.type !== 'int' && lo.type !== 'real')
|
|
116
|
+
|| (hi.type !== 'int' && hi.type !== 'real')) {
|
|
117
|
+
throw new ParseError('pdf/tagged/parent-tree/bad-limits',
|
|
118
|
+
'/Limits entries must be numbers',
|
|
119
|
+
{ context: {} });
|
|
120
|
+
}
|
|
121
|
+
return key >= (lo.value | 0) && key <= (hi.value | 0);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
return { lookupParent };
|
|
125
|
+
}
|
|
126
|
+
};
|