@awacloud/pdf 0.0.0-stage → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (345) hide show
  1. package/CHANGELOG.md +609 -0
  2. package/LICENSE +661 -0
  3. package/NOTICE +77 -0
  4. package/README.md +363 -2
  5. package/dist/build/index.js +21 -0
  6. package/dist/build/pdf-full-rw.js +10972 -0
  7. package/dist/build/pdf-full-rw.meta.json +105 -0
  8. package/dist/build/pdf-full-rw.min.js +53 -0
  9. package/dist/build/pdf-full.js +6078 -0
  10. package/dist/build/pdf-full.meta.json +90 -0
  11. package/dist/build/pdf-full.min.js +32 -0
  12. package/dist/build/pdf-large-rw.js +10367 -0
  13. package/dist/build/pdf-large-rw.meta.json +99 -0
  14. package/dist/build/pdf-large-rw.min.js +53 -0
  15. package/dist/build/pdf-large.js +5473 -0
  16. package/dist/build/pdf-large.meta.json +84 -0
  17. package/dist/build/pdf-large.min.js +32 -0
  18. package/dist/build/pdf-legacy-rw.js +12402 -0
  19. package/dist/build/pdf-legacy-rw.meta.json +110 -0
  20. package/dist/build/pdf-legacy-rw.min.js +53 -0
  21. package/dist/build/pdf-legacy.js +7508 -0
  22. package/dist/build/pdf-legacy.meta.json +95 -0
  23. package/dist/build/pdf-legacy.min.js +32 -0
  24. package/dist/build/pdf-rw.js +7578 -0
  25. package/dist/build/pdf-rw.meta.json +77 -0
  26. package/dist/build/pdf-rw.min.js +53 -0
  27. package/dist/build/pdf.js +2684 -0
  28. package/dist/build/pdf.meta.json +62 -0
  29. package/dist/build/pdf.min.js +32 -0
  30. package/dist/standalone/pdf-full-rw.js +16798 -0
  31. package/dist/standalone/pdf-full-rw.meta.json +78 -0
  32. package/dist/standalone/pdf-full-rw.min.js +56 -0
  33. package/dist/standalone/pdf-full.js +11904 -0
  34. package/dist/standalone/pdf-full.meta.json +63 -0
  35. package/dist/standalone/pdf-full.min.js +35 -0
  36. package/dist/standalone/pdf-large-rw.js +16193 -0
  37. package/dist/standalone/pdf-large-rw.meta.json +72 -0
  38. package/dist/standalone/pdf-large-rw.min.js +56 -0
  39. package/dist/standalone/pdf-large.js +11299 -0
  40. package/dist/standalone/pdf-large.meta.json +57 -0
  41. package/dist/standalone/pdf-large.min.js +35 -0
  42. package/dist/standalone/pdf-legacy-rw.js +18228 -0
  43. package/dist/standalone/pdf-legacy-rw.meta.json +83 -0
  44. package/dist/standalone/pdf-legacy-rw.min.js +56 -0
  45. package/dist/standalone/pdf-legacy.js +13334 -0
  46. package/dist/standalone/pdf-legacy.meta.json +68 -0
  47. package/dist/standalone/pdf-legacy.min.js +35 -0
  48. package/dist/standalone/pdf-rw.js +13404 -0
  49. package/dist/standalone/pdf-rw.meta.json +50 -0
  50. package/dist/standalone/pdf-rw.min.js +56 -0
  51. package/dist/standalone/pdf.js +8510 -0
  52. package/dist/standalone/pdf.meta.json +35 -0
  53. package/dist/standalone/pdf.min.js +35 -0
  54. package/docs/README.md +53 -0
  55. package/docs/api/README.md +38 -0
  56. package/docs/api/_shared/README.md +91 -0
  57. package/docs/api/action/README.md +29 -0
  58. package/docs/api/action/action.md +81 -0
  59. package/docs/api/action/goTo.md +66 -0
  60. package/docs/api/action/launch.md +58 -0
  61. package/docs/api/action/named.md +55 -0
  62. package/docs/api/action/uri.md +54 -0
  63. package/docs/api/annot/README.md +53 -0
  64. package/docs/api/annot/annot.md +114 -0
  65. package/docs/api/annot/fileAttach.md +53 -0
  66. package/docs/api/annot/freeText.md +68 -0
  67. package/docs/api/annot/ink.md +69 -0
  68. package/docs/api/annot/link.md +74 -0
  69. package/docs/api/annot/markup.md +83 -0
  70. package/docs/api/annot/popup.md +52 -0
  71. package/docs/api/annot/projection.md +56 -0
  72. package/docs/api/annot/redact.md +67 -0
  73. package/docs/api/annot/square.md +87 -0
  74. package/docs/api/annot/stamp.md +54 -0
  75. package/docs/api/annot/text.md +69 -0
  76. package/docs/api/annot/widget.md +69 -0
  77. package/docs/api/associatedFiles/README.md +9 -0
  78. package/docs/api/associatedFiles/associatedFiles.md +78 -0
  79. package/docs/api/bundles/README.md +68 -0
  80. package/docs/api/bundles/dist-matrix.md +165 -0
  81. package/docs/api/bundles/pdf-full.md +148 -0
  82. package/docs/api/bundles/pdf-large.md +144 -0
  83. package/docs/api/bundles/pdf-legacy.md +169 -0
  84. package/docs/api/content/README.md +29 -0
  85. package/docs/api/content/color.md +99 -0
  86. package/docs/api/content/graphics.md +114 -0
  87. package/docs/api/content/images.md +124 -0
  88. package/docs/api/content/ops.md +100 -0
  89. package/docs/api/content/stream.md +107 -0
  90. package/docs/api/content/text.md +98 -0
  91. package/docs/api/crypto/README.md +29 -0
  92. package/docs/api/crypto/aesGcm.md +72 -0
  93. package/docs/api/crypto/permissions.md +79 -0
  94. package/docs/api/crypto/security.md +98 -0
  95. package/docs/api/crypto/standardV4.md +104 -0
  96. package/docs/api/crypto/standardV5.md +84 -0
  97. package/docs/api/crypto/standardV6.md +93 -0
  98. package/docs/api/destination/README.md +9 -0
  99. package/docs/api/destination/destination.md +79 -0
  100. package/docs/api/document/README.md +29 -0
  101. package/docs/api/document/builder.md +281 -0
  102. package/docs/api/document/catalog.md +98 -0
  103. package/docs/api/document/document.md +187 -0
  104. package/docs/api/document/encryptedWriter.md +149 -0
  105. package/docs/api/document/incrementalWriter.md +148 -0
  106. package/docs/api/document/page.md +99 -0
  107. package/docs/api/document/pages.md +82 -0
  108. package/docs/api/document/resources.md +102 -0
  109. package/docs/api/document/writer.md +157 -0
  110. package/docs/api/document/xrefStreamWriter.md +122 -0
  111. package/docs/api/embedded/README.md +13 -0
  112. package/docs/api/embedded/collection.md +80 -0
  113. package/docs/api/embedded/embeddedFile.md +86 -0
  114. package/docs/api/embedded/fileSpec.md +87 -0
  115. package/docs/api/errors.md +110 -0
  116. package/docs/api/extra/3d-richmedia.md +76 -0
  117. package/docs/api/extra/README.md +99 -0
  118. package/docs/api/extra/annot-extended.md +71 -0
  119. package/docs/api/extra/associated-files.md +70 -0
  120. package/docs/api/extra/ccitt-fax-decoder.md +74 -0
  121. package/docs/api/extra/color-spaces-extended.md +72 -0
  122. package/docs/api/extra/content-ops-extended.md +82 -0
  123. package/docs/api/extra/document-parts.md +69 -0
  124. package/docs/api/extra/embedded-files-portfolio.md +87 -0
  125. package/docs/api/extra/font-cid-typed.md +77 -0
  126. package/docs/api/extra/font-color-tagging.md +76 -0
  127. package/docs/api/extra/form-actions-extended.md +75 -0
  128. package/docs/api/extra/info-dict-deprecated.md +72 -0
  129. package/docs/api/extra/jbig2-read.md +80 -0
  130. package/docs/api/extra/legacy-deprecated-annots.md +89 -0
  131. package/docs/api/extra/legacy-deprecated-filters.md +78 -0
  132. package/docs/api/extra/legacy-rc4-read.md +74 -0
  133. package/docs/api/extra/legacy-xfa-read.md +65 -0
  134. package/docs/api/extra/linearization-write.md +71 -0
  135. package/docs/api/extra/misc.md +93 -0
  136. package/docs/api/extra/optional-content-extended.md +83 -0
  137. package/docs/api/extra/pdf-a-output-intent.md +65 -0
  138. package/docs/api/extra/pdf-sandbox.md +76 -0
  139. package/docs/api/extra/pdf-ua-tagged.md +63 -0
  140. package/docs/api/extra/pdf-x-prepress.md +65 -0
  141. package/docs/api/extra/redaction-iso32005.md +65 -0
  142. package/docs/api/extra/shading-typed.md +73 -0
  143. package/docs/api/extra/sig-aes-gcm.md +69 -0
  144. package/docs/api/extra/sig-pades.md +103 -0
  145. package/docs/api/extra/tagged-pdf-typed.md +78 -0
  146. package/docs/api/extra/transparency-typed.md +74 -0
  147. package/docs/api/extra/well-tagged-pdf.md +61 -0
  148. package/docs/api/extra/xmp-extended.md +65 -0
  149. package/docs/api/font/README.md +25 -0
  150. package/docs/api/font/embed.md +157 -0
  151. package/docs/api/font/encoding.md +95 -0
  152. package/docs/api/font/font.md +97 -0
  153. package/docs/api/font/type3.md +89 -0
  154. package/docs/api/form/README.md +35 -0
  155. package/docs/api/form/acroform.md +88 -0
  156. package/docs/api/form/appearance.md +87 -0
  157. package/docs/api/form/button.md +97 -0
  158. package/docs/api/form/choice.md +96 -0
  159. package/docs/api/form/fieldTree.md +93 -0
  160. package/docs/api/form/signature.md +90 -0
  161. package/docs/api/form/text.md +88 -0
  162. package/docs/api/linearization/README.md +11 -0
  163. package/docs/api/linearization/linearization.md +81 -0
  164. package/docs/api/main.md +116 -0
  165. package/docs/api/metadata/README.md +10 -0
  166. package/docs/api/metadata/info.md +70 -0
  167. package/docs/api/metadata/xmp.md +62 -0
  168. package/docs/api/ocg/README.md +23 -0
  169. package/docs/api/ocg/config.md +95 -0
  170. package/docs/api/ocg/ocg.md +77 -0
  171. package/docs/api/outline/README.md +11 -0
  172. package/docs/api/outline/outline.md +107 -0
  173. package/docs/api/pdf.md +152 -0
  174. package/docs/api/prepress/README.md +10 -0
  175. package/docs/api/prepress/outputIntent.md +79 -0
  176. package/docs/api/prepress/pageBoundary.md +75 -0
  177. package/docs/api/sig/README.md +32 -0
  178. package/docs/api/sig/byteRange.md +120 -0
  179. package/docs/api/sig/certChain.md +84 -0
  180. package/docs/api/sig/dss.md +111 -0
  181. package/docs/api/sig/oids.md +76 -0
  182. package/docs/api/sig/sha1.md +72 -0
  183. package/docs/api/sig/sign.md +317 -0
  184. package/docs/api/sig/signature.md +178 -0
  185. package/docs/api/sig/timestamp.md +84 -0
  186. package/docs/api/syntax/README.md +29 -0
  187. package/docs/api/syntax/crossRefStream.md +115 -0
  188. package/docs/api/syntax/filters/README.md +50 -0
  189. package/docs/api/syntax/filters/ascii85.md +76 -0
  190. package/docs/api/syntax/filters/asciiHex.md +73 -0
  191. package/docs/api/syntax/filters/dispatch.md +125 -0
  192. package/docs/api/syntax/filters/flate.md +134 -0
  193. package/docs/api/syntax/filters/runLength.md +78 -0
  194. package/docs/api/syntax/objStream.md +88 -0
  195. package/docs/api/syntax/parser-obj.md +97 -0
  196. package/docs/api/syntax/parser.md +151 -0
  197. package/docs/api/syntax/serializer.md +109 -0
  198. package/docs/api/syntax/tokenizer.md +104 -0
  199. package/docs/api/syntax/trailer.md +85 -0
  200. package/docs/api/syntax/xref.md +139 -0
  201. package/docs/api/tagged/README.md +25 -0
  202. package/docs/api/tagged/classMap.md +67 -0
  203. package/docs/api/tagged/markedContent.md +62 -0
  204. package/docs/api/tagged/parentTree.md +67 -0
  205. package/docs/api/tagged/roleMap.md +67 -0
  206. package/docs/api/tagged/structElement.md +76 -0
  207. package/docs/api/tagged/structTree.md +75 -0
  208. package/docs/guide/coverage.md +113 -0
  209. package/docs/guide/crypto.md +121 -0
  210. package/docs/guide/extending.md +76 -0
  211. package/docs/guide/getting-started.md +75 -0
  212. package/docs/guide/legacy-1.7.md +42 -0
  213. package/docs/guide/pades-integration.md +579 -0
  214. package/docs/guide/read-pdf.md +89 -0
  215. package/package.json +97 -4
  216. package/src/_shared/index.js +179 -0
  217. package/src/action/action.js +119 -0
  218. package/src/action/goTo.js +89 -0
  219. package/src/action/launch.js +61 -0
  220. package/src/action/named.js +54 -0
  221. package/src/action/uri.js +51 -0
  222. package/src/annot/annot.js +212 -0
  223. package/src/annot/fileAttach.js +55 -0
  224. package/src/annot/freeText.js +82 -0
  225. package/src/annot/ink.js +77 -0
  226. package/src/annot/link.js +77 -0
  227. package/src/annot/markup.js +91 -0
  228. package/src/annot/popup.js +53 -0
  229. package/src/annot/projection.js +52 -0
  230. package/src/annot/redact.js +87 -0
  231. package/src/annot/square.js +132 -0
  232. package/src/annot/stamp.js +48 -0
  233. package/src/annot/text.js +54 -0
  234. package/src/annot/widget.js +61 -0
  235. package/src/associatedFiles/associatedFiles.js +86 -0
  236. package/src/bundles/pdf-full.js +91 -0
  237. package/src/bundles/pdf-large.js +81 -0
  238. package/src/bundles/pdf-legacy.js +107 -0
  239. package/src/content/color.js +114 -0
  240. package/src/content/graphics.js +192 -0
  241. package/src/content/images.js +160 -0
  242. package/src/content/ops.js +137 -0
  243. package/src/content/stream.js +154 -0
  244. package/src/content/text.js +125 -0
  245. package/src/crypto/aesGcm.js +123 -0
  246. package/src/crypto/permissions.js +112 -0
  247. package/src/crypto/security.js +327 -0
  248. package/src/crypto/standardV4.js +443 -0
  249. package/src/crypto/standardV5.js +306 -0
  250. package/src/crypto/standardV6.js +334 -0
  251. package/src/destination/destination.js +183 -0
  252. package/src/document/builder.js +618 -0
  253. package/src/document/catalog.js +100 -0
  254. package/src/document/document.js +472 -0
  255. package/src/document/encryptedWriter.js +554 -0
  256. package/src/document/incrementalWriter.js +514 -0
  257. package/src/document/page.js +131 -0
  258. package/src/document/pages.js +103 -0
  259. package/src/document/resources.js +146 -0
  260. package/src/document/writer.js +211 -0
  261. package/src/document/xrefStreamWriter.js +353 -0
  262. package/src/embedded/collection.js +102 -0
  263. package/src/embedded/embeddedFile.js +99 -0
  264. package/src/embedded/fileSpec.js +137 -0
  265. package/src/errors.js +78 -0
  266. package/src/extra/3d-richmedia.js +171 -0
  267. package/src/extra/annot-extended.js +200 -0
  268. package/src/extra/associated-files.js +131 -0
  269. package/src/extra/ccitt-fax-decoder.js +776 -0
  270. package/src/extra/color-spaces-extended.js +196 -0
  271. package/src/extra/content-ops-extended.js +153 -0
  272. package/src/extra/document-parts.js +149 -0
  273. package/src/extra/embedded-files-portfolio.js +234 -0
  274. package/src/extra/font-cid-typed.js +185 -0
  275. package/src/extra/font-color-tagging.js +144 -0
  276. package/src/extra/form-actions-extended.js +196 -0
  277. package/src/extra/info-dict-deprecated.js +137 -0
  278. package/src/extra/jbig2-read.js +169 -0
  279. package/src/extra/legacy-deprecated-annots.js +198 -0
  280. package/src/extra/legacy-deprecated-filters.js +167 -0
  281. package/src/extra/legacy-rc4-read.js +235 -0
  282. package/src/extra/legacy-xfa-read.js +104 -0
  283. package/src/extra/linearization-write.js +97 -0
  284. package/src/extra/misc.js +217 -0
  285. package/src/extra/optional-content-extended.js +142 -0
  286. package/src/extra/pdf-a-output-intent.js +112 -0
  287. package/src/extra/pdf-sandbox.js +88 -0
  288. package/src/extra/pdf-ua-tagged.js +116 -0
  289. package/src/extra/pdf-x-prepress.js +114 -0
  290. package/src/extra/redaction-iso32005.js +136 -0
  291. package/src/extra/shading-typed.js +222 -0
  292. package/src/extra/sig-aes-gcm.js +135 -0
  293. package/src/extra/sig-pades.js +242 -0
  294. package/src/extra/tagged-pdf-typed.js +203 -0
  295. package/src/extra/transparency-typed.js +135 -0
  296. package/src/extra/well-tagged-pdf.js +138 -0
  297. package/src/extra/xmp-extended.js +190 -0
  298. package/src/font/embed.js +480 -0
  299. package/src/font/encoding.js +92 -0
  300. package/src/font/font.js +101 -0
  301. package/src/font/type3.js +75 -0
  302. package/src/form/acroform.js +94 -0
  303. package/src/form/appearance.js +90 -0
  304. package/src/form/button.js +105 -0
  305. package/src/form/choice.js +152 -0
  306. package/src/form/fieldTree.js +120 -0
  307. package/src/form/signature.js +100 -0
  308. package/src/form/text.js +101 -0
  309. package/src/linearization/linearization.js +107 -0
  310. package/src/main.js +411 -0
  311. package/src/metadata/info.js +87 -0
  312. package/src/metadata/xmp.js +62 -0
  313. package/src/ocg/config.js +156 -0
  314. package/src/ocg/ocg.js +124 -0
  315. package/src/outline/outline.js +157 -0
  316. package/src/pdf.js +133 -0
  317. package/src/prepress/outputIntent.js +118 -0
  318. package/src/prepress/pageBoundary.js +108 -0
  319. package/src/sig/byteRange.js +306 -0
  320. package/src/sig/certChain.js +247 -0
  321. package/src/sig/dss.js +317 -0
  322. package/src/sig/oids.js +157 -0
  323. package/src/sig/sha1.js +142 -0
  324. package/src/sig/sign.js +1899 -0
  325. package/src/sig/signature.js +1441 -0
  326. package/src/sig/timestamp.js +236 -0
  327. package/src/syntax/crossRefStream.js +133 -0
  328. package/src/syntax/filters/ascii85.js +122 -0
  329. package/src/syntax/filters/asciiHex.js +83 -0
  330. package/src/syntax/filters/dispatch.js +176 -0
  331. package/src/syntax/filters/flate.js +316 -0
  332. package/src/syntax/filters/runLength.js +96 -0
  333. package/src/syntax/objStream.js +99 -0
  334. package/src/syntax/parser-obj.js +52 -0
  335. package/src/syntax/parser.js +321 -0
  336. package/src/syntax/serializer.js +221 -0
  337. package/src/syntax/tokenizer.js +290 -0
  338. package/src/syntax/trailer.js +76 -0
  339. package/src/syntax/xref.js +341 -0
  340. package/src/tagged/classMap.js +81 -0
  341. package/src/tagged/markedContent.js +123 -0
  342. package/src/tagged/parentTree.js +126 -0
  343. package/src/tagged/roleMap.js +107 -0
  344. package/src/tagged/structElement.js +138 -0
  345. package/src/tagged/structTree.js +94 -0
@@ -0,0 +1,100 @@
1
+ // Copyright (c) 2026 AwaCloud SAS
2
+ // Author: Matthieu Bouilloux
3
+ // SPDX-License-Identifier: AGPL-3.0-only
4
+ // Dual-licensed; see the NOTICE file for licensing and any additional terms.
5
+
6
+ /**
7
+ * @fileoverview Catalog typing per ISO 32000-2:2020 §7.7.2.
8
+ *
9
+ * The Catalog (`/Type /Catalog`) is the root of the document tree
10
+ * referenced by the trailer's `/Root` entry. It holds the entry point
11
+ * to pages, outlines, form fields, structure tree, metadata, version
12
+ * override and viewer preferences.
13
+ *
14
+ * The L0 typing exposes the most common, well-defined entries:
15
+ * `version`, `pages`, `pageLabels`, `names`, `dests`, `viewerPrefs`,
16
+ * `pageLayout`, `pageMode`, `outlines`, `metadata`, `structTreeRoot`,
17
+ * `markInfo`, `lang`, `acroForm`, `oCProperties`, `outputIntents`.
18
+ *
19
+ * Unknown entries are preserved in `_extras`.
20
+ *
21
+ * @module pdf/document/catalog
22
+ */
23
+
24
+ /**
25
+ * Module factory — worker-safe, self-contained.
26
+ */
27
+ import { pdfErrors } from '../errors.js';
28
+ import { pdfParser } from '../syntax/parser.js';
29
+
30
+ export const pdfCatalog = {
31
+ name: 'pdfCatalog',
32
+ dependencies: ['pdfErrors', 'pdfParser'],
33
+ deps: [pdfErrors, pdfParser],
34
+ factory(errors, parserMod) {
35
+ const { ParseError } = errors;
36
+ const isType = (parserMod && parserMod.isType)
37
+ || ((v, kind) => !!(v && v.type === kind));
38
+
39
+ const KNOWN = new Set([
40
+ 'Type', 'Version', 'Pages', 'PageLabels', 'Names', 'Dests',
41
+ 'ViewerPreferences', 'PageLayout', 'PageMode', 'Outlines',
42
+ 'Threads', 'OpenAction', 'AA', 'URI', 'AcroForm', 'Metadata',
43
+ 'StructTreeRoot', 'MarkInfo', 'Lang', 'SpiderInfo', 'OutputIntents',
44
+ 'PieceInfo', 'OCProperties', 'Perms', 'Legal', 'Requirements',
45
+ 'Collection', 'NeedsRendering', 'DSS', 'AF', 'DPartRoot'
46
+ ]);
47
+
48
+ function typeCatalog(dict) {
49
+ if (!isType(dict, 'dict')) {
50
+ throw new ParseError('pdf/catalog/not-dict',
51
+ 'Catalog must be a dictionary',
52
+ { context: { type: dict && dict.type } });
53
+ }
54
+ const e = dict.entries;
55
+
56
+ if (e.Type && (e.Type.type !== 'name' || e.Type.value !== 'Catalog')) {
57
+ throw new ParseError('pdf/catalog/bad-type',
58
+ '/Type entry must be /Catalog',
59
+ { context: { actual: e.Type.value } });
60
+ }
61
+
62
+ if (!e.Pages || e.Pages.type !== 'ref') {
63
+ throw new ParseError('pdf/catalog/missing-pages',
64
+ 'Catalog is missing required /Pages reference',
65
+ { context: { type: e.Pages && e.Pages.type } });
66
+ }
67
+
68
+ const out = {
69
+ pages: { num: e.Pages.num, gen: e.Pages.gen },
70
+ raw: dict,
71
+ _extras: {}
72
+ };
73
+
74
+ if (e.Version && e.Version.type === 'name') out.version = e.Version.value;
75
+ if (e.PageLayout && e.PageLayout.type === 'name') out.pageLayout = e.PageLayout.value;
76
+ if (e.PageMode && e.PageMode.type === 'name') out.pageMode = e.PageMode.value;
77
+ if (e.Lang && e.Lang.type === 'string') out.lang = e.Lang.value;
78
+
79
+ if (e.Outlines && e.Outlines.type === 'ref') out.outlines = e.Outlines;
80
+ if (e.Metadata && e.Metadata.type === 'ref') out.metadata = e.Metadata;
81
+ if (e.StructTreeRoot && e.StructTreeRoot.type === 'ref') out.structTreeRoot = e.StructTreeRoot;
82
+ if (e.AcroForm) out.acroForm = e.AcroForm;
83
+ if (e.Names) out.names = e.Names;
84
+ if (e.Dests) out.dests = e.Dests;
85
+ if (e.ViewerPreferences) out.viewerPrefs = e.ViewerPreferences;
86
+ if (e.PageLabels) out.pageLabels = e.PageLabels;
87
+ if (e.MarkInfo) out.markInfo = e.MarkInfo;
88
+ if (e.OCProperties) out.ocProperties = e.OCProperties;
89
+ if (e.OutputIntents) out.outputIntents = e.OutputIntents;
90
+
91
+ for (const k of Object.keys(e)) {
92
+ if (!KNOWN.has(k)) out._extras[k] = e[k];
93
+ }
94
+
95
+ return out;
96
+ }
97
+
98
+ return { typeCatalog };
99
+ }
100
+ };
@@ -0,0 +1,472 @@
1
+ // Copyright (c) 2026 AwaCloud SAS
2
+ // Author: Matthieu Bouilloux
3
+ // SPDX-License-Identifier: AGPL-3.0-only
4
+ // Dual-licensed; see the NOTICE file for licensing and any additional terms.
5
+
6
+ /**
7
+ * @fileoverview Top-level document reader.
8
+ *
9
+ * Orchestrates the syntax layer (`pdfTokenizer`, `pdfParser`, `pdfXref`,
10
+ * `pdfTrailer`) and the document layer (`pdfCatalog`, `pdfPages`,
11
+ * `pdfPage`) to turn a `Uint8Array` of PDF bytes into a navigable
12
+ * typed model.
13
+ *
14
+ * Cross-reference streams (`/Type /XRef`, §7.5.8), object streams
15
+ * (`/Type /ObjStm`, §7.5.7) and hybrid-reference files (a classical
16
+ * table whose trailer carries `/XRefStm`) are read automatically: the
17
+ * section walk inspects the bytes at each `startxref`/`/Prev` offset
18
+ * and takes the table path or the stream path accordingly, so mixed
19
+ * update chains (table → stream, stream → table) resolve too. Objects
20
+ * stored inside an object stream are materialised on demand through
21
+ * `pdfObjStream`, with the container decoded once per document.
22
+ *
23
+ * Encrypted documents (trailer `/Encrypt` present) fail loud by
24
+ * default: `readDocument` throws `pdf/document/encrypted` instead of
25
+ * silently handing back a ciphertext model. The full compose-decrypt
26
+ * read path (password API, V4/V5/V6 handler selection, per-object
27
+ * decrypt through `resolveByKey`) is not provided by this module — pass
28
+ * `{ allowEncrypted: true }` to opt into the raw behaviour.
29
+ *
30
+ * @module pdf/document/document
31
+ */
32
+
33
+ /**
34
+ * Module factory — worker-safe, self-contained.
35
+ */
36
+ import { pdfErrors } from '../errors.js';
37
+ import { pdfTokenizer } from '../syntax/tokenizer.js';
38
+ import { pdfParser } from '../syntax/parser.js';
39
+ import { pdfXref } from '../syntax/xref.js';
40
+ import { pdfTrailer } from '../syntax/trailer.js';
41
+ import { pdfCatalog } from './catalog.js';
42
+ import { pdfPage } from './page.js';
43
+ import { pdfPages } from './pages.js';
44
+ import { pdfCrossRefStream } from '../syntax/crossRefStream.js';
45
+ import { pdfObjStream } from '../syntax/objStream.js';
46
+ import { pdfFilterDispatch } from '../syntax/filters/dispatch.js';
47
+
48
+ export const pdfDocument = {
49
+ name: 'pdfDocument',
50
+ dependencies: [
51
+ 'pdfErrors', 'pdfTokenizer', 'pdfParser',
52
+ 'pdfXref', 'pdfTrailer',
53
+ 'pdfCatalog', 'pdfPage', 'pdfPages',
54
+ 'pdfCrossRefStream', 'pdfObjStream', 'pdfFilterDispatch'
55
+ ],
56
+ deps: [pdfErrors, pdfTokenizer, pdfParser, pdfXref, pdfTrailer, pdfCatalog, pdfPage, pdfPages, pdfCrossRefStream, pdfObjStream, pdfFilterDispatch],
57
+ factory(errors, tokenizerMod, parserMod, xrefMod, trailerMod,
58
+ catalogMod, pageMod, pagesMod,
59
+ crossRefStreamMod, objStreamMod, filterDispatchMod) {
60
+ const { ParseError } = errors;
61
+ const tokenize = tokenizerMod && tokenizerMod.tokenize;
62
+ const parseIndirect = parserMod && parserMod.parseIndirect;
63
+ const locateStartXref = xrefMod && xrefMod.locateStartXref;
64
+ const readStartXref = xrefMod && xrefMod.readStartXref;
65
+ const parseXrefTable = xrefMod && xrefMod.parseXrefTable;
66
+ const parseTrailerDict = xrefMod && xrefMod.parseTrailerDict;
67
+ const typeTrailer = trailerMod && trailerMod.typeTrailer;
68
+ const typeCatalog = catalogMod && catalogMod.typeCatalog;
69
+ const typePage = pageMod && pageMod.typePage;
70
+ const walkPageTree = pagesMod && pagesMod.walkPageTree;
71
+
72
+ const HEADER_PREFIX = new Uint8Array([0x25, 0x50, 0x44, 0x46, 0x2D]); // %PDF-
73
+ const XREF_KEYWORD = new Uint8Array([0x78, 0x72, 0x65, 0x66]); // xref
74
+ const PDF_NULL = Object.freeze({ type: 'null' });
75
+ // Trailer / xref-stream dict keys that describe ONE section only and
76
+ // are never inherited by the merged trailer (§7.5.5, §7.5.8.2).
77
+ const SECTION_LOCAL_KEYS = new Set([
78
+ 'Prev', 'XRefStm', 'Type', 'W', 'Index', 'Length',
79
+ 'Filter', 'DecodeParms', 'F', 'FFilter', 'FDecodeParms', 'DL'
80
+ ]);
81
+ const EMPTY_PAGES_NODE = Object.freeze({
82
+ type: 'dict',
83
+ entries: Object.freeze({
84
+ Type: Object.freeze({ type: 'name', value: 'Pages' }),
85
+ Kids: Object.freeze({ type: 'array', items: Object.freeze([]) })
86
+ })
87
+ });
88
+
89
+ /** Skip the PDF white-space run (§7.2.3) starting at `at`. */
90
+ function skipWhitespace(bytes, at) {
91
+ let p = at < 0 ? 0 : at;
92
+ while (p < bytes.length) {
93
+ const b = bytes[p];
94
+ if (b === 0x00 || b === 0x09 || b === 0x0A
95
+ || b === 0x0C || b === 0x0D || b === 0x20) { p++; continue; }
96
+ break;
97
+ }
98
+ return p;
99
+ }
100
+
101
+ /** True when the four bytes at `at` spell the `xref` keyword. */
102
+ function startsXrefTable(bytes, at) {
103
+ for (let k = 0; k < XREF_KEYWORD.length; k++) {
104
+ if (bytes[at + k] !== XREF_KEYWORD[k]) return false;
105
+ }
106
+ return true;
107
+ }
108
+
109
+ /**
110
+ * A cross-reference stream is parsed before any xref exists, so an
111
+ * indirect `/Length` cannot be resolved — fail loud instead of
112
+ * silently falling back to an `endstream` scan.
113
+ */
114
+ function refuseIndirectLength(ref) {
115
+ throw new ParseError('pdf/document/xrefstm-indirect-length',
116
+ 'a cross-reference stream may not use an indirect /Length',
117
+ { context: { num: ref && ref.num, gen: ref && ref.gen } });
118
+ }
119
+
120
+ function requireStreamWiring(offset) {
121
+ if (!crossRefStreamMod || !objStreamMod || !filterDispatchMod) {
122
+ throw new ParseError('pdf/document/xref-stream-unwired',
123
+ 'this pdfDocument was built without pdfCrossRefStream, ' +
124
+ 'pdfObjStream and pdfFilterDispatch — cross-reference ' +
125
+ 'streams and object streams cannot be read',
126
+ { context: { offset } });
127
+ }
128
+ }
129
+
130
+ /**
131
+ * Read one `/Type /XRef` cross-reference stream section at `at`
132
+ * (§7.5.8) and return its entries plus its dict, which doubles as
133
+ * the section's trailer.
134
+ */
135
+ function readXrefStreamSection(bytes, at) {
136
+ requireStreamWiring(at);
137
+ const tok = tokenize(bytes, { start: at });
138
+ const def = parseIndirect(tok, refuseIndirectLength);
139
+ const streamObj = def.value;
140
+ const typeEntry = streamObj && streamObj.dict
141
+ && streamObj.dict.entries && streamObj.dict.entries.Type;
142
+ if (!streamObj || streamObj.type !== 'stream'
143
+ || !typeEntry || typeEntry.type !== 'name'
144
+ || typeEntry.value !== 'XRef') {
145
+ throw new ParseError('pdf/document/bad-xref-section',
146
+ 'neither an xref table nor a cross-reference stream at startxref',
147
+ { context: { offset: at } });
148
+ }
149
+ const decoded = filterDispatchMod.decode(streamObj);
150
+ const parsed = crossRefStreamMod.parseCrossRefStream(decoded, streamObj.dict);
151
+ return { entries: parsed.entries, dict: streamObj.dict };
152
+ }
153
+
154
+ /**
155
+ * Merge the trailer dicts of every cross-reference section, given
156
+ * newest first: the newest dict is kept whole, and each document
157
+ * trailer key it lacks (`/Root`, `/Info`, `/ID`, `/Encrypt`, `/Size`,
158
+ * …) comes from the first older dict that carries it. Keys that
159
+ * describe one section only (`/Prev`, `/XRefStm` and the
160
+ * cross-reference stream's own `/Type`, `/W`, `/Index`, `/Length`,
161
+ * filter entries) are never inherited. A single dict is returned
162
+ * unchanged, and so is a non-dict, so typeTrailer keeps its
163
+ * pdf/trailer/not-dict refusal.
164
+ */
165
+ function mergeTrailerDicts(dicts) {
166
+ if (dicts.length === 1) return dicts[0];
167
+ for (const d of dicts) {
168
+ if (!d || d.type !== 'dict') return d;
169
+ }
170
+ const entries = { ...dicts[0].entries };
171
+ for (let i = 1; i < dicts.length; i++) {
172
+ for (const k of Object.keys(dicts[i].entries)) {
173
+ if (SECTION_LOCAL_KEYS.has(k)) continue;
174
+ if (!(k in entries)) entries[k] = dicts[i].entries[k];
175
+ }
176
+ }
177
+ return { ...dicts[0], entries };
178
+ }
179
+
180
+ function readHeader(bytes) {
181
+ if (bytes.length < 8) {
182
+ throw new ParseError('pdf/document/short',
183
+ 'input too short to contain a PDF header',
184
+ { context: { length: bytes.length } });
185
+ }
186
+ let off = -1;
187
+ const maxScan = Math.min(bytes.length - 5, 1024);
188
+ outer: for (let i = 0; i <= maxScan; i++) {
189
+ for (let k = 0; k < 5; k++) {
190
+ if (bytes[i + k] !== HEADER_PREFIX[k]) continue outer;
191
+ }
192
+ off = i; break;
193
+ }
194
+ if (off < 0) {
195
+ throw new ParseError('pdf/document/bad-header',
196
+ 'no %PDF- header found in first 1024 bytes');
197
+ }
198
+ let p = off + 5;
199
+ const digits = [];
200
+ while (p < bytes.length) {
201
+ const b = bytes[p];
202
+ if (b === 0x0A || b === 0x0D) break;
203
+ digits.push(b); p++;
204
+ }
205
+ const version = new TextDecoder('latin1').decode(Uint8Array.from(digits));
206
+ if (p < bytes.length && bytes[p] === 0x0D) p++;
207
+ if (p < bytes.length && bytes[p] === 0x0A) p++;
208
+ return { version, end: p };
209
+ }
210
+
211
+ function readDocument(bytes, opts = {}) {
212
+ if (!(bytes instanceof Uint8Array)) {
213
+ throw new ParseError('pdf/document/bad-input',
214
+ 'readDocument expects a Uint8Array',
215
+ { context: { 'typeof': typeof bytes } });
216
+ }
217
+ const { version, end: headerEnd } = readHeader(bytes);
218
+
219
+ const sxAt = locateStartXref(bytes);
220
+ if (sxAt < 0) {
221
+ throw new ParseError('pdf/document/no-startxref',
222
+ 'startxref keyword not found near EOF');
223
+ }
224
+ const xrefAt = readStartXref(bytes, sxAt);
225
+
226
+ const xref = { entries: {}, sections: [] };
227
+ // Read-path loss ledger: degradations the reader
228
+ // tolerated instead of throwing. Live array — later calls to
229
+ // `_raw.resolve` append to it too.
230
+ const losses = [];
231
+ // Every section's trailer dict, in visit order (newest first).
232
+ const trailerDicts = [];
233
+ let cursor = xrefAt;
234
+ const seenSections = new Set();
235
+ function mergeSection(at, kind, entries) {
236
+ xref.sections.push({ at, kind, entries });
237
+ for (const k of Object.keys(entries)) {
238
+ if (!(k in xref.entries)) xref.entries[k] = entries[k];
239
+ }
240
+ }
241
+ for (let safety = 0; safety < 32 && cursor >= 0; safety++) {
242
+ if (seenSections.has(cursor)) break;
243
+ seenSections.add(cursor);
244
+ let dict;
245
+ if (startsXrefTable(bytes, skipWhitespace(bytes, cursor))) {
246
+ const section = parseXrefTable(bytes, cursor);
247
+ mergeSection(cursor, 'table', section.entries);
248
+ dict = parseTrailerDict(bytes, section.end).dict;
249
+ // Hybrid-reference file (§7.5.8.4): the trailer points at
250
+ // a companion xref stream holding the entries a 1.4 reader
251
+ // is not expected to see. Its own /Prev is ignored — the
252
+ // classical chain drives the walk.
253
+ const stmAt = dict && dict.entries && dict.entries.XRefStm;
254
+ if (stmAt && stmAt.type === 'int'
255
+ && stmAt.value >= 0 && stmAt.value < bytes.length) {
256
+ const hybrid = readXrefStreamSection(bytes, stmAt.value);
257
+ mergeSection(stmAt.value, 'stream', hybrid.entries);
258
+ }
259
+ } else {
260
+ const section = readXrefStreamSection(bytes, cursor);
261
+ mergeSection(cursor, 'stream', section.entries);
262
+ dict = section.dict;
263
+ }
264
+ trailerDicts.push(dict);
265
+ // A section's own /Prev drives the walk; its other entries
266
+ // are typed once, on the merged trailer below.
267
+ const prev = dict && dict.type === 'dict' && dict.entries
268
+ && dict.entries.Prev;
269
+ if (prev && prev.type === 'int'
270
+ && prev.value >= 0 && prev.value !== cursor) {
271
+ cursor = prev.value;
272
+ } else break;
273
+ }
274
+ if (trailerDicts.length === 0) {
275
+ throw new ParseError('pdf/document/no-trailer',
276
+ 'no usable trailer dictionary found');
277
+ }
278
+ // The reader's trailer is the MERGE of every section's
279
+ // dict, newest first — an entry (notably /Root) comes from the
280
+ // newest section that supplies it, not only from the newest
281
+ // section. A linearized file's first-page xref stream carries
282
+ // /Root while the main stream it chains to does not; an
283
+ // incremental update may omit it the other way round.
284
+ // typeTrailer runs once, on the merge, so a chain where NO
285
+ // section supplies /Root still throws pdf/trailer/missing-root.
286
+ const trailerTyped = typeTrailer(mergeTrailerDicts(trailerDicts));
287
+
288
+ if (trailerTyped.encrypt && opts.allowEncrypted !== true) {
289
+ // Fail-loud arm — the trailer carries /Encrypt but
290
+ // this package composes no decrypt path (deferred, see
291
+ // pdf/document/document.js @fileoverview). Returning the
292
+ // model here would silently hand back ciphertext for
293
+ // strings/streams. Callers that actually want the raw
294
+ // encrypted container (tests, tooling) opt in explicitly.
295
+ throw new ParseError('pdf/document/encrypted',
296
+ 'document is encrypted (trailer /Encrypt present) — ' +
297
+ 'no decrypt path is composed for readDocument; pass ' +
298
+ '{ allowEncrypted: true } to read the raw ciphertext container',
299
+ { context: { encrypt: trailerTyped.encrypt } });
300
+ }
301
+
302
+ const indirects = new Map();
303
+ const lossKeys = new Set();
304
+ // Keys of objects every section marks free (read as null).
305
+ const freeKeys = new Set();
306
+ // Decoded members of every /Type /ObjStm container touched by
307
+ // this read — one decode + parse per container, per document.
308
+ const objStmMembers = new Map();
309
+
310
+ function resolve(ref) {
311
+ if (!ref || ref.type !== 'ref') return ref;
312
+ return resolveByKey(ref.num, ref.gen);
313
+ }
314
+ function resolveByKey(num, gen) {
315
+ const key = num + ':' + gen;
316
+ if (indirects.has(key)) return indirects.get(key).value;
317
+ let entry = xref.entries[num];
318
+ if (!entry) {
319
+ throw new ParseError('pdf/document/missing-xref',
320
+ `object ${num} ${gen} not in xref`,
321
+ { context: { num, gen } });
322
+ }
323
+ if (entry.free) {
324
+ entry = resolveFreeEntry(num, gen);
325
+ if (!entry) return PDF_NULL;
326
+ }
327
+ if (entry.type === 2) {
328
+ const value = resolveCompressed(num, gen, entry);
329
+ indirects.set(key, { value, offset: 0, objStm: entry.objStm });
330
+ return value;
331
+ }
332
+ const offset = entry.offset;
333
+ if (offset <= 0 || offset >= bytes.length) {
334
+ throw new ParseError('pdf/document/bad-offset',
335
+ `object ${num} ${gen} xref offset out of range`,
336
+ { context: { num, gen, offset, total: bytes.length } });
337
+ }
338
+ const tok = tokenize(bytes, { start: offset });
339
+ const def = parseIndirect(tok, resolve);
340
+ if (def.num !== num || def.gen !== gen) {
341
+ throw new ParseError('pdf/document/xref-mismatch',
342
+ `xref points to a different object`,
343
+ { context: { expected: { num, gen }, found: { num: def.num, gen: def.gen } } });
344
+ }
345
+ indirects.set(key, { value: def.value, offset });
346
+ return def.value;
347
+ }
348
+ /**
349
+ * The winning (newest) xref entry for `num` is free.
350
+ * Look through the sections newest first for the newest one that
351
+ * still DEFINES the object in use and resolve through it,
352
+ * recording `pdf/document/free-entry-fallback`. When every
353
+ * section agrees the object is free, record
354
+ * `pdf/document/free-object` and return null — ISO 32000-2
355
+ * §7.3.10 reads a reference to a free object as the null
356
+ * object — instead of throwing.
357
+ */
358
+ function resolveFreeEntry(num, gen) {
359
+ const key = num + ':' + gen;
360
+ for (const section of xref.sections) {
361
+ const e = section.entries[num];
362
+ if (e && !e.free) {
363
+ recordLoss('fallback:' + key, {
364
+ code: 'pdf/document/free-entry-fallback',
365
+ message: `object ${num} is free in the newest xref section; ` +
366
+ 'resolved through an older section that defines it',
367
+ context: { num, gen, section: { at: section.at, kind: section.kind } }
368
+ });
369
+ return e;
370
+ }
371
+ }
372
+ freeKeys.add(key);
373
+ recordLoss('free:' + key, {
374
+ code: 'pdf/document/free-object',
375
+ message: `object ${num} is free in every xref section; read as null`,
376
+ context: { num, gen }
377
+ });
378
+ return null;
379
+ }
380
+ /** Append `loss` to the ledger once per `dedupeKey`. */
381
+ function recordLoss(dedupeKey, loss) {
382
+ if (lossKeys.has(dedupeKey)) return;
383
+ lossKeys.add(dedupeKey);
384
+ losses.push(loss);
385
+ }
386
+ /**
387
+ * Materialise object `num` from the `/Type /ObjStm` container
388
+ * its type-2 xref entry names (§7.5.7). Compressed objects
389
+ * always carry generation 0.
390
+ */
391
+ function resolveCompressed(num, gen, entry) {
392
+ if (trailerTyped && trailerTyped.encrypt) {
393
+ throw new ParseError('pdf/document/objstm-encrypted',
394
+ 'object streams of an encrypted document cannot be ' +
395
+ 'read — no decrypt path is composed for readDocument',
396
+ { context: { num, gen, objStm: entry.objStm } });
397
+ }
398
+ const containerNum = entry.objStm;
399
+ let members = objStmMembers.get(containerNum);
400
+ if (!members) {
401
+ requireStreamWiring(0);
402
+ const containerEntry = xref.entries[containerNum];
403
+ if (containerEntry && containerEntry.type === 2) {
404
+ throw new ParseError('pdf/document/objstm-nested',
405
+ `object stream ${containerNum} is itself stored in an object stream`,
406
+ { context: { num, objStm: containerNum } });
407
+ }
408
+ const container = resolveByKey(containerNum, 0);
409
+ if (!container || container.type !== 'stream') {
410
+ throw new ParseError('pdf/document/objstm-not-stream',
411
+ `object ${containerNum} is not a stream and cannot hold compressed objects`,
412
+ { context: { num, objStm: containerNum,
413
+ type: container && container.type } });
414
+ }
415
+ members = objStreamMod.parseObjectStream(
416
+ filterDispatchMod.decode(container), container.dict);
417
+ objStmMembers.set(containerNum, members);
418
+ }
419
+ const member = members[entry.index];
420
+ if (!member || member.num !== num) {
421
+ throw new ParseError('pdf/document/objstm-mismatch',
422
+ 'object stream member does not carry the expected object number',
423
+ { context: { num, objStm: containerNum, index: entry.index,
424
+ found: member ? member.num : null } });
425
+ }
426
+ return member.value;
427
+ }
428
+
429
+ const rootRef = trailerTyped.root;
430
+ const catalogDict = resolve({ type: 'ref', num: rootRef.num, gen: rootRef.gen });
431
+ if (freeKeys.has(rootRef.num + ':' + rootRef.gen)) {
432
+ // No document without a catalog: a /Root free in every
433
+ // section stays a refusal, never a degraded read.
434
+ throw new ParseError('pdf/document/free-object',
435
+ `object ${rootRef.num} is free`,
436
+ { context: { num: rootRef.num, gen: rootRef.gen, role: 'catalog' } });
437
+ }
438
+ const catalog = typeCatalog(catalogDict);
439
+
440
+ // A page-tree /Kids reference to an object free in every section
441
+ // resolves to null; hand the walker an empty /Pages node in its
442
+ // place so the dangling kid contributes no page (its loss is
443
+ // already recorded) instead of failing pdf/pages/not-dict.
444
+ function resolvePageTreeNode(ref) {
445
+ const value = resolve(ref);
446
+ if (ref && ref.type === 'ref' && freeKeys.has(ref.num + ':' + ref.gen)) {
447
+ return EMPTY_PAGES_NODE;
448
+ }
449
+ return value;
450
+ }
451
+ const pageRefs = walkPageTree(catalog.pages, resolvePageTreeNode);
452
+ const pages = pageRefs.map(r => typePage(resolve({ type: 'ref', num: r.num, gen: r.gen })));
453
+
454
+ return {
455
+ version,
456
+ catalog,
457
+ pages,
458
+ trailer: trailerTyped,
459
+ xref,
460
+ losses,
461
+ _raw: {
462
+ resolve,
463
+ bytes,
464
+ headerEnd,
465
+ indirects
466
+ }
467
+ };
468
+ }
469
+
470
+ return { readDocument, readHeader };
471
+ }
472
+ };