pdf-codec.js 3.0.7 → 4.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +105 -54
- package/dist/{afm-widths-Dxucrg7D.js → afm-widths-BTu0IDp2.cjs} +1517 -4
- package/dist/{afm-widths-DyDq56Ph.d.cts → afm-widths-CneRyu5l.d.cts} +1 -1
- package/dist/{afm-widths-DyDq56Ph.d.ts → afm-widths-CneRyu5l.d.ts} +1 -1
- package/dist/{afm-widths-BoeTOK2r.cjs → afm-widths-De_QMWgC.js} +1416 -51
- package/dist/afm-widths.cjs +1 -1
- package/dist/afm-widths.d.cts +1 -1
- package/dist/afm-widths.d.ts +1 -1
- package/dist/afm-widths.js +1 -1
- package/dist/annotations.cjs +104 -0
- package/dist/annotations.d.cts +9 -0
- package/dist/annotations.d.ts +9 -0
- package/dist/annotations.js +103 -0
- package/dist/attachments.cjs +65 -0
- package/dist/attachments.d.cts +8 -0
- package/dist/attachments.d.ts +8 -0
- package/dist/attachments.js +64 -0
- package/dist/builtin-encoding.cjs +221 -0
- package/dist/builtin-encoding.d.cts +8 -0
- package/dist/builtin-encoding.d.ts +8 -0
- package/dist/builtin-encoding.js +220 -0
- package/dist/bytes/flate.cjs +1 -1
- package/dist/bytes/flate.js +1 -1
- package/dist/cff.cjs +14 -0
- package/dist/cff.d.cts +4 -1
- package/dist/cff.d.ts +4 -1
- package/dist/cff.js +12 -1
- package/dist/cmap-table.cjs +76 -13
- package/dist/cmap-table.d.cts +9 -1
- package/dist/cmap-table.d.ts +9 -1
- package/dist/cmap-table.js +77 -15
- package/dist/codec.cjs +1 -1
- package/dist/codec.d.cts +134 -0
- package/dist/codec.d.ts +134 -0
- package/dist/codec.js +1 -1
- package/dist/content-read.cjs +1 -1
- package/dist/content-read.d.cts +3 -3
- package/dist/content-read.d.ts +3 -3
- package/dist/content-read.js +1 -1
- package/dist/content-write.cjs +32 -10
- package/dist/content-write.d.cts +12 -5
- package/dist/content-write.d.ts +12 -5
- package/dist/content-write.js +32 -10
- package/dist/crypto/random.cjs +9 -0
- package/dist/crypto/random.d.cts +4 -0
- package/dist/crypto/random.d.ts +4 -0
- package/dist/crypto/random.js +8 -0
- package/dist/diagnostics.d.cts +1 -1
- package/dist/diagnostics.d.ts +1 -1
- package/dist/document.cjs +12 -3
- package/dist/document.d.cts +3 -1
- package/dist/document.d.ts +3 -1
- package/dist/document.js +12 -3
- package/dist/{embedded-font-fREdbxnr.d.ts → embedded-font-B4p9a3O3.d.ts} +3 -1
- package/dist/{embedded-font-BU6Jgcof.d.cts → embedded-font-oc65VGlr.d.cts} +3 -1
- package/dist/embedded-font-write.cjs +1 -1
- package/dist/embedded-font-write.d.cts +3 -3
- package/dist/embedded-font-write.d.ts +3 -3
- package/dist/embedded-font-write.js +1 -1
- package/dist/embedded-font.cjs +46 -15
- package/dist/embedded-font.d.cts +1 -1
- package/dist/embedded-font.d.ts +1 -1
- package/dist/embedded-font.js +46 -15
- package/dist/encoding.cjs +10 -1
- package/dist/encoding.d.cts +10 -1
- package/dist/encoding.d.ts +10 -1
- package/dist/encoding.js +2 -2
- package/dist/encrypt-write.cjs +234 -0
- package/dist/encrypt-write.d.cts +31 -0
- package/dist/encrypt-write.d.ts +31 -0
- package/dist/encrypt-write.js +231 -0
- package/dist/encrypt.cjs +51 -21
- package/dist/encrypt.d.cts +19 -3
- package/dist/encrypt.d.ts +19 -3
- package/dist/encrypt.js +36 -22
- package/dist/filters.cjs +18 -14
- package/dist/filters.d.cts +1 -1
- package/dist/filters.d.ts +1 -1
- package/dist/filters.js +18 -14
- package/dist/font-read.cjs +68 -13
- package/dist/font-read.d.cts +1 -1
- package/dist/font-read.d.ts +1 -1
- package/dist/font-read.js +69 -14
- package/dist/font-registry.cjs +1 -1
- package/dist/font-registry.d.cts +4 -4
- package/dist/font-registry.d.ts +4 -4
- package/dist/font-registry.js +1 -1
- package/dist/font-substitutes.d.cts +1 -1
- package/dist/font-substitutes.d.ts +1 -1
- package/dist/font-tables.cjs +31 -0
- package/dist/font-tables.d.cts +3 -1
- package/dist/font-tables.d.ts +3 -1
- package/dist/font-tables.js +31 -2
- package/dist/fonts.d.cts +2 -2
- package/dist/fonts.d.ts +2 -2
- package/dist/form.cjs +138 -0
- package/dist/form.d.cts +9 -0
- package/dist/form.d.ts +9 -0
- package/dist/form.js +137 -0
- package/dist/gsub-table-BRTQ8EUW.d.ts +10 -0
- package/dist/gsub-table-URczLd85.d.cts +10 -0
- package/dist/gsub-table.cjs +225 -0
- package/dist/gsub-table.d.cts +2 -0
- package/dist/gsub-table.d.ts +2 -0
- package/dist/gsub-table.js +224 -0
- package/dist/image/ccitt-encode.cjs +119 -0
- package/dist/image/ccitt-encode.d.cts +9 -0
- package/dist/image/ccitt-encode.d.ts +9 -0
- package/dist/image/ccitt-encode.js +118 -0
- package/dist/image/ccitt.cjs +3 -0
- package/dist/image/ccitt.d.cts +4 -1
- package/dist/image/ccitt.d.ts +4 -1
- package/dist/image/ccitt.js +1 -1
- package/dist/image/jbig2-bitmap.d.cts +1 -1
- package/dist/image/jbig2-bitmap.d.ts +1 -1
- package/dist/image/jbig2.cjs +3 -1
- package/dist/image/jbig2.js +3 -1
- package/dist/image/jp2-boxes.d.cts +1 -1
- package/dist/image/jp2-boxes.d.ts +1 -1
- package/dist/image/jpeg2000-codestream.d.cts +3 -3
- package/dist/image/jpeg2000-codestream.d.ts +3 -3
- package/dist/image/jpeg2000-t1.d.cts +1 -1
- package/dist/image/jpeg2000-t1.d.ts +1 -1
- package/dist/image/jpeg2000.cjs +1 -0
- package/dist/image/jpeg2000.d.cts +3 -0
- package/dist/image/jpeg2000.d.ts +3 -0
- package/dist/image/jpeg2000.js +1 -0
- package/dist/image/png-decode.cjs +1 -1
- package/dist/image/png-decode.js +1 -1
- package/dist/image/png-filter.d.cts +1 -1
- package/dist/image/png-filter.d.ts +1 -1
- package/dist/images-read.cjs +18 -12
- package/dist/images-read.d.cts +2 -2
- package/dist/images-read.d.ts +2 -2
- package/dist/images-read.js +15 -9
- package/dist/index.cjs +17 -4
- package/dist/index.d.cts +7 -6
- package/dist/index.d.ts +7 -6
- package/dist/index.js +6 -5
- package/dist/interpret.cjs +178 -56
- package/dist/interpret.d.cts +27 -12
- package/dist/interpret.d.ts +27 -12
- package/dist/interpret.js +178 -56
- package/dist/layout.cjs +149 -2
- package/dist/layout.d.cts +394 -1
- package/dist/layout.d.ts +394 -1
- package/dist/layout.js +140 -4
- package/dist/lexer.d.cts +9 -9
- package/dist/lexer.d.ts +9 -9
- package/dist/math-content-write.cjs +1 -1
- package/dist/math-content-write.js +1 -1
- package/dist/math-font-write.cjs +1 -1
- package/dist/math-font-write.d.cts +1 -1
- package/dist/math-font-write.d.ts +1 -1
- package/dist/math-font-write.js +1 -1
- package/dist/math-font.cjs +2 -2
- package/dist/math-font.d.cts +1 -1
- package/dist/math-font.d.ts +1 -1
- package/dist/math-font.js +2 -2
- package/dist/{math-stretch-a_rHNRvc.d.ts → math-stretch-BbC2o4Br.d.ts} +1 -1
- package/dist/{math-stretch-E26z6byT.d.cts → math-stretch-C2JanDAz.d.cts} +1 -1
- package/dist/math-stretch.d.cts +1 -1
- package/dist/math-stretch.d.ts +1 -1
- package/dist/measure.cjs +1 -1
- package/dist/measure.d.cts +1 -1
- package/dist/measure.d.ts +1 -1
- package/dist/measure.js +1 -1
- package/dist/names.cjs +41 -0
- package/dist/names.d.cts +11 -0
- package/dist/names.d.ts +11 -0
- package/dist/names.js +40 -0
- package/dist/navigation.cjs +210 -0
- package/dist/navigation.d.cts +18 -0
- package/dist/navigation.d.ts +18 -0
- package/dist/navigation.js +207 -0
- package/dist/notes-annotation-author.cjs +5 -0
- package/dist/notes-annotation-author.d.cts +4 -0
- package/dist/notes-annotation-author.d.ts +4 -0
- package/dist/notes-annotation-author.js +4 -0
- package/dist/objects-B7HdMeWS.d.cts +59 -0
- package/dist/objects-B7HdMeWS.d.ts +59 -0
- package/dist/objects.d.cts +1 -58
- package/dist/objects.d.ts +1 -58
- package/dist/optional-content.cjs +63 -0
- package/dist/optional-content.d.cts +12 -0
- package/dist/optional-content.d.ts +12 -0
- package/dist/optional-content.js +62 -0
- package/dist/parse.cjs +1 -1
- package/dist/parse.d.cts +2 -2
- package/dist/parse.d.ts +2 -2
- package/dist/parse.js +1 -1
- package/dist/pdf-text.cjs +21 -0
- package/dist/pdf-text.d.cts +5 -0
- package/dist/pdf-text.d.ts +5 -0
- package/dist/pdf-text.js +19 -0
- package/dist/predictors.d.cts +1 -1
- package/dist/predictors.d.ts +1 -1
- package/dist/read.cjs +287 -72
- package/dist/read.d.cts +2 -4
- package/dist/read.d.ts +2 -4
- package/dist/read.js +284 -67
- package/dist/serialize.cjs +5 -1
- package/dist/serialize.d.cts +3 -2
- package/dist/serialize.d.ts +3 -2
- package/dist/serialize.js +5 -2
- package/dist/sfnt-subset.cjs +10 -3
- package/dist/sfnt-subset.d.cts +1 -1
- package/dist/sfnt-subset.d.ts +1 -1
- package/dist/sfnt-subset.js +10 -3
- package/dist/structure.cjs +184 -0
- package/dist/structure.d.cts +12 -0
- package/dist/structure.d.ts +12 -0
- package/dist/structure.js +183 -0
- package/dist/tounicode.cjs +8 -3
- package/dist/tounicode.d.cts +2 -2
- package/dist/tounicode.d.ts +2 -2
- package/dist/tounicode.js +8 -3
- package/dist/util/base64.cjs +4 -4
- package/dist/util/base64.js +4 -4
- package/dist/winansi.cjs +1 -1
- package/dist/winansi.d.cts +1 -1
- package/dist/winansi.d.ts +1 -1
- package/dist/winansi.js +1 -1
- package/dist/write.cjs +464 -21
- package/dist/write.d.cts +4 -3
- package/dist/write.d.ts +4 -3
- package/dist/write.js +464 -20
- package/dist/xmp.cjs +46 -0
- package/dist/xmp.d.cts +14 -0
- package/dist/xmp.d.ts +14 -0
- package/dist/xmp.js +45 -0
- package/dist/xref.cjs +2 -2
- package/dist/xref.d.cts +3 -3
- package/dist/xref.d.ts +3 -3
- package/dist/xref.js +2 -2
- package/package.json +46 -31
- package/dist/image/png-encode.cjs +0 -64
- package/dist/image/png-encode.d.cts +0 -8
- package/dist/image/png-encode.d.ts +0 -8
- package/dist/image/png-encode.js +0 -63
package/README.md
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
# pdf-codec
|
|
2
2
|
|
|
3
|
-
[](https://github.com/ExaDev/pdf-codec) [](https://www.npmjs.com/package/pdf-codec) [](https://github.com/ExaDev/documents.js/tree/main/packages/pdf-codec) [](https://www.npmjs.com/package/pdf-codec) [](https://www.npmjs.com/package/pdf-codec) [](https://github.com/ExaDev/documents.js/actions)
|
|
4
4
|
|
|
5
5
|
> A hand-written, dependency-minimal PDF codec: parses arbitrary real-world PDFs into a structured, positioned-content document and generates new PDFs from one, built on its own codec-owned `LayoutDocument` item model and [Zod 4](https://zod.dev) codecs.
|
|
6
6
|
|
|
7
|
-
`pdf-codec` is the PDF-reading-and-writing half of [`documents.js`](https://github.com/ExaDev/documents.js), extracted into its own package: every layer of the PDF format — the object model, the cross-reference table, the content-stream operators, standard-font metrics, the parser's cross-reference/object-stream resolution and content-stream interpreter — is hand-written against the ISO 32000-1 specification, with no external PDF library (`pdf-lib`, `pdfjs-dist`, `mupdf`, or any other) as a dependency. The one exception is [`fflate`](https://github.com/101arrowz/fflate) for raw DEFLATE/zlib compression. The OpenType/CFF font parsing this package's own writer uses to embed a real math font (`sfnt.ts`/`math-*.ts`) is hand-written too, as are the cryptographic primitives
|
|
7
|
+
`pdf-codec` is the PDF-reading-and-writing half of [`documents.js`](https://github.com/ExaDev/documents.js), extracted into its own package: every layer of the PDF format — the object model, the cross-reference table, the content-stream operators, standard-font metrics, the parser's cross-reference/object-stream resolution and content-stream interpreter — is hand-written against the ISO 32000-1 specification, with no external PDF library (`pdf-lib`, `pdfjs-dist`, `mupdf`, or any other) as a dependency. The one exception is [`fflate`](https://github.com/101arrowz/fflate) for raw DEFLATE/zlib compression. The OpenType/CFF font parsing this package's own writer uses to embed a real math font (`sfnt.ts`/`math-*.ts`) is hand-written too, as are the cryptographic primitives the reader needs to open an encrypted PDF and the writer needs to produce one (`crypto/` — MD5, SHA-2, RC4, AES, and a `getRandomValues`-backed CSPRNG for the writer's own salts/keys/IVs), because `node:crypto` would end this package's platform neutrality and WebCrypto's `crypto.subtle` offers neither MD5 nor RC4 nor a synchronous API (`crypto.getRandomValues` itself is synchronous and used directly). The one bundled binary asset is the vendored STIX Two Math font itself (OFL-1.1, see [Fidelity](#fidelity) and `assets/fonts/NOTICE.md`).
|
|
8
8
|
|
|
9
9
|
This is a genuinely large undertaking with an honest trade-off spelled out in [Fidelity](#fidelity): this is not, and does not attempt to be, as robust against adversarial or badly malformed real-world PDFs as a library with 15+ years of hardening. What it buys instead is a dependency-free, fully auditable PDF implementation with no supply-chain surface beyond `document-schema.js`, `fflate`, and `zod`.
|
|
10
10
|
|
|
11
|
-
`documents.js` uses this package to convert docx/pptx/odt/odp/ods/odg to and from PDF, and to render MathML formulas (typeset by its own `src/mathml/` engine) through the embedded math font this package parses and writes. That MathML
|
|
11
|
+
`documents.js` uses this package to convert docx/pptx/odt/odp/ods/odg to and from PDF, and to render MathML formulas (typeset by its own `src/mathml/` engine) through the embedded math font this package parses and writes. That MathML _layout_ engine deliberately stays in `documents.js` — see [Architecture](#architecture) for exactly where the boundary sits and why a real `MathBox` value crosses it with zero cast or wrapper. [`document-outline.js`](https://github.com/ExaDev/documents.js/tree/main/packages/document-outline.js)'s `segmentPdfRegions` ([ExaDev/documents.js#931](https://github.com/ExaDev/documents.js/issues/931)) is a second, independent consumer of this package's `LayoutPage`/`LayoutItem` types alone — a recursive X-Y cut page-segmentation pass classifying a page's own positioned items into columns, tables, figures, and captions, entirely without the semantic reconstruction this package deliberately keeps out of itself.
|
|
12
12
|
|
|
13
13
|
```mermaid
|
|
14
14
|
graph TD
|
|
@@ -21,33 +21,39 @@ graph TD
|
|
|
21
21
|
documents("documents.js")
|
|
22
22
|
mcp("document-mcp")
|
|
23
23
|
cli("document-cli")
|
|
24
|
+
outline("document-outline.js")
|
|
24
25
|
|
|
25
26
|
schema --> ooxml
|
|
26
27
|
schema --> odf
|
|
27
28
|
schema --> pdfcodec
|
|
28
29
|
schema --> mdcodec
|
|
29
30
|
schema --> documents
|
|
31
|
+
schema --> outline
|
|
30
32
|
ooxml --> documents
|
|
31
33
|
odf --> documents
|
|
32
34
|
pdfcodec --> documents
|
|
33
35
|
mdcodec --> documents
|
|
34
36
|
bytecodec --> pdfcodec
|
|
35
37
|
bytecodec --> documents
|
|
38
|
+
pdfcodec --> outline
|
|
36
39
|
documents --> mcp
|
|
37
40
|
pdfcodec --> mcp
|
|
38
41
|
documents --> cli
|
|
39
42
|
odf --> cli
|
|
40
43
|
pdfcodec --> cli
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
click
|
|
45
|
-
click
|
|
46
|
-
click
|
|
47
|
-
click
|
|
44
|
+
outline --> mcp
|
|
45
|
+
outline --> cli
|
|
46
|
+
|
|
47
|
+
click schema "https://github.com/ExaDev/documents.js/tree/main/packages/document-schema.js" "document-schema.js"
|
|
48
|
+
click ooxml "https://github.com/ExaDev/documents.js/tree/main/packages/ooxml.js" "ooxml.js"
|
|
49
|
+
click odf "https://github.com/ExaDev/documents.js/tree/main/packages/odf.js" "odf.js"
|
|
50
|
+
click pdfcodec "https://github.com/ExaDev/documents.js/tree/main/packages/pdf-codec" "pdf-codec"
|
|
51
|
+
click mdcodec "https://github.com/ExaDev/documents.js/tree/main/packages/markdown-codec" "markdown-codec"
|
|
52
|
+
click bytecodec "https://github.com/ExaDev/documents.js/tree/main/packages/byte-codec" "byte-codec"
|
|
48
53
|
click documents "https://github.com/ExaDev/documents.js" "documents.js"
|
|
49
|
-
click mcp "https://github.com/ExaDev/document-mcp" "document-mcp"
|
|
50
|
-
click cli "https://github.com/ExaDev/document-cli" "document-cli"
|
|
54
|
+
click mcp "https://github.com/ExaDev/documents.js/tree/main/packages/document-mcp" "document-mcp"
|
|
55
|
+
click cli "https://github.com/ExaDev/documents.js/tree/main/packages/document-cli" "document-cli"
|
|
56
|
+
click outline "https://github.com/ExaDev/documents.js/tree/main/packages/document-outline.js" "document-outline.js"
|
|
51
57
|
|
|
52
58
|
style pdfcodec fill:#f9a825,stroke:#333,stroke-width:3px
|
|
53
59
|
```
|
|
@@ -73,7 +79,7 @@ npm install pdf-codec
|
|
|
73
79
|
Reading and writing PDF bytes:
|
|
74
80
|
|
|
75
81
|
```ts
|
|
76
|
-
import { readPdf, writePdf } from
|
|
82
|
+
import { readPdf, writePdf } from "pdf-codec";
|
|
77
83
|
|
|
78
84
|
const layout = readPdf(pdfBytes); // -> LayoutDocument: pages of positioned text/image/rect/line/ellipse/path/link items
|
|
79
85
|
const bytes = writePdf(layout);
|
|
@@ -81,17 +87,21 @@ const bytes = writePdf(layout);
|
|
|
81
87
|
|
|
82
88
|
`LayoutDocument` and its whole item family — every item/page/image-asset type and schema, plus `LAYOUT_FORMAT_VERSION` — are this package's own exports, ported from `document-schema.js` (which dropped them) so a codec's native model lives in the codec, the same family pattern as `ooxml.js`'s `Package`/`XmlElement` and `markdown-codec`'s AST. `readPdf`/`writePdf` keep their signatures; callers see the same names from a new home. `documents.js` re-exports the family onward from its own barrel — those re-exports now source from `pdf-codec` rather than `document-schema.js`, same names, new source.
|
|
83
89
|
|
|
84
|
-
There is deliberately **no `
|
|
90
|
+
There is deliberately **no `DocumentTree`-returning read or `DocumentTree`-accepting write here**, and `readPdf`/`writePdf` are this package's primary API precisely because of that. Only `markdown-codec`'s own read entry point returns `document-schema.js`'s flat `ContentDocument` directly — `ooxml.js`'s `readDocx`/`readPptx` return `DocxDocument`/`PptxDocument` and `odf.js`'s `readOdt`/`readOds` return `OdtDocument`/`OdsDocument`, each codec's own native model, with only `ooxml.js`'s separate `readXlsxContent` producing a `ContentDocument` outright. What lets `documents.js` offer one tree-native entry point uniformly across those formats (`decompose`/`assembleTree` outward, `flattenTree` back) is its own `readXContent` wrapper layer (`readDocxContent`, `readOdtContent`, and siblings), projecting each codec's native model into the flat `ContentDocument` — a step the codecs themselves don't take. PDF has no such wrapper to project through: `readPdf` yields _layout_ cheaply, because positioned glyphs and paths are all the format actually states, and semantic content only through a separate, expensive, lossy reconstruction pass that infers paragraphs, headings, tables, and shapes back out of geometry. That inference is semantic policy rather than codec business, so it lives in `documents.js` — a caller wanting a PDF as a `DocumentTree` passes `onDocument` to a named conversion function or to `convertDocument` itself and reads the tree off that callback (`convertDocument` on its own returns only bytes), or uses `createLocalDocumentConverter()`'s `DocumentConverter` port, whose `ConversionResult.package` is populated by wiring that same callback internally. The write direction is asymmetric for a different reason than it might look: turning a `DocumentTree` into PDF bytes is not itself a layout pass — `documents.js`'s `layoutDocumentFromPackage` is a mechanical inverse that walks the positions a _prior_ layout pass already stamped onto the package's own content nodes as `frames`, and it only works at all when those frames exist (a bridge conversion's own dump, e.g. `odt-to-docx`, carries no `pages` and cannot reach PDF this way). The actual font-measuring, line-breaking engine runs earlier, wherever the package first passed through an X-to-PDF or PDF-to-X conversion — both the frame-stamping and the frame-walking are `documents.js`'s. Keeping both edges out of this package is still what makes the item layer an honest record of what a file says, separate from what any consumer thinks it means.
|
|
85
91
|
|
|
86
|
-
An encrypted PDF that opens without a password decrypts transparently — no extra option, no password parameter; one that genuinely needs a user password throws `PdfPasswordRequiredError`. See [Gotchas](#gotchas-and-quirks) for exactly which encryption is supported.
|
|
92
|
+
An encrypted PDF that opens without a password decrypts transparently — no extra option, no password parameter; one that genuinely needs a user password throws `PdfPasswordRequiredError`. `writePdf` encrypts on the way out via its own `encryption` option (`WritePdfOptions.encryption`, `PdfEncryptionOptions` from `pdf-codec`) — a user password, an optional owner password, a choice of scheme (`"rc4-40" | "rc4-128" | "aes-128" | "aes-256"`, default `"aes-256"`), and permission flags. See [Gotchas](#gotchas-and-quirks) for exactly which encryption is supported on both sides.
|
|
87
93
|
|
|
88
94
|
Both accept an optional `signal` (`AbortSignal`); `readPdf` additionally takes a `sink` (`PdfDiagnosticSink`, called once per recoverable parse diagnostic — see the three-tier failure policy under [Conventions](#conventions)), and `writePdf` an `onSubstitution` callback (called once per character not representable in a standard-14 font — see [Fidelity](#fidelity)).
|
|
89
95
|
|
|
96
|
+
**Page boundaries: the crop box is the visible region** (ISO 32000-1 14.11.2). A page's reported `widthPt`/`heightPt` and coordinate frame come from its effective `/CropBox` — the rectangle a viewer displays and prints, inherited through the page tree like `/MediaBox` and defaulting to it — not from the media box: content an author placed outside the crop box (printer's marks, bleed, off-page slugs) is not visible and does not extract. Content wholly outside the crop box is dropped; content straddling the boundary keeps its original unclipped geometry (the item layer records what the file states — a viewer's clipping is a rendering fact, not source data); link annotations are anchored constructs rather than painted content and are never filtered, the same line the optional-content filter draws. The declared boundary rectangles a distinct crop box hides — plus `/BleedBox`/`/TrimBox`/`/ArtBox`, print-production facts with no field in the layout model — are quarantined verbatim as the `page-boxes` package-level residue row rather than silently dropped. A degenerate `/CropBox` (zero width or height) falls back to the media box with a `pdf/invalid-crop-box` warning.
|
|
97
|
+
|
|
98
|
+
**Cancellation granularity and cost, for CPU-metered runtimes.** Both pipelines are synchronous end to end — there is no `await` point for cancellation to hook into implicitly — so the `signal` is checked explicitly, once per page-loop iteration (and once before `readPdf`'s document-open phase begins). A signal aborted mid-parse therefore takes effect at the next page boundary, not instantly: `readPdf`'s document-open phase (cross-reference resolution, object parsing) and a single page's content-stream interpretation are the two spans that cannot be interrupted, and a document consisting of one enormous page is effectively uninterruptible however many pages it claims. Cost is roughly linear in decompressed content length, so budget for the worst single page, not the page count. On Cloudflare Workers this is the honest shape of the trade: the parse holds the isolate for its whole duration with no opportunity to yield or report progress, and an `AbortSignal` shared with whatever can abort concurrently (a binding, another context) makes a deadline enforceable at page granularity — but it cannot convert a synchronous parse into a resumable one. An async page-at-a-time API is a deliberate non-goal of this package's current surface.
|
|
99
|
+
|
|
90
100
|
The same round trip is also available as a schema-validated [`z.codec()`](https://zod.dev) pair:
|
|
91
101
|
|
|
92
102
|
```ts
|
|
93
|
-
import { z } from
|
|
94
|
-
import { pdfCodec } from
|
|
103
|
+
import { z } from "zod";
|
|
104
|
+
import { pdfCodec } from "pdf-codec";
|
|
95
105
|
|
|
96
106
|
const layout = z.decode(pdfCodec, pdfBytes); // throws a ZodError if pdfBytes has no %PDF- header
|
|
97
107
|
const pdfBytes2 = z.encode(pdfCodec, layout);
|
|
@@ -102,13 +112,19 @@ This is the no-extra-options form only — `readPdf`/`writePdf` remain the entry
|
|
|
102
112
|
Embedding a real math formula: `writePdf`'s own `formulas` option takes an array of `PositionedFormula` — an already-laid-out `MathBox` (positioned glyph runs, fraction/radical rules, radical hook strokes) placed at a page position. This package supplies the font — `loadMathFont()` parses and caches the vendored STIX Two Math font once per process — but it does not lay MathML out itself; that is a separate concern this package deliberately doesn't own (see [Architecture](#architecture)).
|
|
103
113
|
|
|
104
114
|
```ts
|
|
105
|
-
import { loadMathFont, writePdf } from
|
|
106
|
-
import { layoutFormula } from
|
|
115
|
+
import { loadMathFont, writePdf } from "pdf-codec";
|
|
116
|
+
import { layoutFormula } from "documents.js"; // or any other producer of a structurally-compatible MathBox
|
|
107
117
|
|
|
108
118
|
const { metricsAt } = loadMathFont();
|
|
109
|
-
const { box } = layoutFormula(mathml, {
|
|
110
|
-
|
|
111
|
-
|
|
119
|
+
const { box } = layoutFormula(mathml, {
|
|
120
|
+
metrics: metricsAt(12),
|
|
121
|
+
sizePt: 12,
|
|
122
|
+
color: { r: 0, g: 0, b: 0 },
|
|
123
|
+
});
|
|
124
|
+
|
|
125
|
+
const pdfBytes = writePdf(doc, {
|
|
126
|
+
formulas: [{ pageIndex: 0, xPt: 50, yPt: 700, box }],
|
|
127
|
+
});
|
|
112
128
|
```
|
|
113
129
|
|
|
114
130
|
Because `MathBox` and its own constituent types (`MathGlyphRun`/`MathRule`/`MathStroke`/`MathAssembledGlyphs`/`MathColor`) are plain, structurally-typed data — not a class, not branded — any producer whose output matches the shape works here with no cast, no wrapper, and no transformation.
|
|
@@ -116,10 +132,10 @@ Because `MathBox` and its own constituent types (`MathGlyphRun`/`MathRule`/`Math
|
|
|
116
132
|
Sizing a stretchy glyph — a parenthesis tall enough to wrap a big fraction, a radical sign sized to its radicand, an over-brace as wide as the content under it — is the OpenType `MATH` table's `MathVariants` job, and `loadMathFont()` exposes it directly:
|
|
117
133
|
|
|
118
134
|
```ts
|
|
119
|
-
import { loadMathFont } from
|
|
135
|
+
import { loadMathFont } from "pdf-codec";
|
|
120
136
|
|
|
121
137
|
const { stretchGlyph } = loadMathFont();
|
|
122
|
-
const paren = stretchGlyph(0x28,
|
|
138
|
+
const paren = stretchGlyph(0x28, "vertical", 40, 12); // a '(' stretched to 40pt, set at 12pt
|
|
123
139
|
// paren.kind -> 'assembly' (no single pre-built variant reaches 40pt)
|
|
124
140
|
// paren.size -> the extent actually achieved, >= 40 whenever the font can reach it
|
|
125
141
|
// paren.placements -> [{ glyphId, offset, advance }, ...], bottom to top, seams already overlapped
|
|
@@ -136,7 +152,7 @@ Building a layout engine on top of this codec (this is what `documents.js`'s own
|
|
|
136
152
|
Embedding real fonts instead of substituting standard-14 faces, via a `FontRegistry` (see `src/font-registry.ts` for the source-document → caller-supplied → vendored-substitute → standard-14 resolution order). Pass the same registry to both the measurer and `writePdf` so what was measured and what gets drawn come from one font:
|
|
137
153
|
|
|
138
154
|
```ts
|
|
139
|
-
import { createFontMeasurer, createFontRegistry, writePdf } from
|
|
155
|
+
import { createFontMeasurer, createFontRegistry, writePdf } from "pdf-codec";
|
|
140
156
|
|
|
141
157
|
// With no `fonts`/`sourceFonts` of its own, the registry still maps Calibri onto the vendored,
|
|
142
158
|
// metric-compatible Carlito face this package embeds (and Cambria onto Caladea).
|
|
@@ -144,7 +160,10 @@ const fonts = createFontRegistry();
|
|
|
144
160
|
|
|
145
161
|
const measurer = createFontMeasurer(fonts);
|
|
146
162
|
// ... wrap text + build a LayoutDocument using the measurer (documents.js owns wrapRunsToWidth) ...
|
|
147
|
-
const pdfBytes = writePdf(doc, {
|
|
163
|
+
const pdfBytes = writePdf(doc, {
|
|
164
|
+
fonts,
|
|
165
|
+
onMissingGlyph: (m) => console.warn("no glyph for", m.from),
|
|
166
|
+
});
|
|
148
167
|
```
|
|
149
168
|
|
|
150
169
|
Every text run whose family resolves to a real face is subsetted to the glyphs the document actually uses and embedded as its own `/Type0` + `/CIDFontType2` + `/FontFile2` group. **Omit `fonts` and nothing changes at all** — output is byte-identical to a build with no embedded-font support (asserted against golden digests in `src/write-embedded-font.test.ts`).
|
|
@@ -152,10 +171,15 @@ Every text run whose family resolves to a real face is subsetted to the glyphs t
|
|
|
152
171
|
Inspecting a standalone font file before handing it to a `FontRegistry` as a `ProvidedFont`: `readFontFace` reads family/bold/italic straight off the font's own `name`/`OS/2`/`head` tables — exactly the triple `ProvidedFont` needs:
|
|
153
172
|
|
|
154
173
|
```ts
|
|
155
|
-
import { createFontRegistry, readFontFace } from
|
|
156
|
-
|
|
157
|
-
const { family, bold, italic } = readFontFace(
|
|
158
|
-
|
|
174
|
+
import { createFontRegistry, readFontFace } from "pdf-codec";
|
|
175
|
+
|
|
176
|
+
const { family, bold, italic } = readFontFace(
|
|
177
|
+
brandSansTtfBytes,
|
|
178
|
+
"BrandSans-Bold.ttf",
|
|
179
|
+
); // throws FontFaceParseError, naming the source, for a .ttc/.woff file or one with no family name
|
|
180
|
+
const fonts = createFontRegistry({
|
|
181
|
+
fonts: [{ family, bold, italic, bytes: brandSansTtfBytes }],
|
|
182
|
+
});
|
|
159
183
|
```
|
|
160
184
|
|
|
161
185
|
`createFontMeasurer`'s second argument carries `verticalMetrics`, a `VerticalMetricPolicy` of `'hhea'` (default) / `'os2Typo'` / `'os2Win'`, deciding which of the three competing ascent/descent/line-gap sets an sfnt declares should drive line height for an embedded face.
|
|
@@ -163,28 +187,43 @@ const fonts = createFontRegistry({ fonts: [{ family, bold, italic, bytes: brandS
|
|
|
163
187
|
Reading a JPEG 2000 image directly, either as pixels or as metadata alone:
|
|
164
188
|
|
|
165
189
|
```ts
|
|
166
|
-
import { decodeJpeg2000, readJpeg2000Metadata } from
|
|
190
|
+
import { decodeJpeg2000, readJpeg2000Metadata } from "pdf-codec";
|
|
167
191
|
|
|
168
192
|
// Works on any conforming codestream -- including one whose pixels this decoder refuses, which is what `decodable`/`undecodableReason` are for.
|
|
169
193
|
const metadata = readJpeg2000Metadata(jp2OrCodestreamBytes);
|
|
170
|
-
console.log(
|
|
194
|
+
console.log(
|
|
195
|
+
metadata.width,
|
|
196
|
+
metadata.height,
|
|
197
|
+
metadata.transform,
|
|
198
|
+
metadata.layers,
|
|
199
|
+
metadata.decodable,
|
|
200
|
+
metadata.undecodableReason,
|
|
201
|
+
);
|
|
171
202
|
|
|
172
203
|
const image = decodeJpeg2000(jp2OrCodestreamBytes); // -> { width, height, bitDepth, components: Int32Array[] }, one plane per component
|
|
173
204
|
```
|
|
174
205
|
|
|
175
206
|
Both accept a whole JP2 file or a bare codestream; `decodeJpeg2000` takes an optional `onWarning` for recoverable cases and throws `Jpeg2000UnsupportedError` for anything outside [JPEG 2000 scope](#jpeg-2000-scope). Inside a PDF none of this needs calling: `readPdf` decodes a `/JPXDecode` image XObject through the same path automatically.
|
|
176
207
|
|
|
177
|
-
The generic byte- and image-container primitives (`crc32`, `deflate`/`inflate`/`inflateTolerant`, `ByteReader`/`ByteWriter`/`concatBytes`, `readJpegInfo`, `decodePng`/`encodePng`, `unfilterScanlines`/`filterScanlines`) are re-exported from [byte-codec](
|
|
208
|
+
The generic byte- and image-container primitives (`crc32`, `deflate`/`inflate`/`inflateTolerant`, `ByteReader`/`ByteWriter`/`concatBytes`, `readJpegInfo`, `decodePng`/`encodePng`, `unfilterScanlines`/`filterScanlines`) are re-exported from [byte-codec](../byte-codec/README.md). The PDF-specific image codecs (`decodeCcittFax`, `decodeJbig2Embedded`, `decodeJpeg2000`/`readJpeg2000Metadata`/`parseJp2Container`) stay here.
|
|
178
209
|
|
|
179
210
|
Every module under `src/` is also deep-importable directly by its own subpath:
|
|
180
211
|
|
|
181
212
|
```ts
|
|
182
|
-
import { crc32 } from
|
|
183
|
-
import { readJpegInfo } from
|
|
213
|
+
import { crc32 } from "pdf-codec/bytes/crc32";
|
|
214
|
+
import { readJpegInfo } from "pdf-codec/image/jpeg-info";
|
|
184
215
|
```
|
|
185
216
|
|
|
186
217
|
This works via package.json's `"./*"` wildcard export, resolving any `pdf-codec/<path>` subpath to the correspondingly-named file under `dist/` — both ESM `import` and CJS `require` resolve the same way.
|
|
187
218
|
|
|
219
|
+
One subpath is a declared entry point in its own right: **`pdf-codec/read`** (an explicit `exports` entry onto `src/read.ts`, the read pipeline's own module). A consumer that only ever reads PDFs and imports the root barrel statically reaches `write.ts`'s module-scope imports — `math-font.ts` (the vendored STIX Two Math font) and `font-registry.ts` (the four Carlito/Caladea faces), together ~2.9 MB of font binaries it can never execute, which on Cloudflare Workers' free plan (3 MB gzipped for an entire Worker) is most of the budget. The read entry's module graph provably excludes them:
|
|
220
|
+
|
|
221
|
+
```ts
|
|
222
|
+
import { readPdf } from "pdf-codec/read";
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
`src/read-graph.test.ts` walks the entry's static import graph and fails the build if `write.ts`, `math-font.ts`, `font-registry.ts`, or any asset module becomes reachable (type-only imports are exempt — they erase at compile time — so a read-side module can keep typing against the root barrel's types while its runtime graph stays narrow). The entry carries `readPdf` and the read pipeline's own helpers (`normalizeRotation`, `pageRotationTransform`); the other read-adjacent surfaces it does not itself own stay deep-importable through the wildcard and are asset-free the same way — the diagnostics vocabulary (`pdf-codec/diagnostics`), the `LayoutDocument` item family (`pdf-codec/layout`), PDF string/date scalar decoding (`pdf-codec/pdf-text`), and standard-14 resolution/AFM metrics (`pdf-codec/fonts`, `pdf-codec/afm-widths`).
|
|
226
|
+
|
|
188
227
|
## Architecture
|
|
189
228
|
|
|
190
229
|
The package is layered from generic primitives outward to the codec itself:
|
|
@@ -193,16 +232,16 @@ The package is layered from generic primitives outward to the codec itself:
|
|
|
193
232
|
- **The math port types** — `MathColor`/`MathGlyphRun`/`MathRule`/`MathStroke`/`MathLayoutItem`/`MathBox`/`MathGlyphMetrics`/`MathFontMetrics` and `PositionedFormula`, sourced from `document-schema.js`'s math layout port (one shared definition across the family, not a local mirror). Deliberately not imported from `documents.js` — that would be a circular dependency once `documents.js` depends on this package. Because every one of these types is plain data (only `MathFontMetrics` carries a method), a real `MathBox` value `documents.js` produces passes into `writePdf({ formulas })` with zero cast, zero wrapper, and zero transformation.
|
|
194
233
|
- **`src/bytes/`** and **`src/image/`** — generic byte and image-container primitives with zero PDF-specific knowledge: a chunked byte writer, backtracking byte reader, CRC32, a hand-written PNG decoder/encoder, JPEG marker scanning for dimensions only (compressed bytes pass through unchanged), a hand-written CCITT Group 3/Group 4 fax decoder (ITU-T T.4/T.6), a hand-written JBIG2 decoder (ITU-T T.88 — `jbig2-arith.ts` MQ decoder, `jbig2-bitmap.ts`, `jbig2-generic.ts`, `jbig2-text.ts`, `jbig2.ts`), and a hand-written JPEG 2000 decoder (ISO/IEC 15444-1 — `jp2-boxes.ts`, `jpeg2000-codestream.ts`, `jpeg2000-tagtree.ts`, `jpeg2000-t2.ts`, `jpeg2000-t1.ts`, `jpeg2000-dwt.ts`, `jpeg2000.ts`). `src/filters.ts` owns all PDF knowledge for CCITT/JBIG2 (resolving parameters, `/JBIG2Globals`, inverting polarity); `src/images-read.ts` owns the JPEG 2000 PDF integration. `src/bytes/flate.ts` is the only file that imports `fflate`.
|
|
195
234
|
- **`src/util/`** — two small independently-duplicated copies of family-shared logic: `base64.ts` (verbatim copy of `odf.js`'s own, replacing a former `ooxml.js` dependency for this one helper) and `abort.ts` (`throwIfAborted`, called at every page loop boundary — there is no `await` point in this synchronous pipeline for cancellation to hook into implicitly).
|
|
196
|
-
- **`src/crypto/`** — MD5, SHA-256/384/512, RC4, and AES-CBC, hand-written with zero local imports. Not a preference: ISO 32000-1's key-derivation algorithms name MD5 and RC4 directly, neither offered by any portable platform crypto API, and `crypto.subtle` is asynchronous where this codec's read
|
|
235
|
+
- **`src/crypto/`** — MD5, SHA-256/384/512, RC4, and AES-CBC, hand-written with zero local imports, plus a thin `random.ts` wrapper around `globalThis.crypto.getRandomValues` for the writer's own salts/file keys/IVs. Not a preference: ISO 32000-1's key-derivation algorithms name MD5 and RC4 directly, neither offered by any portable platform crypto API, and `crypto.subtle` is asynchronous where this codec's read and write paths are both synchronous end to end — `getRandomValues` itself, unlike `crypto.subtle`, is synchronous and portable, so it is used directly rather than hand-written. Reaching for `node:crypto` would break the browser bundle. Each hash/cipher module cites its specification (RFC 1321, FIPS 180-4, FIPS 197) and is tested against published conformance vectors.
|
|
197
236
|
- **The codec itself, importing only `layout`/`bytes`/`image`/`crypto`/`util` plus `document-schema.js`'s port types (no OOXML or ODF knowledge):**
|
|
198
|
-
- **Write**: `objects.ts` (the `PdfObject` discriminated union), `afm-widths.ts`/`encoding.ts`/`winansi.ts`/`fonts.ts` (standard-14 metrics, WinAnsi encoding, family resolution), `font-registry.ts` (resolution port plus `resolveFaceWithRegistry`, the one step both measurer and writer resolve through so they can never disagree about which face a `LayoutFont` means), `font-face.ts` (`readFontFace`, reading a standalone font file's family/bold/italic triple off its `name`/`OS/2`/`head` tables), `measure.ts`/`text-layout.ts` (greedy line-wrapping against either standard-14 AFM widths plus per-family correction or a resolved face's own real `hmtx` advances — never both), `content-write.ts` (`LayoutItem[]` → content-stream operators, with text branching on standard-14 vs embedded face encoding, pair-kerning split into `TJ` arrays,
|
|
199
|
-
- **sfnt font tables**: `sfnt.ts` (bounds-checked table-directory reader), `cmap-table.ts` (
|
|
237
|
+
- **Write**: `objects.ts` (the `PdfObject` discriminated union), `afm-widths.ts`/`encoding.ts`/`winansi.ts`/`fonts.ts` (standard-14 metrics, WinAnsi encoding, family resolution), `font-registry.ts` (resolution port plus `resolveFaceWithRegistry`, the one step both measurer and writer resolve through so they can never disagree about which face a `LayoutFont` means), `font-face.ts` (`readFontFace`, reading a standalone font file's family/bold/italic triple off its `name`/`OS/2`/`head` tables), `measure.ts`/`text-layout.ts` (greedy line-wrapping against either standard-14 AFM widths plus per-family correction or a resolved face's own real `hmtx` advances — never both), `content-write.ts` (`LayoutItem[]` → content-stream operators, with text branching on standard-14 vs embedded face encoding, pair-kerning split into `TJ` arrays, stroke `style` becoming real dash/line-cap state, and `/P <</MCID n /OC R>> BDC`…`EMC` spans around items carrying a structure owner or an optional-content layer — the one marked-content spelling both read-side channels resolve through), `write.ts` (the full object graph, cross-reference table, trailer, and embedded font groups, plus the document-level round-trip surfaces: the `/Names /EmbeddedFiles` attachments tree and `/Outlines` bookmark tree, `/OCProperties` optional-content groups with each layer stated explicitly into the default configuration's `/ON` or `/OFF` list, the `/AcroForm` field tree with the merged-field/widget spelling (a single widget merges into its field dict; further widgets are separate `/Subtype /Widget` kids, and fully-qualified names decompose back into the `/T` chain), the `/StructTreeRoot` element tree with its `/ParentTree` number tree associating each marked item's MCID to its owning element, and the package-level residue rows restored inline onto the Catalog or trailer — the XMP packet as an uncompressed `/Metadata` stream, and any row whose serialisation references source-file objects skipped rather than emitted as a dangling reference).
|
|
238
|
+
- **sfnt font tables**: `sfnt.ts` (bounds-checked table-directory reader), `cmap-table.ts` (character code → glyph ID, formats 0/4/6/12, exposed both as the best Unicode lookup and as every subtable individually, since a (3, 0) or (1, 0) subtable is keyed by a font's own codes rather than by code points), `hmtx-table.ts` (per-glyph advance widths), `font-tables.ts` (`head`/`maxp`/`OS/2`/`post`/`name`, including `post`'s own glyph names where a font still carries them), `glyf.ts` (`loca` offset index, per-glyph headers, composite component records, `glyphInkBounds`), `math-table.ts` (OpenType `MATH` constants/glyph-info/variants subtables). `ot-layout-common.ts` (Coverage/ClassDef, stored as sorted glyph ranges searched by bisection). `gpos-table.ts` reads `GPOS` for exactly one thing: pair-advance kerning through the `kern` feature, both PairPos formats and LookupType 9 Extension indirection — mark attachment, cursive joining, and contextual positioning have no consumer here. Every parser degrades to `undefined` on a missing/truncated table rather than throwing.
|
|
200
239
|
- **sfnt subsetting**: `sfnt-subset.ts` — a TrueType-outline glyph subsetter (Unicode code points → glyph IDs via `cmap`, transitive closure over `glyf` composite components, rebuilt sfnt carrying only used outlines). **Glyph IDs are preserved, never renumbered**, keeping composite component references valid and making CID == GID trivially true. Output rebuilds `head`/`hhea`/`maxp`/`loca`/`glyf`/`hmtx`, copies hinting programs verbatim, stubs `post`, omits `cmap`/`name`/`OS/2`/`GSUB`/`GPOS`/`kern` (none read through a `CIDFontType2` program per ISO 32000-1 9.9). Applies to `glyf`-flavoured fonts only; CFF returns `undefined`.
|
|
201
240
|
- **Embedded math font**: `math-font.ts` (parses/caches the vendored STIX Two Math font, exposing size-specific `MathFontMetrics` and stretchy-glyph entry points), `math-stretch.ts` (OpenType MATH two-stage stretching: pick smallest pre-built variant reaching target, else assemble from repeated parts with seams overlapped), `math-font-write.ts` (builds the `/Type0`/`/CIDFontType0`/`/FontDescriptor`/`/FontFile3`/ToUnicode group), `math-content-write.ts` (`PositionedFormula[]` → content-stream bytes, Identity-H CIDs for text, `re`/`m`/`l` operators for rules, glyph-ID-addressed text objects for stretched constructions wrapped in `/ActualText`).
|
|
202
241
|
- **Embedded text faces**: `embedded-font.ts` (parses one TrueType-outline face's metrics and `GPOS` pair kerning, and `encodeForShowEmbedded` — the single code path both measurement and text-showing go through so encoding and measuring cannot disagree). Every geometry field is converted into PDF's 1000-units-per-em glyph space. `embedded-font-write.ts` builds the `/Type0`/`/CIDFontType2`/`/FontDescriptor`/`/FontFile2`/ToUnicode group, with `/CIDToGIDMap /Identity` written explicitly and `/Length1` set to the **uncompressed** subset length. Its subset tag is a CRC32 over the face's PostScript name and glyph-ID list, so identical input yields byte-identical output.
|
|
203
242
|
- **ToUnicode CMaps**: `tounicode.ts`, shared by both embedded-font writers — a character code → Unicode code point mapping written as a bfchar CMap (9.10.3), with supplementary-plane code points encoded as UTF-16BE surrogate pairs and entries emitted in blocks of at most 100.
|
|
204
243
|
- **CFF reading**: `cff.ts` (shared `INDEX`/`DICT` container structures), `cff-bounds.ts` (a Type 2 charstring interpreter computing each glyph's tight ink bounding box by tracking the current point through every path operator and solving each cubic's real extrema from the roots of its derivative — a path walker, not a rasteriser; verified against the vendored STIX Two Math font's whole 5,543-glyph repertoire, matching fontTools' `BoundsPen` to within 0.01 design units). `cff-probe.ts` reads a bare CFF program's header/Name INDEX/Top DICT to detect the `ROS` operator defining a CID-keyed font — the guard a future source-embedded-font phase needs before it can trust CID == GID against an arbitrary caller-supplied font.
|
|
205
|
-
- **Read**: `lexer.ts`/`parse.ts` (byte tokenizer and tokens → `PdfObject`), `filters.ts`/`predictors.ts` (Flate/LZW/ASCII85/ASCIIHex/RunLength/CCITTFax, TIFF/PNG predictors), `xref.ts`/`document.ts` (classic and cross-reference-stream resolution, object streams, `/Prev` chains, linear-scan recovery, the page tree with attribute inheritance), `encrypt.ts` (standard security handler: `/Encrypt` parsing, empty-user-password key derivation and `/U` verification, per-object keys, transparent string/stream decryption), `content-read.ts`/`interpret.ts` (content-stream tokenizer and graphics/text state machine, form-XObject recursion, general vector-path tracking), `cmap.ts`/`font-style.ts`/`font-read.ts` (`/ToUnicode` CMaps, font-dictionary resolution), `images-read.ts` (Image XObjects → PNG/JPEG bytes), `read.ts` (`readPdf`, assembling all of the above into a `LayoutDocument`).
|
|
244
|
+
- **Read**: `lexer.ts`/`parse.ts` (byte tokenizer and tokens → `PdfObject`), `filters.ts`/`predictors.ts` (Flate/LZW/ASCII85/ASCIIHex/RunLength/CCITTFax, TIFF/PNG predictors), `xref.ts`/`document.ts` (classic and cross-reference-stream resolution, object streams, `/Prev` chains, linear-scan recovery, the page tree with attribute inheritance), `encrypt.ts` (standard security handler: `/Encrypt` parsing, empty-user-password key derivation and `/U` verification, per-object keys, transparent string/stream decryption), `content-read.ts`/`interpret.ts` (content-stream tokenizer and graphics/text state machine, form-XObject recursion, general vector-path tracking), `cmap.ts`/`font-style.ts`/`font-read.ts` (`/ToUnicode` CMaps, font-dictionary resolution), `builtin-encoding.ts` (a font's own built-in encoding, read out of the embedded program itself — see [Text extraction and font encodings](#text-extraction-and-font-encodings)), `images-read.ts` (Image XObjects → PNG/JPEG bytes), `names.ts` (the document-level name-tree walker: one flattening pass for every `/Names` tenant), `navigation.ts` (named destinations from `/Dests` and `/Names` `/Dests`, reader-minted entries for direct destination arrays, and the `/Outlines` bookmark tree), `attachments.ts` (embedded files from the name tree, `/FileAttachment` filespecs, and `/AF`), `optional-content.ts` (`/OCProperties` groups and default-configuration visibility, plus the `/OC`-to-name resolution `interpret.ts` stamps onto span items), `annotations.ts` (sticky notes, FreeText, the `/QuadPoints` markup family, and residue for the opaque kinds), `form.ts` (the AcroForm field tree), `structure.ts` (the tagged-PDF `/StructTreeRoot` element tree with `/RoleMap` resolution and `/ClassMap` attribute merging, plus the `/ParentTree` walk that resolves each `(page, MCID)` pair to the owning element `read.ts` stamps onto an item), `xmp.ts` (a bounded Dublin Core extractor for the `/Metadata` packet), `pdf-text.ts` (PDF string/date scalar decoding), `read.ts` (`readPdf`, assembling all of the above into a `LayoutDocument`).
|
|
206
245
|
- `codec.ts` — `pdfCodec`, a `z.codec()` pair over `readPdf`/`writePdf`, plus a standalone local copy of the `%PDF-` header check.
|
|
207
246
|
- **`src/test-support/`** — hand-built PDF fixtures (`pdf.ts`) built by literal byte/string concatenation and deliberately importing NOTHING from this package's own writer (a fixture built by `writePdf` would let a writer bug hide from the corresponding reader test). `encrypted-pdfs.ts` applies the same principle: real PDFs encrypted by [qpdf](https://qpdf.sourceforge.io/), embedded as base64, so a bug in key derivation cannot cancel out between write and read halves. `fonts.ts` holds the real vendored Carlito and Caladea faces as raw sfnt bytes, and asserts values read out of the `.ttf` files by a standalone script with a bare `DataView`, not by this package's own parsers — external cross-checks rather than a parser's output compared against itself.
|
|
208
247
|
|
|
@@ -222,21 +261,33 @@ Dependency direction is strictly downward and checkable: `layout` imports only `
|
|
|
222
261
|
|
|
223
262
|
- **Reading arbitrary real-world PDFs is the single largest risk surface in this package**, and the parser targets cleanly-generated output from mainstream producers (Word, PowerPoint, Chrome, LibreOffice, Acrobat), recovering from the malformations those producers actually create, and failing loudly and specifically on anything else — not matching a mature library's robustness against adversarial input.
|
|
224
263
|
- **An encrypted PDF is readable when, and only when, it opens without a password** — the overwhelmingly common real-world case (a permissions-only file whose owner password may be set but whose user password is empty). Supported: `/Filter /Standard` at `/V` 1, 2, 4, and 5 — RC4-40, RC4-128, AES-128, AES-256 — including `/EncryptMetadata false` and `/Identity` crypt filters. A file genuinely needing a user password throws `PdfPasswordRequiredError` (distinct from `PdfEncryptedError`, because "supply the password" and "this codec cannot read this at all" are different things to tell a user).
|
|
225
|
-
- **Nothing in this codec accepts, prompts for, or guesses a password.** Authenticating as owner is a permissions escalation, not a way to read a file you were already allowed to read.
|
|
264
|
+
- **Nothing in this codec's reader accepts, prompts for, or guesses a password.** Authenticating as owner is a permissions escalation, not a way to read a file you were already allowed to read.
|
|
265
|
+
- **`writePdf` produces real, spec-conformant encryption for all four of the schemes above** (`src/encrypt-write.ts`, `WritePdfOptions.encryption`) — genuine `/O`/`/U`/`/OE`/`/UE`/`/Perms` computation per ISO 32000-2 7.6.4.4's Algorithms 3, 8, 9, and 10, a real user and (optional, defaulting to the user password) owner password, and per-`/P`-bit permission flags. Unlike the reader, the writer's own password is not scoped to empty: a genuinely non-empty user password is fully supported and round-trips against the published algorithm (see `encrypt-write.test.ts`), even though this package's own reader — scoped as above — can only ever open the result back up if the user password used to encrypt it was itself empty. Two boundaries are deliberately narrower than the full spec rather than approximated silently: a revision 2-4 (rc4-40/rc4-128/aes-128) password must be plain printable ASCII, not the full 8-bit PDFDocEncoding table, and a revision 6 (aes-256) password is Unicode-normalised with NFKC rather than the full SASLPrep/stringprep profile (prohibited-character tables and the bidirectional-text rule are not implemented). Both throw `PdfEncryptionError` for a password outside that scope rather than silently mis-encoding it into one that will not open the file it was meant to protect.
|
|
226
266
|
- **`CCITTFaxDecode`, `JBIG2Decode` and `JPXDecode` images all decode for real** via hand-written decoders (`src/image/ccitt.ts` for ITU-T T.4/T.6 fax; `src/image/jbig2*.ts` for ITU-T T.88; `src/image/jpeg2000*.ts` for ISO/IEC 15444-1). JPEG images (`DCTDecode`) pass through losslessly in both directions. See [JBIG2 scope](#jbig2-scope) and [JPEG 2000 scope](#jpeg-2000-scope) for exactly what is and is not implemented.
|
|
267
|
+
- **The write side re-encodes bilevel images as CCITT Group 4** (`src/image/ccitt-encode.ts`, a hand-written ITU-T T.6 encoder sharing the decoder's own code-table transcription): a grayscale image whose samples are all 0 or 255, with no soft mask, is written as `/CCITTFaxDecode` with `/K -1` whenever the G4 encoding comes out smaller than Flate over the same pixels — which for vertically coherent bilevel content (a scan's edges and text baselines) it does by an order of magnitude. Whichever encoding is smaller wins, deterministically, so decorrelated or repetitive bilevel content where Flate happens to win keeps Flate. **JBIG2 and JPEG 2000 re-encoding are permanent boundaries, not gaps**: a hand-written JBIG2 encoder is research-grade symbol-dictionary design and a JPEG 2000 encoder is the full EBCOT/wavelet stack, and re-encoding lossy JPX back is not fidelity-preserving anyway. Verbatim passthrough of a source's original JBIG2/JPX compressed bytes would need `LayoutImageAsset` and `document-schema.js`'s own `ContentImageBlock` to carry the original filter and bytes (the model normalises to decoded pixels at read time), a cross-package schema design change recorded in #975's scorecard rather than smuggled into this feature.
|
|
227
268
|
- **`interpret.ts` tracks general vector paths, not just axis-aligned `re` rectangles.** `m`/`l`/`c`/`v`/`y`/`h` (and `re` itself) accumulate real subpaths — CTM-transformed line/cubic segments, open or closed — and any paint operator emits an item built from them. A recovered path matching one of three characteristic shape patterns comes back as that shape's own kind, not a generic `LayoutPath`: an axis-aligned closed four-corner subpath is a `LayoutRect`; a closed subpath of four cubic segments meeting its bounding box at cardinal points with kappa-ratio control points is a `LayoutEllipse`; an open single-straight-segment stroke-only subpath is a `LayoutLine` (tolerance: `max(1e-3pt, 1e-4 × extent)`). **These are deliberate, bounded heuristics** — a false positive changes an item's kind, never its geometry, since every detected shape reproduces its source path's own points exactly. Off-axis rotations, non-quadrant curves, polygons, and multi-subpath figures stay a `LayoutPath`.
|
|
228
269
|
- **`writePdf`/`readPdf` round-trip a page's own `notes` field via a hidden `/Subtype /Text` annotation** with the `Hidden` flag set so it never renders or prints, distinguished from a genuine third-party sticky note by an internal author marker. This is a round-trip mechanism specific to this package's own writer/reader pair. `documents.js` uses this to carry pptx/odp speaker notes through PDF.
|
|
229
270
|
- **STIX Two Math is a CFF-flavoured OpenType font, not TrueType/glyf** — the **entire** `CFF ` table is embedded verbatim as a single `/FontFile3` `/Subtype /CIDFontType0C` stream (a real, correct, working embedded font, just not glyph-subsetted). Everything else genuinely IS built from a targeted parse of only what's used: `cmap` resolves exactly the Unicode code points a document's formulas reference, and the emitted `/W` widths array covers only drawn glyph IDs. A CID-keyed composite font built this way needs no `/CIDToGIDMap` — per ISO 32000-1 9.7.4.2, a `/CIDFontType0` whose `/FontFile3` is a non-CID-keyed CFF program is read with CID directly indexing `CharStrings` by glyph order (CID == GID).
|
|
230
271
|
- **The OpenType `MATH` table's `MathVariants` is parsed, stretchy-glyph assembly implemented, and the result genuinely drawable** via `MathAssembledGlyphs` (glyph-ID-addressed placements, since most construction glyphs have no Unicode code point — they draw directly because CID == GID here). An unencoded glyph gets no ToUnicode entry; the construction is wrapped in an `/ActualText` span carrying the operator's own text so it still extracts as `(`. `MathConstants` and `MathGlyphInfo` (italics correction, top-accent attachment) are parsed in full.
|
|
231
272
|
- **What this package draws for a stretchy glyph is decided entirely by its caller.** Which operators a document actually stretches is a layout-engine decision — `documents.js`'s own `src/mathml/layout.ts` currently stretches vertical fences in an `mrow` and nothing else.
|
|
232
|
-
- **Real per-glyph ink bounds are measured from the outline (`inkAscentPt`/`inkDescentPt`), computed by walking each glyph's Type 2 charstring.** `ascentPerEm`/`descentPerEm` remain alongside them as the uniform face-wide figure, still the right measure for anything sized against the font rather than particular characters, and the fallback for a glyph with no outline to measure. An ink box is genuinely tight, which for a math font is often
|
|
233
|
-
- **A `LayoutLine`/`LayoutPath` `style` of `dashed`/`dotted` becomes a real dash-array (`d`) operator scaled to the stroke's own width; `double` becomes two genuinely separate offset strokes.** Dash lengths are stroke-width multiples so a hairline and a thick rule both read as recognisably dashed: `dashed` emits `[3w 3w] 0 d`, `dotted` emits `[0 2w] 0 d` with a `1 J` round cap (the zero on-length under a round cap paints a filled circle — exactly a dot; under PDF's default butt cap it paints nothing). Both are reset immediately after the paint operator (`[] 0 d`, `0 J`) since the graphics state persists for the whole content stream. `double` has no PDF operator and is drawn as geometry: width `w` splits into three equal bands, each rule `w/3` wide with centreline `w/3` offset, outer edges matching the single stroke's.
|
|
273
|
+
- **Real per-glyph ink bounds are measured from the outline (`inkAscentPt`/`inkDescentPt`), computed by walking each glyph's Type 2 charstring.** `ascentPerEm`/`descentPerEm` remain alongside them as the uniform face-wide figure, still the right measure for anything sized against the font rather than particular characters, and the fallback for a glyph with no outline to measure. An ink box is genuinely tight, which for a math font is often _larger_ than the nominal metrics (over a tenth of STIX Two Math's repertoire draws above its nominal ascent). `inkDescentPt` is negative where the glyph's lowest ink sits above the baseline.
|
|
274
|
+
- **A `LayoutLine`/`LayoutPath` `style` of `dashed`/`dotted` becomes a real dash-array (`d`) operator scaled to the stroke's own width; `double` becomes two genuinely separate offset strokes.** Dash lengths are stroke-width multiples so a hairline and a thick rule both read as recognisably dashed: `dashed` emits `[3w 3w] 0 d`, `dotted` emits `[0 2w] 0 d` with a `1 J` round cap (the zero on-length under a round cap paints a filled circle — exactly a dot; under PDF's default butt cap it paints nothing). Both are reset immediately after the paint operator (`[] 0 d`, `0 J`) since the graphics state persists for the whole content stream. `double` has no PDF operator and is drawn as geometry: width `w` splits into three equal bands, each rule `w/3` wide with centreline `w/3` offset, outer edges matching the single stroke's. The read side tracks the `d` operator's own dash array as graphics state and recovers `dashed`/`dotted` back onto a stroked `LayoutLine`/`LayoutPath` from it — a zero on-length reads as `dotted`, any other non-empty array as `dashed`, an empty array (or none at all) as `solid`. `double` has no operator of its own to recover from, so it reads back as whatever geometry its two offset strokes actually are (typically two plain solid lines or paths), not as `double`.
|
|
234
275
|
- **An embedded `CIDFontType2` program needs no `cmap` table of its own**, and `sfnt-subset.ts`'s output doesn't carry one — character code → CID goes through the `Type0` font's `/Encoding` (Identity-H, so CID == character code), and CID → GID through `/CIDToGIDMap /Identity` (matching the GID-preserving design). Both happen inside the PDF's object graph, before the embedded font program is consulted (ISO 32000-1 9.7.4.2).
|
|
235
|
-
- **`GPOS` pair kerning
|
|
276
|
+
- **`GPOS` pair kerning and `GSUB`'s default ligature features are read and applied for an embedded face; the contextual and opt-in layout features are not.** An `fi` in a face whose `liga` feature declares the ligature draws as that real ligature glyph (advances, kerning, subsetting, and the ToUnicode CMap all describing the one substituted sequence — a ligature's ToUnicode entry maps back to its whole character run, so copy/paste recovers `fi`, not one glyph's worth of it). Still refused by name: the contextual features (`calt`, `clig`), which reach through Contextual/Chaining Contextual lookup machinery out of proportion to apply; the opt-in features (`smcp`, `dlig` and friends), which no real shaper turns on until a caller asks; and lookups with a nonzero `lookupFlag`, whose behaviour depends on GDEF glyph classes this package does not read. The legacy `kern` table is not read either (neither Carlito nor Caladea ships one). A kerned run is shown with `TJ`, and the sign of a `TJ` number is the opposite of the adjustment it expresses (ISO 32000-1 9.4.3: a positive number moves the next glyph closer). A run with no kerning pairs stays as one unsplit hex string with `Tj`.
|
|
236
277
|
- **Kerning applies to whole shown strings, so a wrap decision does not see a pair straddling the boundary between two separately-measured words.** The width a line reports is the width the page draws; making the wrap decision itself exact would mean widening the `TextMeasurer` port for a sub-point difference that only ever errs towards breaking a line early.
|
|
237
|
-
- **`font-substitutes.ts` maps both `Calibri` and `Calibri Light` onto the same ordinary-weight Carlito face** — Carlito ships only one weight per style axis, so `Calibri Light` substitutes to standard Carlito rather than a genuinely lighter face. An honest, documented approximation: width metrics match, visibly thinner strokes do not.
|
|
278
|
+
- **`font-substitutes.ts` maps both `Calibri` and `Calibri Light` onto the same ordinary-weight Carlito face** — Carlito ships only one weight per style axis (no Light face exists upstream to vendor), so `Calibri Light` substitutes to standard Carlito rather than a genuinely lighter face. An honest, documented approximation: width metrics match, visibly thinner strokes do not — and a REPORTED one, never silent: the registry's `onSubstitution` channel fires for it like every vendored substitution, naming `Calibri Light` → `carlito` so a caller can act on the weight mismatch (pinned by test in write-embedded-font.test.ts).
|
|
238
279
|
- **`cff-probe.ts`'s CID-keyed CFF guard exists for a source-embedded-font phase this package hasn't built yet** — it is not wired into any write path today. Every face currently embedded is `glyf`-flavoured TrueType, and `sfnt-subset.ts` already refuses anything else before this guard would run.
|
|
239
280
|
|
|
281
|
+
## Text extraction and font encodings
|
|
282
|
+
|
|
283
|
+
A character code in a content stream means nothing on its own: what character it draws is decided by the font it is shown in. `readPdf` resolves that through every source the format offers, in the order ISO 32000-1 clause 9.6.6 puts them, and reports an honestly unmapped code where they are all silent rather than guessing.
|
|
284
|
+
|
|
285
|
+
For a simple font, `/ToUnicode` wins wherever it covers a code, then `/Encoding`'s own `/Differences` array, then the base encoding. What "the base encoding" is depends on the font: a symbolic font (one whose `/FontDescriptor` sets the Symbolic flag, or whose `/BaseFont` is Symbol or ZapfDingbats) is encoded by its own font program, so the **embedded program's built-in encoding is read first** and the two fixed standard-14 symbol tables next; an ordinary text font takes an explicitly named `/WinAnsiEncoding`, `/MacRomanEncoding` or `/StandardEncoding` first, falls through to its own program when it names none, and only then to WinAnsi. For a composite font, `/ToUnicode` is the authority, and an Identity-H font whose CIDs are its program's own glyph IDs falls back to the program the same way.
|
|
286
|
+
|
|
287
|
+
Reading a program's built-in encoding is what stops a symbol-encoded subset — a handful of glyphs embedded to draw Ω, µ, ± or ≤ at whatever codes the producing tool picked — from decoding as whatever character the assumed default encoding happens to put at that code. That failure is silent by construction: `W` for an ohm sign is not missing or garbled text, it is a plausible different character that nothing downstream can detect. `builtin-encoding.ts` reads the encoding out of all three program shapes a PDF can embed — a TrueType `cmap`'s (3, 0) or (1, 0) subtable with the glyph identified through `post` names or the font's own Unicode subtable read backwards, a CFF Top DICT's Encoding operator with glyphs named through its charset, and a Type 1 program's cleartext `/Encoding` array — resolving each through the Adobe Glyph List, including its constructed `uniXXXX`/`uXXXXXX` name forms.
|
|
288
|
+
|
|
289
|
+
**Where nothing states an answer, the answer is the replacement character plus a `text/unmapped-encoding` diagnostic, never a guess.** Two cases reach it in practice: a symbolic font with no embedded program and no `/ToUnicode`, and a subsetted font that both strips its glyph names and maps its glyphs only from private-use code points — a private-use code point identifies a glyph inside one font and says nothing about the character it draws, so it is treated as no answer rather than a wrong one.
|
|
290
|
+
|
|
240
291
|
## JBIG2 scope
|
|
241
292
|
|
|
242
293
|
`src/image/jbig2*.ts` is a hand-written ITU-T T.88 decoder covering what real scanned PDFs actually contain.
|
|
@@ -247,7 +298,7 @@ Dependency direction is strictly downward and checkable: `layout` imports only `
|
|
|
247
298
|
|
|
248
299
|
**Verification.** `src/test-support/jbig2.ts` holds real streams from three independent producers: jbig2enc (the encoder behind essentially every JBIG2-in-PDF in the wild), libtiff (MMR payloads), and a hand-written T.88 Annex E arithmetic encoder for templates jbig2enc will not emit. Every stream — hand-encoded ones included — is decoded by jbig2dec (Ghostscript's independent implementation) before being written out, and the bitmap recorded as each fixture's expected output is jbig2dec's, not this package's. The symbol-mode fixtures exist in six variants with only `REFCORNER`/`TRANSPOSED` bits rewritten, turning jbig2dec's output into a real differential test of the placement rules jbig2enc never exercises.
|
|
249
300
|
|
|
250
|
-
A differential test pins the
|
|
301
|
+
A differential test pins the _set_ of template positions and offsets but not their _order_ — a context index is only a label for a neighbourhood pattern, so any consistent permutation cancels between encoder and decoder. TPGRON is refused rather than shipped unverified: jbig2enc's refinement support is disabled upstream, so the only available stream is one this package encoded itself, which cannot pin the pseudo-context constant even in principle (brute-forcing all 1024 candidates confirmed different unrelated bands of constants pass depending on the test image).
|
|
251
302
|
|
|
252
303
|
## JPEG 2000 scope
|
|
253
304
|
|
|
@@ -265,23 +316,21 @@ A differential test pins the *set* of template positions and offsets but not the
|
|
|
265
316
|
|
|
266
317
|
**Supply a `FontRegistry` and Calibri/Cambria stop being an approximation at all** — the vendored Carlito/Caladea faces are the real metric-compatible TrueType families, resolved automatically. Aptos still has no vendored substitute. A resolved face is measured at its own real `hmtx` advances (never the width-correction table — applying both would silently draw text narrower than measured), subsetted to used glyphs, and embedded as a real `/Type0` + `/CIDFontType2` + `/FontFile2` group. Remaining limits: only TrueType (`glyf`) outlines can be embedded; a character with no glyph is drawn as `.notdef` and reported through `onMissingGlyph`; and vertical-metric policy is caller-chosen.
|
|
267
318
|
|
|
268
|
-
**The acceptance bar for embedded-font fidelity is "no page-count drift on a real corpus", not "line-identical".** An embedded run is placed at real `hmtx` advances adjusted by real `GPOS` pair kerning — measured and drawn from one shared computation. What separates this from line-identical:
|
|
319
|
+
**The acceptance bar for embedded-font fidelity is "no page-count drift on a real corpus", not "line-identical".** An embedded run is placed at real `hmtx` advances adjusted by real `GPOS` pair kerning, shaped through real `GSUB` default ligatures — measured and drawn from one shared computation. What separates this from line-identical: contextual alternates and the opt-in layout features are never applied, and kerning is within each shown string rather than across whitespace boundaries a wrap decision measures separately.
|
|
269
320
|
|
|
270
321
|
**The one exception is math-formula rendering (`WritePdfOptions.formulas`): this genuinely embeds a real, hand-parsed font.** Real box-model glyph runs through the embedded STIX Two Math font with genuine per-glyph metrics and font-wide layout constants parsed directly from the `MATH` table. Stretchy constructions are real: a `MathVariants` variant or assembly resolved, measured against actual outlines, and drawn by glyph ID.
|
|
271
322
|
|
|
272
|
-
**`readPdf(writePdf(doc))` is not guaranteed to reproduce `doc` exactly, and `writePdf(readPdf(bytes))` is not guaranteed to reproduce `bytes` exactly.** A PDF page is fundamentally positioned drawing operators, not a structured document — a shape drawn any way other than the recognised characteristic patterns, or rotated off-axis, collapses to a generic `LayoutPath`. This is a deliberate, permanent contrast with format-preserving codecs like `ooxml.js`'s `packageCodec`. `pdfCodec` shares `z.codec()`'s
|
|
323
|
+
**`readPdf(writePdf(doc))` is not guaranteed to reproduce `doc` exactly, and `writePdf(readPdf(bytes))` is not guaranteed to reproduce `bytes` exactly.** A PDF page is fundamentally positioned drawing operators, not a structured document — a shape drawn any way other than the recognised characteristic patterns, or rotated off-axis, collapses to a generic `LayoutPath`. This is a deliberate, permanent contrast with format-preserving codecs like `ooxml.js`'s `packageCodec`. `pdfCodec` shares `z.codec()`'s _mechanism_ (schema-validated both ways) but not that _guarantee_.
|
|
273
324
|
|
|
274
325
|
**Optional real-world corpus.** `test/corpus/` (gitignored) holds a `pnpm test:corpus` vitest project for manual conformance checking against real PDFs — Word/PowerPoint/Chrome/LibreOffice exports. Not part of `pnpm test` and does not gate CI; drop files in locally before a significant parser change.
|
|
275
326
|
|
|
276
327
|
## Release and publishing
|
|
277
328
|
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
Whether that release actually published a new version is detected by diffing `package.json`'s version before and after the release step. Two further jobs gate on that: one republishes the same build under the scoped `@exadev/pdf-codec` alias to GitHub Packages (authenticating with `GITHUB_TOKEN`), and one packs the release, generates an SPDX SBOM (`pnpm sbom`), and signs both an SBOM and a build-provenance attestation against that exact tarball — verifiable independently of the registry, and still present if the package is later unpublished.
|
|
329
|
+
Release, CI, and commit-message conventions are all workspace-wide, not package-local — see the [monorepo root README](../../README.md#releases) for the mechanism (topological per-package `semantic-release` via `@exadev/semantic-release-workspace`, OIDC trusted npm publishing, automatic sibling dependency-range rewriting) and its [post-release republishing and attestation](../../README.md#releases) note on the restored GitHub Packages mirrors, npm aliases, and SBOM/provenance signing.
|
|
281
330
|
|
|
282
331
|
## Contributing
|
|
283
332
|
|
|
284
|
-
|
|
333
|
+
Conventional Commits, enforced workspace-wide by commitlint through a root `commit-msg` hook. Work inside `packages/pdf-codec/`; see [CONTRIBUTING.md](../../CONTRIBUTING.md) for the shared git hooks and history conventions. The package's own scripts are turbo-wrapped:
|
|
285
334
|
|
|
286
335
|
```sh
|
|
287
336
|
pnpm build # tsdown (ESM + CJS + .d.ts)
|
|
@@ -298,18 +347,20 @@ There is a single `main` branch and no open pull request workflow established so
|
|
|
298
347
|
## References
|
|
299
348
|
|
|
300
349
|
- [documents.js](https://github.com/ExaDev/documents.js) — the package this codec was extracted from, and its principal downstream consumer: docx/pptx/odt/odp/ods/odg ⇄ PDF conversion, and MathML formula rendering (its own `src/mathml/` typesetting engine feeds a real `MathBox` into this package's `writePdf({ formulas })` with zero cast).
|
|
301
|
-
- [document-schema.js](
|
|
350
|
+
- [document-schema.js](../document-schema.js/README.md) — the sibling package that owns the canonical `ContentDocument`/`DocumentTree` pivots the wider family shares, plus the shared leaf shapes and port types this package imports (`Color`, `LayoutFont`, `LayoutMetadata`, `TextMeasurer`, the math family). The `LayoutDocument` item family itself lived there until moving into this package.
|
|
302
351
|
- [qpdf](https://qpdf.sourceforge.io/) — the independent implementation that produces this package's encrypted-PDF test fixtures. A build-time and test-time tool only, never a dependency of the package itself.
|
|
303
|
-
- The specifications `src/crypto/` implements, each cited in the module that implements it and checked against published conformance vectors: [RFC 1321](https://www.rfc-editor.org/rfc/rfc1321) (MD5), [FIPS 180-4](https://csrc.nist.gov/pubs/fips/180-4/upd1/final) (SHA-256/384/512), [FIPS 197](https://csrc.nist.gov/pubs/fips/197/final) (AES), and [NIST SP 800-38A](https://csrc.nist.gov/pubs/sp/800/38/a/final) (CBC mode). The standard security handler is ISO 32000-1 7.6, extended for revisions 5 and 6 by ISO 32000-2 7.6.4.3.
|
|
352
|
+
- The specifications `src/crypto/` implements, each cited in the module that implements it and checked against published conformance vectors: [RFC 1321](https://www.rfc-editor.org/rfc/rfc1321) (MD5), [FIPS 180-4](https://csrc.nist.gov/pubs/fips/180-4/upd1/final) (SHA-256/384/512), [FIPS 197](https://csrc.nist.gov/pubs/fips/197/final) (AES), and [NIST SP 800-38A](https://csrc.nist.gov/pubs/sp/800/38/a/final) (CBC mode). The standard security handler is ISO 32000-1 7.6, extended for revisions 5 and 6 by ISO 32000-2 7.6.4.3; the write side's own password algorithms (`src/encrypt-write.ts`) are ISO 32000-2 7.6.4.4's Algorithms 3, 8, 9, and 10.
|
|
304
353
|
- [STIX Two Math](https://github.com/stipub/stixfonts) — the embedded math font, vendored at `assets/fonts/STIXTwoMath-Regular.otf` and embedded into `dist/` as a base64 string (`src/assets/stix-two-math-font.ts`, generated by `scripts/generate-math-font-asset.mjs`). Copyright 2001-2021 The STIX Fonts Project Authors, licensed [OFL-1.1](assets/fonts/OFL.txt) — see `assets/fonts/NOTICE.md` for the exact source commit and version.
|
|
305
354
|
|
|
306
355
|
## npm aliases
|
|
307
356
|
|
|
308
|
-
This package also
|
|
357
|
+
This package also published under the following alternate npm names from the pre-monorepo pipeline:
|
|
309
358
|
|
|
310
359
|
- [pdf-codec.js](https://www.npmjs.com/package/pdf-codec.js)
|
|
311
360
|
- [pdf-parser.js](https://www.npmjs.com/package/pdf-parser.js)
|
|
312
361
|
|
|
362
|
+
**Frozen since the monorepo migration** — see the [root README's release note](../../README.md#releases): the alias republish step was dropped along with GitHub Packages mirroring and SBOM/provenance signing, and nothing today keeps either name in sync with `pdf-codec`'s own releases. Tracked in [ExaDev/documents.js#729](https://github.com/ExaDev/documents.js/issues/729).
|
|
363
|
+
|
|
313
364
|
## License
|
|
314
365
|
|
|
315
366
|
MIT
|