reamkit 1.15.0 → 1.15.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/arc-to-bezier.d.ts +23 -0
- package/dist/esm/core/arc-to-bezier.js +23 -0
- package/dist/esm/core/bidi/algorithm.d.ts +26 -0
- package/dist/esm/core/bidi/algorithm.js +21 -0
- package/dist/esm/core/bidi/char-types.d.ts +17 -0
- package/dist/esm/core/bidi/char-types.js +10 -0
- package/dist/esm/core/bidi/index.d.ts +42 -0
- package/dist/esm/core/bidi/index.js +27 -0
- package/dist/esm/core/bidi/segments.d.ts +22 -0
- package/dist/esm/core/bidi/segments.js +14 -0
- package/dist/esm/core/bytes.d.ts +9 -0
- package/dist/esm/core/bytes.js +9 -0
- package/dist/esm/core/converter/facade.d.ts +38 -1
- package/dist/esm/core/converter/facade.js +25 -0
- package/dist/esm/core/converter/project.d.ts +12 -0
- package/dist/esm/core/converter/project.js +11 -0
- package/dist/esm/core/converter/ream.d.ts +107 -0
- package/dist/esm/core/converter/ream.js +76 -0
- package/dist/esm/core/crypto/asn1.d.ts +77 -0
- package/dist/esm/core/crypto/asn1.js +65 -0
- package/dist/esm/core/crypto/cms.d.ts +25 -0
- package/dist/esm/core/crypto/cms.js +8 -0
- package/dist/esm/core/document-model/index.d.ts +7 -0
- package/dist/esm/core/document-model/types.d.ts +328 -0
- package/dist/esm/core/drawingml/chart-geometry.d.ts +125 -0
- package/dist/esm/core/drawingml/chart-geometry.js +96 -0
- package/dist/esm/core/drawingml/chart-parser.d.ts +34 -0
- package/dist/esm/core/drawingml/chart-parser.js +34 -0
- package/dist/esm/core/drawingml/chart-serializer.d.ts +12 -0
- package/dist/esm/core/drawingml/chart-serializer.js +12 -0
- package/dist/esm/core/drawingml/colors.d.ts +53 -0
- package/dist/esm/core/drawingml/colors.js +41 -0
- package/dist/esm/core/drawingml/preset-geometry.d.ts +30 -0
- package/dist/esm/core/drawingml/preset-geometry.js +30 -0
- package/dist/esm/core/drawingml/shape-render.d.ts +54 -0
- package/dist/esm/core/drawingml/shape-render.js +54 -0
- package/dist/esm/core/drawingml/sparkline-geometry.d.ts +15 -0
- package/dist/esm/core/drawingml/sparkline-geometry.js +13 -0
- package/dist/esm/core/drawingml/theme-parser.d.ts +11 -0
- package/dist/esm/core/drawingml/theme-parser.js +11 -0
- package/dist/esm/core/font/arabic-joining.d.ts +21 -0
- package/dist/esm/core/font/arabic-joining.js +16 -0
- package/dist/esm/core/font/binary-reader.d.ts +17 -0
- package/dist/esm/core/font/binary-reader.js +17 -0
- package/dist/esm/core/font/font-registry.d.ts +37 -0
- package/dist/esm/core/font/font-registry.js +31 -0
- package/dist/esm/core/font/measure.d.ts +20 -0
- package/dist/esm/core/font/measure.js +11 -0
- package/dist/esm/core/font/opentype-layout.d.ts +49 -0
- package/dist/esm/core/font/opentype-layout.js +37 -0
- package/dist/esm/core/font/ttf-parser.d.ts +32 -0
- package/dist/esm/core/font/ttf-parser.js +7 -0
- package/dist/esm/core/font/ttf-subset.d.ts +21 -0
- package/dist/esm/core/font/ttf-subset.js +21 -0
- package/dist/esm/core/fonts/provider.d.ts +8 -1
- package/dist/esm/core/fonts/provider.js +1 -0
- package/dist/esm/core/fonts/remote-fonts.d.ts +22 -0
- package/dist/esm/core/fonts/remote-fonts.js +16 -0
- package/dist/esm/core/hyphenation/index.d.ts +25 -0
- package/dist/esm/core/hyphenation/index.js +17 -0
- package/dist/esm/core/hyphenation/liang.d.ts +27 -0
- package/dist/esm/core/hyphenation/liang.js +14 -0
- package/dist/esm/core/images.d.ts +32 -0
- package/dist/esm/core/images.js +12 -0
- package/dist/esm/core/ir/adapters.d.ts +23 -0
- package/dist/esm/core/ir/features.d.ts +8 -0
- package/dist/esm/core/ir/features.js +1 -0
- package/dist/esm/core/ir/flow.d.ts +20 -0
- package/dist/esm/core/ir/index.d.ts +7 -0
- package/dist/esm/core/ir/loss.d.ts +12 -0
- package/dist/esm/core/ir/loss.js +8 -0
- package/dist/esm/core/ir/resources.d.ts +12 -0
- package/dist/esm/core/ir/resources.js +11 -0
- package/dist/esm/core/ir/sheet.d.ts +81 -0
- package/dist/esm/core/ir/units.d.ts +9 -0
- package/dist/esm/core/ir/units.js +2 -0
- package/dist/esm/core/line-breaker/cjk.d.ts +26 -0
- package/dist/esm/core/line-breaker/cjk.js +60 -0
- package/dist/esm/core/line-breaker/greedy.d.ts +14 -0
- package/dist/esm/core/line-breaker/greedy.js +14 -0
- package/dist/esm/core/line-breaker/index.d.ts +1 -0
- package/dist/esm/core/line-breaker/index.js +3 -0
- package/dist/esm/core/line-breaker/knuth-plass.d.ts +29 -5
- package/dist/esm/core/line-breaker/knuth-plass.js +10 -5
- package/dist/esm/core/numbering/apply.d.ts +19 -0
- package/dist/esm/core/numbering/apply.js +19 -0
- package/dist/esm/core/numbering/state.d.ts +15 -0
- package/dist/esm/core/numbering/state.js +15 -0
- package/dist/esm/core/ole/cfb.d.ts +19 -0
- package/dist/esm/core/ole/cfb.js +50 -4
- package/dist/esm/core/opc/core-properties.d.ts +12 -0
- package/dist/esm/core/opc/core-properties.js +5 -0
- package/dist/esm/core/opc/opc-writer.d.ts +24 -4
- package/dist/esm/core/opc/opc-writer.js +12 -0
- package/dist/esm/core/opc/package.d.ts +55 -0
- package/dist/esm/core/opc/package.js +46 -0
- package/dist/esm/core/opc/relationship-types.d.ts +8 -0
- package/dist/esm/core/opc/relationship-types.js +5 -0
- package/dist/esm/core/opc/relationships.d.ts +12 -0
- package/dist/esm/core/opc/relationships.js +7 -0
- package/dist/esm/core/po-helpers.d.ts +38 -0
- package/dist/esm/core/po-helpers.js +33 -0
- package/dist/esm/core/spreadsheet-model/types.d.ts +281 -0
- package/dist/esm/core/style-cascade/resolver.d.ts +31 -0
- package/dist/esm/core/style-cascade/resolver.js +31 -0
- package/dist/esm/core/style-cascade/table.d.ts +7 -0
- package/dist/esm/core/style-cascade/table.js +7 -0
- package/dist/esm/core/style-cascade/types.d.ts +27 -0
- package/dist/esm/core/style-cascade/types.js +2 -0
- package/dist/esm/core/vector.d.ts +67 -0
- package/dist/esm/core/vector.js +23 -0
- package/dist/esm/excel/activex-parser.d.ts +31 -0
- package/dist/esm/excel/activex-parser.js +24 -0
- package/dist/esm/excel/cell-reference.d.ts +14 -0
- package/dist/esm/excel/cell-reference.js +9 -0
- package/dist/esm/excel/column-bands.d.ts +29 -0
- package/dist/esm/excel/column-bands.js +26 -0
- package/dist/esm/excel/comments-parser.d.ts +19 -0
- package/dist/esm/excel/comments-parser.js +19 -0
- package/dist/esm/excel/conditional-format.d.ts +38 -0
- package/dist/esm/excel/conditional-format.js +25 -0
- package/dist/esm/excel/defined-name-ref.d.ts +21 -0
- package/dist/esm/excel/defined-name-ref.js +20 -0
- package/dist/esm/excel/form-control-parser.d.ts +13 -0
- package/dist/esm/excel/form-control-parser.js +5 -0
- package/dist/esm/excel/formula/context.d.ts +37 -0
- package/dist/esm/excel/formula/dates.d.ts +23 -0
- package/dist/esm/excel/formula/dates.js +20 -0
- package/dist/esm/excel/formula/eval.d.ts +21 -0
- package/dist/esm/excel/formula/eval.js +11 -0
- package/dist/esm/excel/formula/functions.d.ts +12 -0
- package/dist/esm/excel/formula/functions.js +12 -0
- package/dist/esm/excel/formula/index.d.ts +21 -0
- package/dist/esm/excel/formula/index.js +16 -0
- package/dist/esm/excel/formula/lexer.d.ts +12 -0
- package/dist/esm/excel/formula/lexer.js +10 -0
- package/dist/esm/excel/formula/parser.d.ts +20 -0
- package/dist/esm/excel/formula/parser.js +53 -0
- package/dist/esm/excel/formula/value.d.ts +53 -0
- package/dist/esm/excel/formula/value.js +30 -0
- package/dist/esm/excel/header-footer.d.ts +9 -0
- package/dist/esm/excel/header-footer.js +9 -0
- package/dist/esm/excel/number-format.d.ts +34 -0
- package/dist/esm/excel/number-format.js +34 -0
- package/dist/esm/excel/pivot-table-parser.d.ts +8 -0
- package/dist/esm/excel/pivot-table-parser.js +8 -0
- package/dist/esm/excel/print-model.d.ts +56 -0
- package/dist/esm/excel/print-model.js +43 -0
- package/dist/esm/excel/shared-strings-parser.d.ts +13 -0
- package/dist/esm/excel/shared-strings-parser.js +5 -0
- package/dist/esm/excel/sheet-drawing.d.ts +34 -0
- package/dist/esm/excel/sheet-drawing.js +21 -0
- package/dist/esm/excel/sheet-shape-parser.d.ts +12 -0
- package/dist/esm/excel/sheet-shape-parser.js +12 -0
- package/dist/esm/excel/sheet-to-flow.d.ts +19 -0
- package/dist/esm/excel/sheet-to-flow.js +10 -0
- package/dist/esm/excel/slicer-parser.d.ts +23 -0
- package/dist/esm/excel/slicer-parser.js +10 -0
- package/dist/esm/excel/styles-parser.d.ts +8 -0
- package/dist/esm/excel/styles-parser.js +8 -0
- package/dist/esm/excel/table-parser.d.ts +24 -0
- package/dist/esm/excel/table-parser.js +5 -0
- package/dist/esm/excel/workbook-parser.d.ts +17 -0
- package/dist/esm/excel/workbook-parser.js +5 -0
- package/dist/esm/excel/worksheet-parser.d.ts +9 -0
- package/dist/esm/excel/worksheet-parser.js +9 -0
- package/dist/esm/excel/xls/biff-chart.d.ts +10 -0
- package/dist/esm/excel/xls/biff-chart.js +10 -0
- package/dist/esm/excel/xls/biff-reader.d.ts +27 -0
- package/dist/esm/excel/xls/biff-reader.js +44 -0
- package/dist/esm/excel/xls/biff-styles.d.ts +20 -0
- package/dist/esm/excel/xls/biff-styles.js +20 -0
- package/dist/esm/excel/xls/escher.d.ts +27 -0
- package/dist/esm/excel/xls/escher.js +18 -0
- package/dist/esm/excel/xls/xls-reader.d.ts +6 -0
- package/dist/esm/excel/xls/xls-reader.js +6 -0
- package/dist/esm/excel/xlsx-reader.d.ts +26 -0
- package/dist/esm/excel/xlsx-reader.js +26 -0
- package/dist/esm/excel/xlsx-to-pdf.d.ts +27 -0
- package/dist/esm/excel/xlsx-writer.d.ts +14 -0
- package/dist/esm/excel/xlsx-writer.js +13 -0
- package/dist/esm/html/html-writer.d.ts +18 -0
- package/dist/esm/html/html-writer.js +18 -0
- package/dist/esm/index.d.ts +23 -0
- package/dist/esm/layout/math-layout.d.ts +39 -0
- package/dist/esm/layout/math-layout.js +18 -0
- package/dist/esm/layout/page-doc.d.ts +91 -0
- package/dist/esm/layout/styled-layout.d.ts +175 -0
- package/dist/esm/layout/styled-layout.js +117 -2
- package/dist/esm/pdf/builtin-fonts.d.ts +5 -0
- package/dist/esm/pdf/cid-font.d.ts +22 -0
- package/dist/esm/pdf/cid-font.js +10 -0
- package/dist/esm/pdf/embedded-file.d.ts +16 -0
- package/dist/esm/pdf/embedded-file.js +9 -0
- package/dist/esm/pdf/encryption.d.ts +59 -0
- package/dist/esm/pdf/encryption.js +37 -0
- package/dist/esm/pdf/icc-profile.d.ts +8 -0
- package/dist/esm/pdf/icc-profile.js +8 -0
- package/dist/esm/pdf/image-xobject.d.ts +19 -0
- package/dist/esm/pdf/image-xobject.js +9 -0
- package/dist/esm/pdf/objects.d.ts +40 -0
- package/dist/esm/pdf/objects.js +36 -0
- package/dist/esm/pdf/serialize.d.ts +18 -0
- package/dist/esm/pdf/serialize.js +18 -0
- package/dist/esm/pdf/shading.d.ts +16 -0
- package/dist/esm/pdf/shading.js +16 -0
- package/dist/esm/pdf/signature.d.ts +36 -0
- package/dist/esm/pdf/signature.js +22 -0
- package/dist/esm/pdf/struct-tree.d.ts +64 -0
- package/dist/esm/pdf/struct-tree.js +60 -0
- package/dist/esm/pdf/styled-page-emitter.d.ts +33 -0
- package/dist/esm/pdf/styled-page-emitter.js +21 -0
- package/dist/esm/pdf/styled-page-renderer.d.ts +18 -0
- package/dist/esm/pdf/styled-page-renderer.js +18 -0
- package/dist/esm/pdf/text-encoding.d.ts +16 -0
- package/dist/esm/pdf/text-page-renderer.d.ts +11 -0
- package/dist/esm/pdf/vector-graphics.d.ts +12 -0
- package/dist/esm/pdf/vector-graphics.js +12 -0
- package/dist/esm/pdf/writer.d.ts +39 -0
- package/dist/esm/pdf/writer.js +27 -0
- package/dist/esm/pdf/xmp.d.ts +16 -0
- package/dist/esm/pdf/xmp.js +9 -0
- package/dist/esm/pdf-reader/ccitt.d.ts +25 -0
- package/dist/esm/pdf-reader/ccitt.js +15 -0
- package/dist/esm/pdf-reader/cmap.d.ts +12 -0
- package/dist/esm/pdf-reader/cmap.js +9 -0
- package/dist/esm/pdf-reader/content.d.ts +69 -0
- package/dist/esm/pdf-reader/content.js +16 -0
- package/dist/esm/pdf-reader/crypto.d.ts +5 -0
- package/dist/esm/pdf-reader/crypto.js +5 -0
- package/dist/esm/pdf-reader/decrypt.d.ts +17 -0
- package/dist/esm/pdf-reader/decrypt.js +12 -0
- package/dist/esm/pdf-reader/document.d.ts +53 -0
- package/dist/esm/pdf-reader/document.js +50 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +51 -2
- package/dist/esm/pdf-reader/flow-build.js +67 -2
- package/dist/esm/pdf-reader/font.d.ts +12 -0
- package/dist/esm/pdf-reader/font.js +12 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +21 -0
- package/dist/esm/pdf-reader/image-decode.js +14 -0
- package/dist/esm/pdf-reader/images.d.ts +19 -0
- package/dist/esm/pdf-reader/images.js +9 -0
- package/dist/esm/pdf-reader/layout.d.ts +15 -0
- package/dist/esm/pdf-reader/layout.js +17 -2
- package/dist/esm/pdf-reader/lexer.d.ts +43 -0
- package/dist/esm/pdf-reader/lexer.js +38 -0
- package/dist/esm/pdf-reader/parser.d.ts +15 -0
- package/dist/esm/pdf-reader/parser.js +8 -0
- package/dist/esm/pdf-reader/png-encode.d.ts +9 -0
- package/dist/esm/pdf-reader/png-encode.js +8 -0
- package/dist/esm/pdf-reader/predictor.d.ts +9 -0
- package/dist/esm/pdf-reader/predictor.js +8 -0
- package/dist/esm/pdf-reader/reader.d.ts +16 -0
- package/dist/esm/pdf-reader/reader.js +16 -0
- package/dist/esm/pdf-reader/shading.d.ts +10 -0
- package/dist/esm/pdf-reader/shading.js +10 -0
- package/dist/esm/pdf-reader/struct-tree.d.ts +24 -0
- package/dist/esm/pdf-reader/struct-tree.js +12 -0
- package/dist/esm/pdf-reader/tagged.d.ts +13 -0
- package/dist/esm/pdf-reader/tagged.js +15 -2
- package/dist/esm/pdf-reader/text.d.ts +11 -0
- package/dist/esm/pdf-reader/text.js +11 -0
- package/dist/esm/pdf-reader/vector.d.ts +19 -0
- package/dist/esm/pdf-reader/vector.js +9 -0
- package/dist/esm/pptx/placeholder-cascade.d.ts +21 -0
- package/dist/esm/pptx/placeholder-cascade.js +10 -0
- package/dist/esm/pptx/ppt/ppt-reader.d.ts +16 -0
- package/dist/esm/pptx/ppt/ppt-reader.js +16 -0
- package/dist/esm/pptx/ppt/ppt-text.d.ts +55 -0
- package/dist/esm/pptx/ppt/ppt-text.js +14 -0
- package/dist/esm/pptx/pptx-reader.d.ts +21 -0
- package/dist/esm/pptx/pptx-reader.js +37 -1
- package/dist/esm/pptx/slide-parser.d.ts +102 -0
- package/dist/esm/pptx/slide-parser.js +78 -0
- package/dist/esm/pptx/sp-helpers.d.ts +28 -0
- package/dist/esm/pptx/sp-helpers.js +22 -0
- package/dist/esm/svg/svg-writer.d.ts +17 -0
- package/dist/esm/svg/svg-writer.js +16 -0
- package/dist/esm/word/doc/doc-reader.d.ts +16 -0
- package/dist/esm/word/doc/doc-reader.js +26 -4
- package/dist/esm/word/doc/doc-text.d.ts +81 -0
- package/dist/esm/word/doc/doc-text.js +38 -0
- package/dist/esm/word/document-parser.d.ts +134 -0
- package/dist/esm/word/document-parser.js +86 -0
- package/dist/esm/word/docx-reader.d.ts +16 -0
- package/dist/esm/word/docx-reader.js +16 -0
- package/dist/esm/word/docx-to-pdf.d.ts +52 -0
- package/dist/esm/word/docx-to-pdf.js +21 -0
- package/dist/esm/word/docx-writer.d.ts +17 -0
- package/dist/esm/word/docx-writer.js +17 -0
- package/dist/esm/word/drawing-parser.d.ts +88 -0
- package/dist/esm/word/drawing-parser.js +68 -0
- package/dist/esm/word/font-table.d.ts +24 -0
- package/dist/esm/word/font-table.js +24 -0
- package/dist/esm/word/numbering-parser.d.ts +11 -0
- package/dist/esm/word/numbering-parser.js +11 -0
- package/dist/esm/word/omml-parser.d.ts +15 -0
- package/dist/esm/word/omml-parser.js +15 -0
- package/dist/esm/word/omml-serializer.d.ts +9 -0
- package/dist/esm/word/omml-serializer.js +9 -0
- package/dist/esm/word/paragraph-properties.d.ts +10 -0
- package/dist/esm/word/paragraph-properties.js +10 -0
- package/dist/esm/word/po-to-flat.d.ts +11 -0
- package/dist/esm/word/po-to-flat.js +11 -0
- package/dist/esm/word/run-properties.d.ts +10 -0
- package/dist/esm/word/run-properties.js +10 -0
- package/dist/esm/word/settings-parser.d.ts +13 -0
- package/dist/esm/word/settings-parser.js +8 -0
- package/dist/esm/word/styles-parser.d.ts +10 -0
- package/dist/esm/word/styles-parser.js +10 -0
- package/dist/esm/word/table-parser.d.ts +11 -0
- package/dist/esm/word/table-parser.js +11 -0
- package/dist/esm/word/text-extractor.d.ts +9 -0
- package/dist/esm/word/xml-helpers.d.ts +44 -0
- package/dist/esm/word/xml-helpers.js +38 -0
- package/package.json +1 -1
|
@@ -1,4 +1,16 @@
|
|
|
1
1
|
import { PdfDict } from '../pdf/objects.js';
|
|
2
2
|
import { ContentFont } from './content.js';
|
|
3
3
|
import { PdfFile } from './document.js';
|
|
4
|
+
/**
|
|
5
|
+
* Build a {@link ContentFont} (the interpreter's decode + advance hooks) from a
|
|
6
|
+
* `/Font` dictionary (E-PDF EP2). Unicode comes from the `/ToUnicode` CMap;
|
|
7
|
+
* glyph advances from a simple font's `/Widths` (§9.6.2.1) or a composite
|
|
8
|
+
* `/Type0` font's descendant `/W` array (§9.7.4.3). A code with no `/ToUnicode`
|
|
9
|
+
* entry decodes to its Latin-1 character for a simple font, or to nothing for a
|
|
10
|
+
* composite one.
|
|
11
|
+
*
|
|
12
|
+
* @param file The owning {@link PdfFile}, used to resolve indirect references.
|
|
13
|
+
* @param fontDict The `/Font` dictionary.
|
|
14
|
+
* @returns The decode/advance hooks plus the code width (1 or 2 bytes per code).
|
|
15
|
+
*/
|
|
4
16
|
export declare function buildContentFont(file: PdfFile, fontDict: PdfDict): ContentFont;
|
|
@@ -1,6 +1,18 @@
|
|
|
1
1
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
import { parseToUnicodeCMap } from "./cmap.js";
|
|
3
3
|
//#region src/pdf-reader/font.ts
|
|
4
|
+
/**
|
|
5
|
+
* Build a {@link ContentFont} (the interpreter's decode + advance hooks) from a
|
|
6
|
+
* `/Font` dictionary (E-PDF EP2). Unicode comes from the `/ToUnicode` CMap;
|
|
7
|
+
* glyph advances from a simple font's `/Widths` (§9.6.2.1) or a composite
|
|
8
|
+
* `/Type0` font's descendant `/W` array (§9.7.4.3). A code with no `/ToUnicode`
|
|
9
|
+
* entry decodes to its Latin-1 character for a simple font, or to nothing for a
|
|
10
|
+
* composite one.
|
|
11
|
+
*
|
|
12
|
+
* @param file The owning {@link PdfFile}, used to resolve indirect references.
|
|
13
|
+
* @param fontDict The `/Font` dictionary.
|
|
14
|
+
* @returns The decode/advance hooks plus the code width (1 or 2 bytes per code).
|
|
15
|
+
*/
|
|
4
16
|
function buildContentFont(file, fontDict) {
|
|
5
17
|
const isType0 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type0";
|
|
6
18
|
let toUnicode = /* @__PURE__ */ new Map();
|
|
@@ -1,15 +1,36 @@
|
|
|
1
1
|
import { PdfFile } from './document.js';
|
|
2
2
|
import { PdfStream } from '../pdf/objects.js';
|
|
3
|
+
/**
|
|
4
|
+
* The result of {@link decodePdfImage}: either a decoded standalone raster file
|
|
5
|
+
* (with pixel dimensions and an optional `degraded` note for a partial loss,
|
|
6
|
+
* e.g. a JPEG's alpha dropped) or a typed failure carrying the loss severity and
|
|
7
|
+
* a human-readable reason.
|
|
8
|
+
*/
|
|
3
9
|
export type DecodedImage = {
|
|
4
10
|
readonly ok: true;
|
|
5
11
|
readonly bytes: Uint8Array;
|
|
6
12
|
readonly format: 'png' | 'jpeg' | 'jpeg2000';
|
|
7
13
|
readonly widthPx: number;
|
|
8
14
|
readonly heightPx: number;
|
|
15
|
+
/** A partial loss note (e.g. a JPEG's alpha dropped). */
|
|
9
16
|
readonly degraded?: string;
|
|
10
17
|
} | {
|
|
11
18
|
readonly ok: false;
|
|
12
19
|
readonly severity: 'dropped' | 'degraded';
|
|
13
20
|
readonly detail: string;
|
|
14
21
|
};
|
|
22
|
+
/**
|
|
23
|
+
* Decode a PDF image XObject (ISO 32000-1 §8.9) into a standalone raster file
|
|
24
|
+
* the FlowDoc resource store can hold. JPEG (`/DCTDecode`) and JPEG 2000
|
|
25
|
+
* (`/JPXDecode`) pass through verbatim — only the filters layered before them
|
|
26
|
+
* are stripped; everything else is decoded to raw samples and re-wrapped as PNG.
|
|
27
|
+
* Supports DeviceGray/RGB/CMYK, CalGray/CalRGB, ICCBased (by `/N`) and Indexed
|
|
28
|
+
* colour spaces; Flate, LZW (EP12), RunLength, ASCII85, ASCIIHex and CCITT
|
|
29
|
+
* Group 4 / Group 3 1-D fax (EP15) filters; PNG/TIFF predictors; bit depths
|
|
30
|
+
* 1/2/4/8/16; and an `/SMask` folded in as the PNG alpha channel. Unsupported
|
|
31
|
+
* inputs (stencil `/ImageMask`, Separation/DeviceN/Lab, JBIG2, CCITT Group 3
|
|
32
|
+
* 2-D) return a typed failure so the caller records a loss.
|
|
33
|
+
*
|
|
34
|
+
* @returns The decoded image, or `{ ok: false }` with the loss severity and reason.
|
|
35
|
+
*/
|
|
15
36
|
export declare function decodePdfImage(file: PdfFile, stream: PdfStream): DecodedImage;
|
|
@@ -5,6 +5,20 @@ import { encodePng } from "./png-encode.js";
|
|
|
5
5
|
import { unzlibSync } from "fflate";
|
|
6
6
|
//#region src/pdf-reader/image-decode.ts
|
|
7
7
|
var MAX_PIXELS = 4e7;
|
|
8
|
+
/**
|
|
9
|
+
* Decode a PDF image XObject (ISO 32000-1 §8.9) into a standalone raster file
|
|
10
|
+
* the FlowDoc resource store can hold. JPEG (`/DCTDecode`) and JPEG 2000
|
|
11
|
+
* (`/JPXDecode`) pass through verbatim — only the filters layered before them
|
|
12
|
+
* are stripped; everything else is decoded to raw samples and re-wrapped as PNG.
|
|
13
|
+
* Supports DeviceGray/RGB/CMYK, CalGray/CalRGB, ICCBased (by `/N`) and Indexed
|
|
14
|
+
* colour spaces; Flate, LZW (EP12), RunLength, ASCII85, ASCIIHex and CCITT
|
|
15
|
+
* Group 4 / Group 3 1-D fax (EP15) filters; PNG/TIFF predictors; bit depths
|
|
16
|
+
* 1/2/4/8/16; and an `/SMask` folded in as the PNG alpha channel. Unsupported
|
|
17
|
+
* inputs (stencil `/ImageMask`, Separation/DeviceN/Lab, JBIG2, CCITT Group 3
|
|
18
|
+
* 2-D) return a typed failure so the caller records a loss.
|
|
19
|
+
*
|
|
20
|
+
* @returns The decoded image, or `{ ok: false }` with the loss severity and reason.
|
|
21
|
+
*/
|
|
8
22
|
function decodePdfImage(file, stream) {
|
|
9
23
|
const d = stream.dict;
|
|
10
24
|
const width = intOf(file.get(d, "Width")) || intOf(file.get(d, "W"));
|
|
@@ -1,16 +1,35 @@
|
|
|
1
1
|
import { Loss } from '../core/ir/index.js';
|
|
2
2
|
import { PdfFile, PdfPage } from './document.js';
|
|
3
|
+
/**
|
|
4
|
+
* One raster image lifted off a page (E-PDF EP6): the standalone image file
|
|
5
|
+
* (`png`/`jpeg`/`jpeg2000`), its page-space rectangle (computed from the CTM
|
|
6
|
+
* that maps the unit square) and the enclosing marked-content id so the tagged
|
|
7
|
+
* path can attach it to a `/Figure`.
|
|
8
|
+
*/
|
|
3
9
|
export interface PdfImage {
|
|
4
10
|
readonly bytes: Uint8Array;
|
|
5
11
|
readonly format: 'png' | 'jpeg' | 'jpeg2000';
|
|
12
|
+
/** Display size in page points (from the CTM). */
|
|
6
13
|
readonly widthPt: number;
|
|
7
14
|
readonly heightPt: number;
|
|
15
|
+
/** Page-space lower-left corner (points, y-up). */
|
|
8
16
|
readonly x: number;
|
|
9
17
|
readonly y: number;
|
|
18
|
+
/** Enclosing marked-content id, if the placement was inside a `/Figure`. */
|
|
10
19
|
readonly mcid?: number;
|
|
11
20
|
}
|
|
21
|
+
/** The images lifted off one page plus any losses for images that could not be reconstructed. */
|
|
12
22
|
export interface PageImages {
|
|
13
23
|
readonly images: Array<PdfImage>;
|
|
14
24
|
readonly losses: Array<Loss>;
|
|
15
25
|
}
|
|
26
|
+
/**
|
|
27
|
+
* Lift the raster images off a page (E-PDF EP6). Runs the content interpreter
|
|
28
|
+
* (EP2) for its `Do` placements, resolves each name against the page's
|
|
29
|
+
* `/Resources` `/XObject`, and either decodes an `/Image` (via `decodePdfImage`)
|
|
30
|
+
* or recurses into a `/Form` XObject — composing the form's `/Matrix` onto the
|
|
31
|
+
* placement CTM, depth-guarded against cyclic forms. Each surviving image
|
|
32
|
+
* carries its page-space rectangle and enclosing structure id. Unsupported
|
|
33
|
+
* images become {@link Loss} entries rather than broken pictures.
|
|
34
|
+
*/
|
|
16
35
|
export declare function collectPageImages(file: PdfFile, page: PdfPage): PageImages;
|
|
@@ -6,6 +6,15 @@ import { decodePdfImage } from "./image-decode.js";
|
|
|
6
6
|
var NO_FONTS = /* @__PURE__ */ new Map();
|
|
7
7
|
var MAX_FORM_DEPTH = 12;
|
|
8
8
|
var MAX_IMAGES = 4096;
|
|
9
|
+
/**
|
|
10
|
+
* Lift the raster images off a page (E-PDF EP6). Runs the content interpreter
|
|
11
|
+
* (EP2) for its `Do` placements, resolves each name against the page's
|
|
12
|
+
* `/Resources` `/XObject`, and either decodes an `/Image` (via `decodePdfImage`)
|
|
13
|
+
* or recurses into a `/Form` XObject — composing the form's `/Matrix` onto the
|
|
14
|
+
* placement CTM, depth-guarded against cyclic forms. Each surviving image
|
|
15
|
+
* carries its page-space rectangle and enclosing structure id. Unsupported
|
|
16
|
+
* images become {@link Loss} entries rather than broken pictures.
|
|
17
|
+
*/
|
|
9
18
|
function collectPageImages(file, page) {
|
|
10
19
|
const images = [];
|
|
11
20
|
const lossByDetail = /* @__PURE__ */ new Map();
|
|
@@ -1,3 +1,18 @@
|
|
|
1
1
|
import { PdfFile } from './document.js';
|
|
2
2
|
import { Reconstruction } from './flow-build.js';
|
|
3
|
+
/**
|
|
4
|
+
* Heuristically reconstruct an untagged PDF into a {@link Reconstruction}
|
|
5
|
+
* (E-PDF EP4). With no structure tree there is only positioned content, so
|
|
6
|
+
* reading order is recovered the way a human eye does: split a clean two-column
|
|
7
|
+
* page at its central gutter (EP17), then within each column cluster runs
|
|
8
|
+
* sharing a baseline into lines, order each line left-to-right inserting spaces
|
|
9
|
+
* across gaps, group lines into paragraphs by their vertical spacing, and
|
|
10
|
+
* interleave the column's images and filled vector paths (EP10) by their top
|
|
11
|
+
* edge. Each run's `href` is carried through as a span so links survive, and a
|
|
12
|
+
* font size well above the document median is guessed as a heading. The result
|
|
13
|
+
* is inherently approximate.
|
|
14
|
+
*
|
|
15
|
+
* @param file The PDF to reconstruct.
|
|
16
|
+
* @returns The reconstructed {@link FlowDoc} plus any read-time losses.
|
|
17
|
+
*/
|
|
3
18
|
export declare function reconstructByLayout(file: PdfFile): Reconstruction;
|
|
@@ -1,9 +1,24 @@
|
|
|
1
1
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
2
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, shapeBlock } from "./flow-build.js";
|
|
2
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
|
|
3
3
|
import { collectPageImages } from "./images.js";
|
|
4
4
|
import { extractPageText } from "./text.js";
|
|
5
5
|
import { collectPageVectors } from "./vector.js";
|
|
6
6
|
//#region src/pdf-reader/layout.ts
|
|
7
|
+
/**
|
|
8
|
+
* Heuristically reconstruct an untagged PDF into a {@link Reconstruction}
|
|
9
|
+
* (E-PDF EP4). With no structure tree there is only positioned content, so
|
|
10
|
+
* reading order is recovered the way a human eye does: split a clean two-column
|
|
11
|
+
* page at its central gutter (EP17), then within each column cluster runs
|
|
12
|
+
* sharing a baseline into lines, order each line left-to-right inserting spaces
|
|
13
|
+
* across gaps, group lines into paragraphs by their vertical spacing, and
|
|
14
|
+
* interleave the column's images and filled vector paths (EP10) by their top
|
|
15
|
+
* edge. Each run's `href` is carried through as a span so links survive, and a
|
|
16
|
+
* font size well above the document median is guessed as a heading. The result
|
|
17
|
+
* is inherently approximate.
|
|
18
|
+
*
|
|
19
|
+
* @param file The PDF to reconstruct.
|
|
20
|
+
* @returns The reconstructed {@link FlowDoc} plus any read-time losses.
|
|
21
|
+
*/
|
|
7
22
|
function reconstructByLayout(file) {
|
|
8
23
|
const pages = file.pages();
|
|
9
24
|
const pageRuns = pages.map((page) => extractPageText(file, page));
|
|
@@ -45,7 +60,7 @@ function reconstructByLayout(file) {
|
|
|
45
60
|
for (const block of blocks) body.push(block.el);
|
|
46
61
|
});
|
|
47
62
|
return {
|
|
48
|
-
doc: buildFlowDoc(body, resources),
|
|
63
|
+
doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages)),
|
|
49
64
|
losses: dedupeLosses(losses)
|
|
50
65
|
};
|
|
51
66
|
}
|
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One COS lexical token (ISO 32000-1 §7.2/§7.3): a number, name, string (literal
|
|
3
|
+
* or hex), an array/dictionary delimiter, a bare keyword (`obj` / `R` / `stream`
|
|
4
|
+
* / `true` / …), or end-of-input.
|
|
5
|
+
*/
|
|
1
6
|
export type Token = {
|
|
2
7
|
readonly kind: 'num';
|
|
3
8
|
readonly value: number;
|
|
@@ -24,20 +29,58 @@ export type Token = {
|
|
|
24
29
|
} | {
|
|
25
30
|
readonly kind: 'eof';
|
|
26
31
|
};
|
|
32
|
+
/**
|
|
33
|
+
* COS lexer (ISO 32000-1 §7.2/§7.3): scans a PDF byte buffer into a stream of
|
|
34
|
+
* {@link Token}s. The parser drives it with look-ahead to recover the object
|
|
35
|
+
* grammar. `pos` is the public cursor; callers (and the parser's rewind logic)
|
|
36
|
+
* read and assign it directly.
|
|
37
|
+
*/
|
|
27
38
|
export declare class Lexer {
|
|
28
39
|
private readonly buf;
|
|
29
40
|
pos: number;
|
|
41
|
+
/**
|
|
42
|
+
* @param buf The PDF byte buffer to tokenize.
|
|
43
|
+
* @param pos The starting byte offset (defaults to the start of the buffer).
|
|
44
|
+
*/
|
|
30
45
|
constructor(buf: Uint8Array, pos?: number);
|
|
46
|
+
/** The length of the underlying byte buffer. */
|
|
31
47
|
get length(): number;
|
|
48
|
+
/** The byte at index `i`, or −1 when out of range. */
|
|
32
49
|
byteAt(i: number): number;
|
|
50
|
+
/** §7.2.3 — skip whitespace and `%`-to-end-of-line comments. */
|
|
33
51
|
skipWhitespace(): void;
|
|
52
|
+
/** Read and consume the next {@link Token} from the current position. */
|
|
34
53
|
nextToken(): Token;
|
|
54
|
+
/** Read a numeric token (optional sign, digits and a decimal point). */
|
|
35
55
|
private readNumber;
|
|
56
|
+
/** Read a `/Name` token, decoding `#XX` hex escapes (§7.3.5). */
|
|
36
57
|
private readName;
|
|
58
|
+
/** Read a bare keyword token: a run of regular bytes (`obj`, `R`, `true`, …). */
|
|
37
59
|
private readKeyword;
|
|
60
|
+
/** Read a `<…>` hex string (§7.3.4.3); an odd trailing digit takes a low nibble of 0. */
|
|
38
61
|
private readHexString;
|
|
62
|
+
/**
|
|
63
|
+
* Read a `(…)` literal string (§7.3.4.2): handles nested parentheses, backslash
|
|
64
|
+
* escapes and octal codes. The bytes are decoded latin1.
|
|
65
|
+
*/
|
|
39
66
|
private readLiteralString;
|
|
67
|
+
/**
|
|
68
|
+
* First index of an ASCII `needle` at or after `from` (−1 if none). Used to find
|
|
69
|
+
* `endstream` when a stream's `/Length` is missing or an unresolved reference.
|
|
70
|
+
*/
|
|
40
71
|
indexOfAscii(needle: string, from: number): number;
|
|
72
|
+
/**
|
|
73
|
+
* §7.3.8.1 — read a stream's raw bytes. `pos` must sit right after the `stream`
|
|
74
|
+
* keyword. The keyword is followed by CRLF (or a lone LF); the data then runs
|
|
75
|
+
* for `length` bytes, or — when the length is unknown — up to `endstream`.
|
|
76
|
+
* Leaves `pos` at the `endstream` keyword.
|
|
77
|
+
*/
|
|
41
78
|
readStreamBody(length: number | undefined): Uint8Array;
|
|
42
79
|
}
|
|
80
|
+
/**
|
|
81
|
+
* Bytes → a Latin-1 (ISO-8859-1) string: each byte becomes the code point of the
|
|
82
|
+
* same value, so the string round-trips back to the exact bytes. PDF text in
|
|
83
|
+
* strings is decoded to Unicode later (via the font's `/ToUnicode`); at the COS
|
|
84
|
+
* layer a string is just bytes.
|
|
85
|
+
*/
|
|
43
86
|
export declare function latin1(bytes: Uint8Array): string;
|
|
@@ -14,18 +14,31 @@ function hexVal(b) {
|
|
|
14
14
|
if (b >= 97 && b <= 102) return b - 97 + 10;
|
|
15
15
|
return -1;
|
|
16
16
|
}
|
|
17
|
+
/**
|
|
18
|
+
* COS lexer (ISO 32000-1 §7.2/§7.3): scans a PDF byte buffer into a stream of
|
|
19
|
+
* {@link Token}s. The parser drives it with look-ahead to recover the object
|
|
20
|
+
* grammar. `pos` is the public cursor; callers (and the parser's rewind logic)
|
|
21
|
+
* read and assign it directly.
|
|
22
|
+
*/
|
|
17
23
|
var Lexer = class {
|
|
18
24
|
pos;
|
|
25
|
+
/**
|
|
26
|
+
* @param buf The PDF byte buffer to tokenize.
|
|
27
|
+
* @param pos The starting byte offset (defaults to the start of the buffer).
|
|
28
|
+
*/
|
|
19
29
|
constructor(buf, pos = 0) {
|
|
20
30
|
this.buf = buf;
|
|
21
31
|
this.pos = pos;
|
|
22
32
|
}
|
|
33
|
+
/** The length of the underlying byte buffer. */
|
|
23
34
|
get length() {
|
|
24
35
|
return this.buf.length;
|
|
25
36
|
}
|
|
37
|
+
/** The byte at index `i`, or −1 when out of range. */
|
|
26
38
|
byteAt(i) {
|
|
27
39
|
return i >= 0 && i < this.buf.length ? this.buf[i] : -1;
|
|
28
40
|
}
|
|
41
|
+
/** §7.2.3 — skip whitespace and `%`-to-end-of-line comments. */
|
|
29
42
|
skipWhitespace() {
|
|
30
43
|
const buf = this.buf;
|
|
31
44
|
while (this.pos < buf.length) {
|
|
@@ -37,6 +50,7 @@ var Lexer = class {
|
|
|
37
50
|
} else break;
|
|
38
51
|
}
|
|
39
52
|
}
|
|
53
|
+
/** Read and consume the next {@link Token} from the current position. */
|
|
40
54
|
nextToken() {
|
|
41
55
|
this.skipWhitespace();
|
|
42
56
|
const buf = this.buf;
|
|
@@ -77,6 +91,7 @@ var Lexer = class {
|
|
|
77
91
|
this.pos++;
|
|
78
92
|
return this.nextToken();
|
|
79
93
|
}
|
|
94
|
+
/** Read a numeric token (optional sign, digits and a decimal point). */
|
|
80
95
|
readNumber() {
|
|
81
96
|
const buf = this.buf;
|
|
82
97
|
const start = this.pos;
|
|
@@ -93,6 +108,7 @@ var Lexer = class {
|
|
|
93
108
|
value: Number.isFinite(value) ? value : 0
|
|
94
109
|
};
|
|
95
110
|
}
|
|
111
|
+
/** Read a `/Name` token, decoding `#XX` hex escapes (§7.3.5). */
|
|
96
112
|
readName() {
|
|
97
113
|
const buf = this.buf;
|
|
98
114
|
this.pos++;
|
|
@@ -117,6 +133,7 @@ var Lexer = class {
|
|
|
117
133
|
value: latin1(Uint8Array.from(out))
|
|
118
134
|
};
|
|
119
135
|
}
|
|
136
|
+
/** Read a bare keyword token: a run of regular bytes (`obj`, `R`, `true`, …). */
|
|
120
137
|
readKeyword() {
|
|
121
138
|
const buf = this.buf;
|
|
122
139
|
const start = this.pos;
|
|
@@ -126,6 +143,7 @@ var Lexer = class {
|
|
|
126
143
|
value: latin1(buf.subarray(start, this.pos))
|
|
127
144
|
};
|
|
128
145
|
}
|
|
146
|
+
/** Read a `<…>` hex string (§7.3.4.3); an odd trailing digit takes a low nibble of 0. */
|
|
129
147
|
readHexString() {
|
|
130
148
|
const buf = this.buf;
|
|
131
149
|
this.pos++;
|
|
@@ -149,6 +167,10 @@ var Lexer = class {
|
|
|
149
167
|
bytes: Uint8Array.from(out)
|
|
150
168
|
};
|
|
151
169
|
}
|
|
170
|
+
/**
|
|
171
|
+
* Read a `(…)` literal string (§7.3.4.2): handles nested parentheses, backslash
|
|
172
|
+
* escapes and octal codes. The bytes are decoded latin1.
|
|
173
|
+
*/
|
|
152
174
|
readLiteralString() {
|
|
153
175
|
const buf = this.buf;
|
|
154
176
|
this.pos++;
|
|
@@ -212,6 +234,10 @@ var Lexer = class {
|
|
|
212
234
|
value: latin1(Uint8Array.from(out))
|
|
213
235
|
};
|
|
214
236
|
}
|
|
237
|
+
/**
|
|
238
|
+
* First index of an ASCII `needle` at or after `from` (−1 if none). Used to find
|
|
239
|
+
* `endstream` when a stream's `/Length` is missing or an unresolved reference.
|
|
240
|
+
*/
|
|
215
241
|
indexOfAscii(needle, from) {
|
|
216
242
|
const buf = this.buf;
|
|
217
243
|
const n = needle.length;
|
|
@@ -221,6 +247,12 @@ var Lexer = class {
|
|
|
221
247
|
}
|
|
222
248
|
return -1;
|
|
223
249
|
}
|
|
250
|
+
/**
|
|
251
|
+
* §7.3.8.1 — read a stream's raw bytes. `pos` must sit right after the `stream`
|
|
252
|
+
* keyword. The keyword is followed by CRLF (or a lone LF); the data then runs
|
|
253
|
+
* for `length` bytes, or — when the length is unknown — up to `endstream`.
|
|
254
|
+
* Leaves `pos` at the `endstream` keyword.
|
|
255
|
+
*/
|
|
224
256
|
readStreamBody(length) {
|
|
225
257
|
const buf = this.buf;
|
|
226
258
|
if (buf[this.pos] === 13 && buf[this.pos + 1] === 10) this.pos += 2;
|
|
@@ -241,6 +273,12 @@ var Lexer = class {
|
|
|
241
273
|
return buf.subarray(start, dataEnd);
|
|
242
274
|
}
|
|
243
275
|
};
|
|
276
|
+
/**
|
|
277
|
+
* Bytes → a Latin-1 (ISO-8859-1) string: each byte becomes the code point of the
|
|
278
|
+
* same value, so the string round-trips back to the exact bytes. PDF text in
|
|
279
|
+
* strings is decoded to Unicode later (via the font's `/ToUnicode`); at the COS
|
|
280
|
+
* layer a string is just bytes.
|
|
281
|
+
*/
|
|
244
282
|
function latin1(bytes) {
|
|
245
283
|
let s = "";
|
|
246
284
|
for (const b of bytes) s += String.fromCharCode(b);
|
|
@@ -1,10 +1,25 @@
|
|
|
1
1
|
import { Lexer } from './lexer.js';
|
|
2
2
|
import { PdfValue, PdfRef } from '../pdf/objects.js';
|
|
3
|
+
/**
|
|
4
|
+
* Resolves an indirect-reference `/Length` to its numeric value. When a stream's
|
|
5
|
+
* `/Length` is an indirect reference it cannot be resolved while parsing the
|
|
6
|
+
* object in isolation; the document layer passes a resolver. Without one the
|
|
7
|
+
* parser falls back to scanning for `endstream`.
|
|
8
|
+
*/
|
|
3
9
|
export type LengthResolver = (ref: PdfRef) => number | undefined;
|
|
10
|
+
/** A parsed `N G obj … endobj` definition: its id, generation and contained value. */
|
|
4
11
|
export interface IndirectObject {
|
|
5
12
|
readonly id: number;
|
|
6
13
|
readonly generation: number;
|
|
7
14
|
readonly value: PdfValue;
|
|
8
15
|
}
|
|
16
|
+
/** Parse one object value at the lexer's current position. */
|
|
9
17
|
export declare function parseObject(lexer: Lexer, resolveLength?: LengthResolver): PdfValue;
|
|
18
|
+
/**
|
|
19
|
+
* Parse an `N G obj … endobj` definition (the lexer must sit at the leading
|
|
20
|
+
* integer).
|
|
21
|
+
*
|
|
22
|
+
* @returns The {@link IndirectObject} (id, generation and contained value), or
|
|
23
|
+
* `undefined` when the `N G obj` header does not parse.
|
|
24
|
+
*/
|
|
10
25
|
export declare function parseIndirectObject(lexer: Lexer, resolveLength?: LengthResolver): IndirectObject | undefined;
|
|
@@ -1,8 +1,16 @@
|
|
|
1
1
|
import { PDF_NULL, PdfHexString, PdfName, PdfRef, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
//#region src/pdf-reader/parser.ts
|
|
3
|
+
/** Parse one object value at the lexer's current position. */
|
|
3
4
|
function parseObject(lexer, resolveLength) {
|
|
4
5
|
return parseValue(lexer, lexer.nextToken(), resolveLength);
|
|
5
6
|
}
|
|
7
|
+
/**
|
|
8
|
+
* Parse an `N G obj … endobj` definition (the lexer must sit at the leading
|
|
9
|
+
* integer).
|
|
10
|
+
*
|
|
11
|
+
* @returns The {@link IndirectObject} (id, generation and contained value), or
|
|
12
|
+
* `undefined` when the `N G obj` header does not parse.
|
|
13
|
+
*/
|
|
6
14
|
function parseIndirectObject(lexer, resolveLength) {
|
|
7
15
|
const idTok = lexer.nextToken();
|
|
8
16
|
if (idTok.kind !== "num") return void 0;
|
|
@@ -1,2 +1,11 @@
|
|
|
1
|
+
/** A PNG colour type: `gray` = 1ch, `rgb` = 3ch, `gray-alpha` = 2ch, `rgba` = 4ch. */
|
|
1
2
|
export type PngColor = 'gray' | 'rgb' | 'gray-alpha' | 'rgba';
|
|
3
|
+
/**
|
|
4
|
+
* Wrap raw 8-bit samples into a minimal PNG file (RFC 2083): a single `IDAT`
|
|
5
|
+
* with filter-none scanlines, no interlacing, colour types 0/2/4/6. Produces a
|
|
6
|
+
* format every writer already embeds — `detectImageFormat` recognises this
|
|
7
|
+
* output (the HTML data-URI and docx media paths both rely on it).
|
|
8
|
+
*
|
|
9
|
+
* @param samples Row-major, 8-bit, `CHANNELS[color]` interleaved values per pixel.
|
|
10
|
+
*/
|
|
2
11
|
export declare function encodePng(width: number, height: number, color: PngColor, samples: Uint8Array): Uint8Array;
|
|
@@ -22,6 +22,14 @@ var SIGNATURE = Uint8Array.from([
|
|
|
22
22
|
26,
|
|
23
23
|
10
|
|
24
24
|
]);
|
|
25
|
+
/**
|
|
26
|
+
* Wrap raw 8-bit samples into a minimal PNG file (RFC 2083): a single `IDAT`
|
|
27
|
+
* with filter-none scanlines, no interlacing, colour types 0/2/4/6. Produces a
|
|
28
|
+
* format every writer already embeds — `detectImageFormat` recognises this
|
|
29
|
+
* output (the HTML data-URI and docx media paths both rely on it).
|
|
30
|
+
*
|
|
31
|
+
* @param samples Row-major, 8-bit, `CHANNELS[color]` interleaved values per pixel.
|
|
32
|
+
*/
|
|
25
33
|
function encodePng(width, height, color, samples) {
|
|
26
34
|
const stride = width * CHANNELS[color];
|
|
27
35
|
const raw = new Uint8Array(height * (stride + 1));
|
|
@@ -1,7 +1,16 @@
|
|
|
1
|
+
/** The `/DecodeParms` that parameterize a `/Predictor` (ISO 32000-1 §7.4.4.4). */
|
|
1
2
|
export interface PredictorParams {
|
|
2
3
|
readonly predictor: number;
|
|
3
4
|
readonly colors: number;
|
|
4
5
|
readonly bitsPerComponent: number;
|
|
5
6
|
readonly columns: number;
|
|
6
7
|
}
|
|
8
|
+
/**
|
|
9
|
+
* Reverse a stream's `/Predictor` (ISO 32000-1 §7.4.4.4) after FlateDecode. Undoes
|
|
10
|
+
* a PNG predictor (`/Predictor` ≥ 10, each row prefixed with a filter-type byte:
|
|
11
|
+
* None/Sub/Up/Average/Paeth) or TIFF horizontal differencing (`/Predictor` 2,
|
|
12
|
+
* 8-bit components only). Shared by image XObjects (image-decode.ts) and
|
|
13
|
+
* cross-reference / object streams (document.ts). A `/Predictor` < 2, or
|
|
14
|
+
* unsupported parameters, returns the data unchanged.
|
|
15
|
+
*/
|
|
7
16
|
export declare function reversePredictor(data: Uint8Array, p: PredictorParams): Uint8Array;
|
|
@@ -1,4 +1,12 @@
|
|
|
1
1
|
//#region src/pdf-reader/predictor.ts
|
|
2
|
+
/**
|
|
3
|
+
* Reverse a stream's `/Predictor` (ISO 32000-1 §7.4.4.4) after FlateDecode. Undoes
|
|
4
|
+
* a PNG predictor (`/Predictor` ≥ 10, each row prefixed with a filter-type byte:
|
|
5
|
+
* None/Sub/Up/Average/Paeth) or TIFF horizontal differencing (`/Predictor` 2,
|
|
6
|
+
* 8-bit components only). Shared by image XObjects (image-decode.ts) and
|
|
7
|
+
* cross-reference / object streams (document.ts). A `/Predictor` < 2, or
|
|
8
|
+
* unsupported parameters, returns the data unchanged.
|
|
9
|
+
*/
|
|
2
10
|
function reversePredictor(data, p) {
|
|
3
11
|
if (p.predictor < 2) return data;
|
|
4
12
|
const bpp = Math.max(1, Math.ceil(p.colors * p.bitsPerComponent / 8));
|
|
@@ -1,4 +1,20 @@
|
|
|
1
1
|
import { DocumentReader, ReadResult } from '../core/ir/adapters.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
|
+
/**
|
|
4
|
+
* Parse PDF bytes into a {@link FlowDoc} (E-PDF EP5): reconstruct via the tagged
|
|
5
|
+
* structure tree when present (EP3), else the layout heuristic (EP4), lifting
|
|
6
|
+
* raster images back into reading order (EP6). Records losses for encryption that
|
|
7
|
+
* could not be opened, heuristic (untagged) reconstruction, and the vector
|
|
8
|
+
* regions that are not reconstructed (clipping paths, bare `sh` shadings).
|
|
9
|
+
*
|
|
10
|
+
* @param bytes The complete PDF file bytes.
|
|
11
|
+
* @param password The user password for an encrypted source; the empty string
|
|
12
|
+
* opens permissions-only encryption.
|
|
13
|
+
* @returns The reconstructed FlowDoc and its accumulated {@link Loss} report.
|
|
14
|
+
*/
|
|
3
15
|
export declare function readPdf(bytes: Uint8Array, password?: string): ReadResult<FlowDoc>;
|
|
16
|
+
/**
|
|
17
|
+
* The `pdfReader` adapter: a {@link DocumentReader} that sniffs the `%PDF-`
|
|
18
|
+
* header and parses the bytes into a {@link FlowDoc} (E-PDF EP5).
|
|
19
|
+
*/
|
|
4
20
|
export declare const pdfReader: DocumentReader<FlowDoc>;
|
|
@@ -8,6 +8,18 @@ function sniffPdf(bytes) {
|
|
|
8
8
|
for (let i = 0; i <= limit; i++) if (bytes[i] === 37 && bytes[i + 1] === 80 && bytes[i + 2] === 68 && bytes[i + 3] === 70 && bytes[i + 4] === 45) return true;
|
|
9
9
|
return false;
|
|
10
10
|
}
|
|
11
|
+
/**
|
|
12
|
+
* Parse PDF bytes into a {@link FlowDoc} (E-PDF EP5): reconstruct via the tagged
|
|
13
|
+
* structure tree when present (EP3), else the layout heuristic (EP4), lifting
|
|
14
|
+
* raster images back into reading order (EP6). Records losses for encryption that
|
|
15
|
+
* could not be opened, heuristic (untagged) reconstruction, and the vector
|
|
16
|
+
* regions that are not reconstructed (clipping paths, bare `sh` shadings).
|
|
17
|
+
*
|
|
18
|
+
* @param bytes The complete PDF file bytes.
|
|
19
|
+
* @param password The user password for an encrypted source; the empty string
|
|
20
|
+
* opens permissions-only encryption.
|
|
21
|
+
* @returns The reconstructed FlowDoc and its accumulated {@link Loss} report.
|
|
22
|
+
*/
|
|
11
23
|
function readPdf(bytes, password = "") {
|
|
12
24
|
const file = PdfFile.parse(bytes, password);
|
|
13
25
|
const losses = [];
|
|
@@ -34,6 +46,10 @@ function readPdf(bytes, password = "") {
|
|
|
34
46
|
losses
|
|
35
47
|
};
|
|
36
48
|
}
|
|
49
|
+
/**
|
|
50
|
+
* The `pdfReader` adapter: a {@link DocumentReader} that sniffs the `%PDF-`
|
|
51
|
+
* header and parses the bytes into a {@link FlowDoc} (E-PDF EP5).
|
|
52
|
+
*/
|
|
37
53
|
var pdfReader = {
|
|
38
54
|
id: "pdf",
|
|
39
55
|
produces: "flow",
|
|
@@ -1,3 +1,13 @@
|
|
|
1
1
|
import { ShapeGradient } from '../core/vector.js';
|
|
2
2
|
import { PdfFile, PdfPage } from './document.js';
|
|
3
|
+
/**
|
|
4
|
+
* Resolve a page's `/Pattern` resources into gradient fills (E-PDF EP16c, ISO
|
|
5
|
+
* 32000-1 §8.7.4.5). Every `PatternType` 2 (shading) pattern is evaluated — its
|
|
6
|
+
* `/Shading` type 2 (axial) or 3 (radial) plus the `/Function` colour stops —
|
|
7
|
+
* and keyed by resource name; the interpreter looks the name up when a shape is
|
|
8
|
+
* filled with `/Pattern cs /Pn scn`. The bare `sh` operator (clip-bounded) is
|
|
9
|
+
* not captured.
|
|
10
|
+
*
|
|
11
|
+
* @returns A map from pattern resource name to its {@link ShapeGradient}.
|
|
12
|
+
*/
|
|
3
13
|
export declare function buildShadingMap(file: PdfFile, page: PdfPage): Map<string, ShapeGradient>;
|
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
import { PDF_NULL, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
//#region src/pdf-reader/shading.ts
|
|
3
|
+
/**
|
|
4
|
+
* Resolve a page's `/Pattern` resources into gradient fills (E-PDF EP16c, ISO
|
|
5
|
+
* 32000-1 §8.7.4.5). Every `PatternType` 2 (shading) pattern is evaluated — its
|
|
6
|
+
* `/Shading` type 2 (axial) or 3 (radial) plus the `/Function` colour stops —
|
|
7
|
+
* and keyed by resource name; the interpreter looks the name up when a shape is
|
|
8
|
+
* filled with `/Pattern cs /Pn scn`. The bare `sh` operator (clip-bounded) is
|
|
9
|
+
* not captured.
|
|
10
|
+
*
|
|
11
|
+
* @returns A map from pattern resource name to its {@link ShapeGradient}.
|
|
12
|
+
*/
|
|
3
13
|
function buildShadingMap(file, page) {
|
|
4
14
|
const out = /* @__PURE__ */ new Map();
|
|
5
15
|
if (!page.resources) return out;
|
|
@@ -1,14 +1,38 @@
|
|
|
1
1
|
import { PdfFile } from './document.js';
|
|
2
|
+
/** A marked-content reference: a page index plus an MCID on that page. */
|
|
2
3
|
export interface StructMcid {
|
|
4
|
+
/** Zero-based page index the MCID lives on. */
|
|
3
5
|
readonly page: number;
|
|
4
6
|
readonly mcid: number;
|
|
5
7
|
}
|
|
8
|
+
/**
|
|
9
|
+
* One node of the recovered logical structure tree (ISO 32000-1 §14.7): a
|
|
10
|
+
* `/StructElem`'s role, its own marked-content references (the text it owns) and
|
|
11
|
+
* its child elements.
|
|
12
|
+
*/
|
|
6
13
|
export interface StructNode {
|
|
14
|
+
/** The `/S` role name (`Document`, `P`, `H1`, `Table`, `TR`, `TD`, `L`, `LI`, …). */
|
|
7
15
|
readonly type: string;
|
|
16
|
+
/** The element's own marked content, linking it to interpreter-extracted text (EP2). */
|
|
8
17
|
readonly mcids: ReadonlyArray<StructMcid>;
|
|
9
18
|
readonly children: ReadonlyArray<StructNode>;
|
|
19
|
+
/** `/Alt` — alternate text (figures). */
|
|
10
20
|
readonly alt?: string;
|
|
21
|
+
/** `/A /Table /ColSpan` on a table cell. */
|
|
11
22
|
readonly colSpan?: number;
|
|
23
|
+
/** `/A /Table /RowSpan` on a table cell. */
|
|
12
24
|
readonly rowSpan?: number;
|
|
13
25
|
}
|
|
26
|
+
/**
|
|
27
|
+
* Read the `/StructTreeRoot` (ISO 32000-1 §14.7) into a {@link StructNode} tree,
|
|
28
|
+
* recovering reading order and roles. Walks each `/StructElem`'s `/K` children —
|
|
29
|
+
* resolving nested elements, bare-integer and `/MCR` marked-content references
|
|
30
|
+
* (carrying the owning page), and skipping `/OBJR` object references (no text) —
|
|
31
|
+
* guards against cycles and pathological size, and lifts `/Alt` plus table cell
|
|
32
|
+
* `/ColSpan`/`/RowSpan`. Multiple top-level roots are wrapped in a synthetic
|
|
33
|
+
* `Document` node.
|
|
34
|
+
*
|
|
35
|
+
* @returns The structure-tree root, or `undefined` when the catalog has no
|
|
36
|
+
* `/StructTreeRoot` (an untagged PDF).
|
|
37
|
+
*/
|
|
14
38
|
export declare function readStructTree(file: PdfFile): StructNode | undefined;
|
|
@@ -1,6 +1,18 @@
|
|
|
1
1
|
import { PDF_NULL, PdfName } from "../pdf/objects.js";
|
|
2
2
|
//#region src/pdf-reader/struct-tree.ts
|
|
3
3
|
var MAX_NODES = 2e5;
|
|
4
|
+
/**
|
|
5
|
+
* Read the `/StructTreeRoot` (ISO 32000-1 §14.7) into a {@link StructNode} tree,
|
|
6
|
+
* recovering reading order and roles. Walks each `/StructElem`'s `/K` children —
|
|
7
|
+
* resolving nested elements, bare-integer and `/MCR` marked-content references
|
|
8
|
+
* (carrying the owning page), and skipping `/OBJR` object references (no text) —
|
|
9
|
+
* guards against cycles and pathological size, and lifts `/Alt` plus table cell
|
|
10
|
+
* `/ColSpan`/`/RowSpan`. Multiple top-level roots are wrapped in a synthetic
|
|
11
|
+
* `Document` node.
|
|
12
|
+
*
|
|
13
|
+
* @returns The structure-tree root, or `undefined` when the catalog has no
|
|
14
|
+
* `/StructTreeRoot` (an untagged PDF).
|
|
15
|
+
*/
|
|
4
16
|
function readStructTree(file) {
|
|
5
17
|
const stRoot = file.get(file.catalog, "StructTreeRoot");
|
|
6
18
|
if (!(stRoot instanceof Map)) return void 0;
|
|
@@ -1,3 +1,16 @@
|
|
|
1
1
|
import { PdfFile } from './document.js';
|
|
2
2
|
import { Reconstruction } from './flow-build.js';
|
|
3
|
+
/**
|
|
4
|
+
* Reconstruct a {@link Reconstruction} from a tagged PDF's logical structure
|
|
5
|
+
* (E-PDF EP3 — the honest inverse of the tagged PDF Ream writes). Walks the
|
|
6
|
+
* structure tree ({@link readStructTree}), pulls each element's text from the
|
|
7
|
+
* per-page MCID → text map the content interpreter produced (EP2), and rebuilds
|
|
8
|
+
* headings (`H1`–`H6` → outline level), paragraphs, tables (`Table` → `TR` →
|
|
9
|
+
* `TH`/`TD`), list items (each `LI` → label + body) and figures (EP6 — each
|
|
10
|
+
* `/Figure`'s MCID resolves to a lifted image carrying its `/Alt`). Images no
|
|
11
|
+
* `/Figure` claims are appended in page + top-down order so nothing is lost.
|
|
12
|
+
*
|
|
13
|
+
* @returns The reconstructed document and image losses, or `undefined` when the
|
|
14
|
+
* PDF carries no structure tree or yields no body content.
|
|
15
|
+
*/
|
|
3
16
|
export declare function reconstructTaggedPdf(file: PdfFile): Reconstruction | undefined;
|