reamkit 1.24.0 → 1.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -11
- package/dist/esm/core/converter/facade.d.ts +3 -3
- package/dist/esm/core/converter/facade.js +12 -0
- package/dist/esm/core/converter/ream.d.ts +48 -4
- package/dist/esm/core/converter/ream.js +25 -4
- package/dist/esm/core/document-model/index.d.ts +1 -1
- package/dist/esm/core/document-model/types.d.ts +17 -0
- package/dist/esm/core/outline.d.ts +17 -0
- package/dist/esm/core/outline.js +30 -0
- package/dist/esm/core/style-cascade/resolver.js +1 -0
- package/dist/esm/core/style-cascade/types.d.ts +3 -1
- package/dist/esm/excel/sheet-to-flow.d.ts +10 -0
- package/dist/esm/excel/sheet-to-flow.js +14 -1
- package/dist/esm/html/html-writer.js +3 -2
- package/dist/esm/index.d.ts +3 -0
- package/dist/esm/index.js +2 -1
- package/dist/esm/layout/page-doc.js +1 -1
- package/dist/esm/layout/styled-layout.js +41 -15
- package/dist/esm/markdown/markdown-writer.d.ts +41 -0
- package/dist/esm/markdown/markdown-writer.js +733 -0
- package/dist/esm/pdf/styled-page-emitter.js +20 -1
- package/dist/esm/pdf-reader/annots.d.ts +24 -0
- package/dist/esm/pdf-reader/annots.js +126 -0
- package/dist/esm/pdf-reader/content.d.ts +131 -5
- package/dist/esm/pdf-reader/content.js +169 -12
- package/dist/esm/pdf-reader/display.d.ts +56 -0
- package/dist/esm/pdf-reader/display.js +162 -0
- package/dist/esm/pdf-reader/document.d.ts +36 -1
- package/dist/esm/pdf-reader/document.js +92 -25
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +31 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +94 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +61 -6
- package/dist/esm/pdf-reader/flow-build.js +128 -22
- package/dist/esm/pdf-reader/font.js +185 -4
- package/dist/esm/pdf-reader/image-decode.js +55 -4
- package/dist/esm/pdf-reader/images.d.ts +6 -0
- package/dist/esm/pdf-reader/images.js +25 -5
- package/dist/esm/pdf-reader/jpeg.d.ts +18 -0
- package/dist/esm/pdf-reader/jpeg.js +419 -0
- package/dist/esm/pdf-reader/layout.d.ts +1 -1
- package/dist/esm/pdf-reader/layout.js +221 -32
- package/dist/esm/pdf-reader/pattern-tint.d.ts +17 -0
- package/dist/esm/pdf-reader/pattern-tint.js +181 -0
- package/dist/esm/pdf-reader/reader.d.ts +9 -1
- package/dist/esm/pdf-reader/reader.js +22 -6
- package/dist/esm/pdf-reader/shading.d.ts +14 -0
- package/dist/esm/pdf-reader/shading.js +27 -1
- package/dist/esm/pdf-reader/tagged.js +156 -17
- package/dist/esm/pdf-reader/text.d.ts +13 -1
- package/dist/esm/pdf-reader/text.js +70 -3
- package/dist/esm/pdf-reader/vector.d.ts +25 -1
- package/dist/esm/pdf-reader/vector.js +168 -12
- package/dist/esm/pptx/slide-parser.js +5 -0
- package/dist/esm/word/docx-writer.js +11 -1
- package/dist/esm/word/drawing-parser.js +7 -1
- package/package.json +1 -1
|
@@ -2,6 +2,7 @@ import { pt } from "../core/ir/units.js";
|
|
|
2
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
3
|
import { EMPTY_STYLE_SHEET, resolveBodyStyles } from "../core/style-cascade/resolver.js";
|
|
4
4
|
import "../core/style-cascade/index.js";
|
|
5
|
+
import { displayOf } from "./display.js";
|
|
5
6
|
//#region src/pdf-reader/flow-build.ts
|
|
6
7
|
/**
|
|
7
8
|
* Build a paragraph {@link BodyElement} from a single plain-text string,
|
|
@@ -30,16 +31,21 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
30
31
|
const merged = [];
|
|
31
32
|
for (const s of spans) {
|
|
32
33
|
const last = merged[merged.length - 1];
|
|
33
|
-
if (last && last.href === s.href) last.text += s.text;
|
|
34
|
-
else
|
|
34
|
+
if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic) last.text += s.text;
|
|
35
|
+
else merged.push({
|
|
35
36
|
text: s.text,
|
|
36
|
-
href: s.href
|
|
37
|
+
...s.href !== void 0 ? { href: s.href } : {},
|
|
38
|
+
...s.sizePt !== void 0 ? { sizePt: s.sizePt } : {},
|
|
39
|
+
...s.colorHex !== void 0 ? { colorHex: s.colorHex } : {},
|
|
40
|
+
...s.fontName !== void 0 ? { fontName: s.fontName } : {},
|
|
41
|
+
...s.outline !== void 0 ? { outline: s.outline } : {},
|
|
42
|
+
...s.bold !== void 0 ? { bold: s.bold } : {},
|
|
43
|
+
...s.italic !== void 0 ? { italic: s.italic } : {}
|
|
37
44
|
});
|
|
38
|
-
else merged.push({ text: s.text });
|
|
39
45
|
}
|
|
40
46
|
const runs = merged.map((m) => ({
|
|
41
|
-
|
|
42
|
-
|
|
47
|
+
...m,
|
|
48
|
+
text: m.text.replace(/\s+/g, " ")
|
|
43
49
|
})).filter((m) => m.text.length > 0);
|
|
44
50
|
if (runs.length > 0) {
|
|
45
51
|
runs[0].text = runs[0].text.replace(/^ /, "");
|
|
@@ -51,7 +57,14 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
51
57
|
properties: outlineLevel !== void 0 ? { outlineLevel } : {},
|
|
52
58
|
runs: runs.filter((r) => r.text.length > 0).map((r) => ({
|
|
53
59
|
text: r.text,
|
|
54
|
-
properties: {
|
|
60
|
+
properties: {
|
|
61
|
+
...r.sizePt !== void 0 ? { fontSizePt: pt(r.sizePt) } : {},
|
|
62
|
+
...r.colorHex !== void 0 ? { colorHex: r.colorHex } : {},
|
|
63
|
+
...r.fontName !== void 0 ? { fontFamily: { ascii: r.fontName } } : {},
|
|
64
|
+
...r.outline !== void 0 ? { textOutline: r.outline } : {},
|
|
65
|
+
...r.bold ? { bold: true } : {},
|
|
66
|
+
...r.italic ? { italic: true } : {}
|
|
67
|
+
},
|
|
55
68
|
...r.href ? { href: r.href } : {}
|
|
56
69
|
}))
|
|
57
70
|
}
|
|
@@ -62,11 +75,25 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
62
75
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
63
76
|
* CTM. `alt` becomes the block's alt text when given.
|
|
64
77
|
*/
|
|
65
|
-
function imageBlock(image, resources, alt) {
|
|
78
|
+
function imageBlock(image, resources, alt, frame, zOrder) {
|
|
79
|
+
const resource = resources.put(image.bytes);
|
|
80
|
+
const float = frame !== void 0 ? {
|
|
81
|
+
wrap: "none",
|
|
82
|
+
...zOrder !== void 0 ? { zOrder } : {},
|
|
83
|
+
posH: {
|
|
84
|
+
relativeFrom: "page",
|
|
85
|
+
offsetPt: pt(image.x - frame.left)
|
|
86
|
+
},
|
|
87
|
+
posV: {
|
|
88
|
+
relativeFrom: "page",
|
|
89
|
+
offsetPt: pt(Math.max(0, frame.top - image.y - image.heightPt))
|
|
90
|
+
}
|
|
91
|
+
} : void 0;
|
|
66
92
|
return {
|
|
67
93
|
kind: "image",
|
|
68
94
|
image: {
|
|
69
|
-
|
|
95
|
+
...float ? { float } : {},
|
|
96
|
+
resource,
|
|
70
97
|
width: pt(image.widthPt),
|
|
71
98
|
height: pt(image.heightPt),
|
|
72
99
|
paragraphProperties: {},
|
|
@@ -74,6 +101,60 @@ function imageBlock(image, resources, alt) {
|
|
|
74
101
|
}
|
|
75
102
|
};
|
|
76
103
|
}
|
|
104
|
+
/**
|
|
105
|
+
* A line of text as an anchored box, standing where the page set it.
|
|
106
|
+
*
|
|
107
|
+
* The flowed reconstruction reads a document OUT of a page: paragraphs in
|
|
108
|
+
* reading order, re-flowable, free to land wherever the next medium puts them.
|
|
109
|
+
* A form is not that document. 160F-2019.pdf is a grid of ruled boxes with a
|
|
110
|
+
* label in each, and a label means nothing an inch from the box it labels — the
|
|
111
|
+
* artwork is placed absolutely, so text that flows beside it lines up with none
|
|
112
|
+
* of it.
|
|
113
|
+
*
|
|
114
|
+
* @param spans The line's runs.
|
|
115
|
+
* @param box Its page-space rectangle (y-up, as PDF measures).
|
|
116
|
+
* @param frame The page's own corner, to measure the box off.
|
|
117
|
+
* @param zOrder Its place in the page's painting order.
|
|
118
|
+
* @param rotation60k §20.1.7.6 — how far the box turns about its own centre,
|
|
119
|
+
* for a baseline the page did not set flat.
|
|
120
|
+
* @returns A shape carrying the text, anchored where the glyphs were.
|
|
121
|
+
*/
|
|
122
|
+
function positionedText(spans, box, frame, zOrder, rotation60k) {
|
|
123
|
+
const paragraph = paragraphFromRuns(spans);
|
|
124
|
+
return {
|
|
125
|
+
kind: "shape",
|
|
126
|
+
shape: {
|
|
127
|
+
float: {
|
|
128
|
+
wrap: "none",
|
|
129
|
+
zOrder,
|
|
130
|
+
posH: {
|
|
131
|
+
relativeFrom: "page",
|
|
132
|
+
offsetPt: pt(box.x - frame.left)
|
|
133
|
+
},
|
|
134
|
+
posV: {
|
|
135
|
+
relativeFrom: "page",
|
|
136
|
+
offsetPt: pt(Math.max(0, frame.top - box.y - box.height))
|
|
137
|
+
}
|
|
138
|
+
},
|
|
139
|
+
width: pt(Math.max(1, box.width)),
|
|
140
|
+
height: pt(Math.max(1, box.height)),
|
|
141
|
+
...rotation60k !== void 0 ? { transform: { rotation60k } } : {},
|
|
142
|
+
geometry: {
|
|
143
|
+
kind: "preset",
|
|
144
|
+
preset: "rect"
|
|
145
|
+
},
|
|
146
|
+
fill: { kind: "none" },
|
|
147
|
+
text: {
|
|
148
|
+
content: [paragraph],
|
|
149
|
+
insetLeft: pt(0),
|
|
150
|
+
insetTop: pt(0),
|
|
151
|
+
insetRight: pt(0),
|
|
152
|
+
insetBottom: pt(0)
|
|
153
|
+
},
|
|
154
|
+
paragraphProperties: {}
|
|
155
|
+
}
|
|
156
|
+
};
|
|
157
|
+
}
|
|
77
158
|
/** Collapse losses sharing a `detail` message (the same colour space dropped on many pages). */
|
|
78
159
|
function dedupeLosses(losses) {
|
|
79
160
|
const byDetail = /* @__PURE__ */ new Map();
|
|
@@ -84,10 +165,14 @@ function dedupeLosses(losses) {
|
|
|
84
165
|
* Turn a lifted {@link PdfVector} path (filled EP10 / stroked EP11) into a
|
|
85
166
|
* custom-geometry shape {@link BodyElement}. Page-space points (y-up) become
|
|
86
167
|
* path-space (bbox-relative, y-down); the shape is sized from the bounding box
|
|
87
|
-
* (plus the stroke thickness)
|
|
88
|
-
*
|
|
168
|
+
* (plus the stroke thickness). A fill becomes a solid fill, a stroke the outline.
|
|
169
|
+
*
|
|
170
|
+
* Given the page's frame the shape is ANCHORED where the page drew it, behind
|
|
171
|
+
* the text, rather than taking a place of its own in the flow. A drawing is not
|
|
172
|
+
* a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
|
|
173
|
+
* its forty-nine paths one under another spilled it onto a second page.
|
|
89
174
|
*/
|
|
90
|
-
function shapeBlock(v) {
|
|
175
|
+
function shapeBlock(v, frame, zOrder) {
|
|
91
176
|
const w = v.maxX - v.minX;
|
|
92
177
|
const h = v.maxY - v.minY;
|
|
93
178
|
const fx = (x) => x - v.minX;
|
|
@@ -116,22 +201,41 @@ function shapeBlock(v) {
|
|
|
116
201
|
case "close": return { cmd: "close" };
|
|
117
202
|
}
|
|
118
203
|
});
|
|
119
|
-
const
|
|
204
|
+
const HAIRLINE_PT = .1;
|
|
205
|
+
const stated = v.lineWidth ?? .75;
|
|
206
|
+
const pen = v.strokeHex !== void 0 ? stated > 0 ? stated : HAIRLINE_PT : 0;
|
|
207
|
+
const thick = v.strokeHex !== void 0 ? Math.max(pen, .5) : 0;
|
|
208
|
+
const alpha = v.alpha !== void 0 ? { alpha: v.alpha } : {};
|
|
120
209
|
const fill = v.gradient !== void 0 ? {
|
|
121
210
|
kind: "gradient",
|
|
122
|
-
gradient: v.gradient
|
|
211
|
+
gradient: v.gradient,
|
|
212
|
+
...alpha
|
|
123
213
|
} : v.fillHex !== void 0 ? {
|
|
124
214
|
kind: "solid",
|
|
125
|
-
colorHex: v.fillHex
|
|
215
|
+
colorHex: v.fillHex,
|
|
216
|
+
...alpha
|
|
126
217
|
} : { kind: "none" };
|
|
127
218
|
const line = v.strokeHex !== void 0 ? {
|
|
128
|
-
width: pt(
|
|
219
|
+
width: pt(pen),
|
|
129
220
|
colorHex: v.strokeHex,
|
|
130
221
|
fill: "solid"
|
|
131
222
|
} : void 0;
|
|
223
|
+
const float = frame !== void 0 ? {
|
|
224
|
+
wrap: "none",
|
|
225
|
+
...zOrder !== void 0 ? { zOrder } : {},
|
|
226
|
+
posH: {
|
|
227
|
+
relativeFrom: "page",
|
|
228
|
+
offsetPt: pt(v.minX - frame.left)
|
|
229
|
+
},
|
|
230
|
+
posV: {
|
|
231
|
+
relativeFrom: "page",
|
|
232
|
+
offsetPt: pt(Math.max(0, frame.top - v.maxY))
|
|
233
|
+
}
|
|
234
|
+
} : void 0;
|
|
132
235
|
return {
|
|
133
236
|
kind: "shape",
|
|
134
237
|
shape: {
|
|
238
|
+
...float ? { float } : {},
|
|
135
239
|
width: pt(Math.max(w, thick)),
|
|
136
240
|
height: pt(Math.max(h, thick)),
|
|
137
241
|
geometry: {
|
|
@@ -160,10 +264,11 @@ function shapeBlock(v) {
|
|
|
160
264
|
* `A4`. Returns `undefined` when there is no usable first-page box.
|
|
161
265
|
*/
|
|
162
266
|
function sectionFromPdfPages(pages) {
|
|
163
|
-
const
|
|
164
|
-
if (!
|
|
165
|
-
const
|
|
166
|
-
const
|
|
267
|
+
const first = pages[0];
|
|
268
|
+
if (!first) return void 0;
|
|
269
|
+
const shown = displayOf(first);
|
|
270
|
+
const width = shown.width;
|
|
271
|
+
const height = shown.height;
|
|
167
272
|
if (!(width > 0 && height > 0)) return void 0;
|
|
168
273
|
return {
|
|
169
274
|
pageSize: {
|
|
@@ -188,15 +293,16 @@ function sectionFromPdfPages(pages) {
|
|
|
188
293
|
* both reconstruction paths (the tagged fast-path EP3 and the heuristic layout
|
|
189
294
|
* path EP4).
|
|
190
295
|
*/
|
|
191
|
-
function buildFlowDoc(body, resources = new ResourceStore(), section) {
|
|
296
|
+
function buildFlowDoc(body, resources = new ResourceStore(), section, embeddedFonts) {
|
|
192
297
|
return {
|
|
193
298
|
kind: "flow",
|
|
194
299
|
body: resolveBodyStyles([...body], EMPTY_STYLE_SHEET),
|
|
195
300
|
sections: [],
|
|
196
301
|
...section ? { section } : {},
|
|
302
|
+
...embeddedFonts && embeddedFonts.size > 0 ? { embeddedFonts } : {},
|
|
197
303
|
styles: EMPTY_STYLE_SHEET,
|
|
198
304
|
resources
|
|
199
305
|
};
|
|
200
306
|
}
|
|
201
307
|
//#endregion
|
|
202
|
-
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock };
|
|
308
|
+
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock };
|
|
@@ -1,4 +1,6 @@
|
|
|
1
|
+
import { parseTtf } from "../core/font/ttf-parser.js";
|
|
1
2
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
3
|
+
import { embeddedFontName } from "./embedded-fonts.js";
|
|
2
4
|
import { parseToUnicodeCMap } from "./cmap.js";
|
|
3
5
|
//#region src/pdf-reader/font.ts
|
|
4
6
|
/**
|
|
@@ -21,16 +23,195 @@ function buildContentFont(file, fontDict) {
|
|
|
21
23
|
if (tu instanceof PdfStream) {
|
|
22
24
|
const parsed = parseToUnicodeCMap(file.streamData(tu));
|
|
23
25
|
toUnicode = parsed.map;
|
|
24
|
-
codeBytes = parsed.codeBytes;
|
|
26
|
+
if (isType0) codeBytes = parsed.codeBytes;
|
|
25
27
|
}
|
|
26
|
-
const
|
|
28
|
+
const unicode = (isType0 && toUnicode.size === 0 ? embeddedCmap(file, fontDict) : void 0) ?? toUnicode;
|
|
27
29
|
const bytesPerCode = codeBytes;
|
|
30
|
+
const style = faceStyle(file, fontDict, isType0);
|
|
31
|
+
const name = embeddedFontName(file, fontDict);
|
|
32
|
+
const type3 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type3" ? type3Face(file, fontDict) : void 0;
|
|
33
|
+
const simple = simpleWidths(file, fontDict);
|
|
34
|
+
const width = isType0 ? cidWidths(file, fontDict) : type3 ? (code) => simple(code) * type3.matrix[0] * 1e3 : simple;
|
|
28
35
|
return {
|
|
29
36
|
bytesPerCode,
|
|
30
|
-
|
|
31
|
-
|
|
37
|
+
...type3 ? { type3 } : {},
|
|
38
|
+
...name !== void 0 ? { name } : {},
|
|
39
|
+
decode: (codes) => codes.map((c) => unicode.get(c) ?? (bytesPerCode === 1 ? String.fromCharCode(c) : "")).join(""),
|
|
40
|
+
width,
|
|
41
|
+
...style
|
|
32
42
|
};
|
|
33
43
|
}
|
|
44
|
+
/** The last Unicode code point in the BMP, and the surrogate block inside it. */
|
|
45
|
+
var BMP_END = 65535;
|
|
46
|
+
var SURROGATE_FIRST = 55296;
|
|
47
|
+
var SURROGATE_LAST = 57343;
|
|
48
|
+
/**
|
|
49
|
+
* §9.10.2 — code → Unicode read out of an embedded TrueType program's `cmap`,
|
|
50
|
+
* for a composite font that states no `/ToUnicode`.
|
|
51
|
+
*
|
|
52
|
+
* A `cmap` maps the other way, code point → glyph, so it is walked once and
|
|
53
|
+
* turned round. With `Identity-H` and no `/CIDToGIDMap` a code IS a glyph
|
|
54
|
+
* index, which is the case this exists for; a `/CIDToGIDMap` stream is read
|
|
55
|
+
* where one is present.
|
|
56
|
+
*
|
|
57
|
+
* Only TrueType (`/FontFile2`) is read. A CFF program (`/FontFile3`) carries
|
|
58
|
+
* its own charset and is a separate reading; a font with neither says nothing
|
|
59
|
+
* about its glyphs and nothing is invented.
|
|
60
|
+
*/
|
|
61
|
+
function embeddedCmap(file, fontDict) {
|
|
62
|
+
const cidFont = descendantFont(file, fontDict);
|
|
63
|
+
const descriptor = file.resolve(cidFont.get("FontDescriptor") ?? PDF_NULL);
|
|
64
|
+
if (!(descriptor instanceof Map)) return void 0;
|
|
65
|
+
const program = file.resolve(descriptor.get("FontFile2") ?? PDF_NULL);
|
|
66
|
+
if (!(program instanceof PdfStream)) return void 0;
|
|
67
|
+
let glyphOf;
|
|
68
|
+
try {
|
|
69
|
+
glyphOf = parseTtf(file.streamData(program)).glyphForCodepoint;
|
|
70
|
+
} catch {
|
|
71
|
+
return;
|
|
72
|
+
}
|
|
73
|
+
const byGlyph = /* @__PURE__ */ new Map();
|
|
74
|
+
for (let cp = 32; cp <= BMP_END; cp++) {
|
|
75
|
+
if (cp >= SURROGATE_FIRST && cp <= SURROGATE_LAST) continue;
|
|
76
|
+
let gid = 0;
|
|
77
|
+
try {
|
|
78
|
+
gid = glyphOf(cp);
|
|
79
|
+
} catch {
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
if (gid > 0 && !byGlyph.has(gid)) byGlyph.set(gid, String.fromCodePoint(cp));
|
|
83
|
+
}
|
|
84
|
+
if (byGlyph.size === 0) return void 0;
|
|
85
|
+
const cidToGid = readCidToGid(file, cidFont);
|
|
86
|
+
if (!cidToGid) return byGlyph;
|
|
87
|
+
const out = /* @__PURE__ */ new Map();
|
|
88
|
+
cidToGid.forEach((gid, cid) => {
|
|
89
|
+
const text = byGlyph.get(gid);
|
|
90
|
+
if (text !== void 0) out.set(cid, text);
|
|
91
|
+
});
|
|
92
|
+
return out.size > 0 ? out : void 0;
|
|
93
|
+
}
|
|
94
|
+
/** §9.7.4.2 `/CIDToGIDMap` — a stream of two-byte glyph indices, CID by CID. */
|
|
95
|
+
function readCidToGid(file, cidFont) {
|
|
96
|
+
const map = file.resolve(cidFont.get("CIDToGIDMap") ?? PDF_NULL);
|
|
97
|
+
if (!(map instanceof PdfStream)) return void 0;
|
|
98
|
+
const bytes = file.streamData(map);
|
|
99
|
+
const out = [];
|
|
100
|
+
for (let i = 0; i + 1 < bytes.length; i += 2) out.push(bytes[i] << 8 | bytes[i + 1]);
|
|
101
|
+
return out;
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* §9.6.5 — a Type 3 font's glyphs are content streams, not outlines: what the
|
|
105
|
+
* face draws is whatever each procedure paints, in the resources the font
|
|
106
|
+
* states. `/Encoding` `/Differences` names the procedure a code selects and
|
|
107
|
+
* `/CharProcs` holds it.
|
|
108
|
+
*
|
|
109
|
+
* ContentStreamCycleType3insideType3.pdf is a page of them — a stroked square
|
|
110
|
+
* and a stroked triangle, with a second Type 3 font shown from inside the
|
|
111
|
+
* square — and with the procedures unread the page came back as two letters of
|
|
112
|
+
* substituted type an eighth of an inch tall.
|
|
113
|
+
*/
|
|
114
|
+
function type3Face(file, fontDict) {
|
|
115
|
+
const procs = file.resolve(fontDict.get("CharProcs") ?? PDF_NULL);
|
|
116
|
+
if (!(procs instanceof Map)) return void 0;
|
|
117
|
+
const names = differences(file, fontDict);
|
|
118
|
+
const resourcesVal = file.resolve(fontDict.get("Resources") ?? PDF_NULL);
|
|
119
|
+
return {
|
|
120
|
+
matrix: fontMatrix(file, fontDict),
|
|
121
|
+
resources: resourcesVal instanceof Map ? resourcesVal : void 0,
|
|
122
|
+
proc: (code) => {
|
|
123
|
+
const glyph = names.get(code);
|
|
124
|
+
if (glyph === void 0) return void 0;
|
|
125
|
+
const stream = file.resolve(procs.get(glyph) ?? PDF_NULL);
|
|
126
|
+
return stream instanceof PdfStream ? stream : void 0;
|
|
127
|
+
}
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
/** §9.6.5 `/FontMatrix` — glyph space to text space; a thousandth by default. */
|
|
131
|
+
function fontMatrix(file, fontDict) {
|
|
132
|
+
const m = file.resolve(fontDict.get("FontMatrix") ?? PDF_NULL);
|
|
133
|
+
if (!Array.isArray(m) || m.length !== 6) return [
|
|
134
|
+
.001,
|
|
135
|
+
0,
|
|
136
|
+
0,
|
|
137
|
+
.001,
|
|
138
|
+
0,
|
|
139
|
+
0
|
|
140
|
+
];
|
|
141
|
+
const n = m.map((v) => asNumber(file.resolve(v), 0));
|
|
142
|
+
return [
|
|
143
|
+
n[0],
|
|
144
|
+
n[1],
|
|
145
|
+
n[2],
|
|
146
|
+
n[3],
|
|
147
|
+
n[4],
|
|
148
|
+
n[5]
|
|
149
|
+
];
|
|
150
|
+
}
|
|
151
|
+
/** §9.6.6.1 `/Encoding` `/Differences` — code → glyph name, as the array runs. */
|
|
152
|
+
function differences(file, fontDict) {
|
|
153
|
+
const out = /* @__PURE__ */ new Map();
|
|
154
|
+
const encoding = file.resolve(fontDict.get("Encoding") ?? PDF_NULL);
|
|
155
|
+
if (!(encoding instanceof Map)) return out;
|
|
156
|
+
const list = file.resolve(encoding.get("Differences") ?? PDF_NULL);
|
|
157
|
+
if (!Array.isArray(list)) return out;
|
|
158
|
+
let code = 0;
|
|
159
|
+
for (const entry of list) {
|
|
160
|
+
const value = file.resolve(entry);
|
|
161
|
+
if (typeof value === "number") code = value;
|
|
162
|
+
else if (value instanceof PdfName) out.set(code++, value.value);
|
|
163
|
+
}
|
|
164
|
+
return out;
|
|
165
|
+
}
|
|
166
|
+
/** §9.8.2 `/Flags` — bit 7 is Italic, bit 19 ForceBold (bits numbered from 1). */
|
|
167
|
+
var FLAG_ITALIC = 64;
|
|
168
|
+
var FLAG_FORCE_BOLD = 1 << 18;
|
|
169
|
+
/** §9.8.1 `/FontWeight` — 400 is normal, 700 bold; 600 is where "bold" begins. */
|
|
170
|
+
var BOLD_WEIGHT = 600;
|
|
171
|
+
/**
|
|
172
|
+
* §9.8.1 — whether the face a run is shown in is bold or slanted.
|
|
173
|
+
*
|
|
174
|
+
* A descriptor is the witness where there is one, and the ONLY witness: it
|
|
175
|
+
* states `/FontWeight`, `/ItalicAngle` and the `/Flags` bits, so one that gives
|
|
176
|
+
* neither a weight nor the ForceBold bit is saying the face is not bold.
|
|
177
|
+
* ArabicCIDTrueType.pdf shows why that matters — two of its four faces are
|
|
178
|
+
* called `NewBasrahBold` and `DamascusBold`, which is the family's own name and
|
|
179
|
+
* not a weight, and reading the name over the descriptor set two lines heavy
|
|
180
|
+
* that no reader sets heavy.
|
|
181
|
+
*
|
|
182
|
+
* The name is read only where no descriptor exists at all, which is the
|
|
183
|
+
* standard-14 case (§9.6.2.2): `Helvetica-BoldOblique` has nothing else to go
|
|
184
|
+
* on. The subset prefix (`ISVAYD+`) is dropped first — six arbitrary capitals
|
|
185
|
+
* may spell anything.
|
|
186
|
+
*/
|
|
187
|
+
function faceStyle(file, fontDict, isType0) {
|
|
188
|
+
const owner = isType0 ? descendantFont(file, fontDict) : fontDict;
|
|
189
|
+
const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
|
|
190
|
+
if (descriptor instanceof Map) {
|
|
191
|
+
const flags = asNumber(file.resolve(descriptor.get("Flags") ?? PDF_NULL), 0);
|
|
192
|
+
const weight = asNumber(file.resolve(descriptor.get("FontWeight") ?? PDF_NULL), 0);
|
|
193
|
+
const slant = asNumber(file.resolve(descriptor.get("ItalicAngle") ?? PDF_NULL), 0);
|
|
194
|
+
const bold = weight >= BOLD_WEIGHT || (flags & FLAG_FORCE_BOLD) !== 0;
|
|
195
|
+
const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0;
|
|
196
|
+
return {
|
|
197
|
+
...bold ? { bold } : {},
|
|
198
|
+
...italic ? { italic } : {}
|
|
199
|
+
};
|
|
200
|
+
}
|
|
201
|
+
const name = asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)).replace(/^[A-Z]{6}\+/u, "");
|
|
202
|
+
const bold = /bold|black|heavy/iu.test(name);
|
|
203
|
+
const italic = /italic|oblique/iu.test(name);
|
|
204
|
+
return {
|
|
205
|
+
...bold ? { bold } : {},
|
|
206
|
+
...italic ? { italic } : {}
|
|
207
|
+
};
|
|
208
|
+
}
|
|
209
|
+
/** §9.7.4 — a `/Type0` font's one descendant CIDFont, which owns the descriptor. */
|
|
210
|
+
function descendantFont(file, fontDict) {
|
|
211
|
+
const descFonts = file.resolve(fontDict.get("DescendantFonts") ?? PDF_NULL);
|
|
212
|
+
const first = Array.isArray(descFonts) ? file.resolve(descFonts[0] ?? PDF_NULL) : PDF_NULL;
|
|
213
|
+
return first instanceof Map ? first : /* @__PURE__ */ new Map();
|
|
214
|
+
}
|
|
34
215
|
function simpleWidths(file, fontDict) {
|
|
35
216
|
const first = asNumber(file.resolve(fontDict.get("FirstChar") ?? PDF_NULL), 0);
|
|
36
217
|
const widthsVal = file.resolve(fontDict.get("Widths") ?? PDF_NULL);
|
|
@@ -3,6 +3,7 @@ import { encodePng } from "../core/png-encode.js";
|
|
|
3
3
|
import { lzwDecodeMsb } from "../core/lzw.js";
|
|
4
4
|
import { reversePredictor } from "./predictor.js";
|
|
5
5
|
import { decodeCcitt } from "./ccitt.js";
|
|
6
|
+
import { decodeJpeg } from "./jpeg.js";
|
|
6
7
|
import { unzlibSync } from "fflate";
|
|
7
8
|
//#region src/pdf-reader/image-decode.ts
|
|
8
9
|
var MAX_PIXELS = 4e7;
|
|
@@ -30,10 +31,13 @@ function decodePdfImage(file, stream) {
|
|
|
30
31
|
const filters = filterNames(file, d);
|
|
31
32
|
const last = filters[filters.length - 1];
|
|
32
33
|
if (last === "DCTDecode" || last === "DCT") {
|
|
33
|
-
const
|
|
34
|
+
const jpegBytes = applyChainExceptLast(filters, stream.data);
|
|
35
|
+
const masked = hasSMask(file, d) ? maskedJpeg(file, d, jpegBytes, width, height) : void 0;
|
|
36
|
+
if (masked) return masked;
|
|
37
|
+
const degraded = hasSMask(file, d) ? "image transparency dropped (JPEG could not be decoded to attach its mask)" : void 0;
|
|
34
38
|
return {
|
|
35
39
|
ok: true,
|
|
36
|
-
bytes:
|
|
40
|
+
bytes: jpegBytes,
|
|
37
41
|
format: "jpeg",
|
|
38
42
|
widthPx: width,
|
|
39
43
|
heightPx: height,
|
|
@@ -276,14 +280,49 @@ function unpackSamples(raw, width, height, ncomp, bpc) {
|
|
|
276
280
|
}
|
|
277
281
|
return out;
|
|
278
282
|
}
|
|
283
|
+
/**
|
|
284
|
+
* A JPEG with its `/SMask` folded in, as a PNG.
|
|
285
|
+
*
|
|
286
|
+
* A JPEG has no alpha channel, so the only way to honour the mask is to decode
|
|
287
|
+
* the picture, decode the mask, put them together and re-encode. Both are
|
|
288
|
+
* usually DCT themselves — 22060_A1_01_Plans.pdf stores each floor plan as a
|
|
289
|
+
* wash and its line work as a grey JPEG mask of the same 2480x2630 — so this
|
|
290
|
+
* costs a full baseline decode of two images, and is spent only where a mask
|
|
291
|
+
* actually exists.
|
|
292
|
+
*
|
|
293
|
+
* @returns The recomposed PNG, or `undefined` when either image is beyond the
|
|
294
|
+
* baseline decoder — the caller then carries the JPEG through as it is.
|
|
295
|
+
*/
|
|
296
|
+
function maskedJpeg(file, d, jpegBytes, width, height) {
|
|
297
|
+
const image = decodeJpeg(jpegBytes);
|
|
298
|
+
if (!image) return void 0;
|
|
299
|
+
const alpha = decodeSMask(file, d, image.width, image.height);
|
|
300
|
+
if (!alpha) return void 0;
|
|
301
|
+
const { color: pngColor, samples } = combineAlpha(image.components === 1 ? {
|
|
302
|
+
color: "gray",
|
|
303
|
+
samples: image.samples
|
|
304
|
+
} : {
|
|
305
|
+
color: "rgb",
|
|
306
|
+
samples: image.samples
|
|
307
|
+
}, alpha);
|
|
308
|
+
return {
|
|
309
|
+
ok: true,
|
|
310
|
+
bytes: encodePng(image.width, image.height, pngColor, samples),
|
|
311
|
+
format: "png",
|
|
312
|
+
widthPx: width,
|
|
313
|
+
heightPx: height
|
|
314
|
+
};
|
|
315
|
+
}
|
|
279
316
|
function decodeSMask(file, d, width, height) {
|
|
280
317
|
const sm = file.resolve(d.get("SMask") ?? PDF_NULL);
|
|
281
318
|
if (!(sm instanceof PdfStream)) return void 0;
|
|
282
319
|
const sw = intOf(file.get(sm.dict, "Width"));
|
|
283
320
|
const sh = intOf(file.get(sm.dict, "Height"));
|
|
284
321
|
if (sw <= 0 || sh <= 0) return void 0;
|
|
285
|
-
const
|
|
286
|
-
|
|
322
|
+
const maskFilters = filterNames(file, sm.dict);
|
|
323
|
+
const maskLast = maskFilters[maskFilters.length - 1];
|
|
324
|
+
const decoded = maskLast === "DCTDecode" || maskLast === "DCT" ? jpegSamples(applyChainExceptLast(maskFilters, sm.data)) : decodeToSamples(file, sm, maskFilters, sw, sh);
|
|
325
|
+
if (decoded === void 0 || typeof decoded === "string") return void 0;
|
|
287
326
|
const ch = decoded.color === "rgb" ? 3 : 1;
|
|
288
327
|
const gray = new Uint8Array(sw * sh);
|
|
289
328
|
for (let i = 0; i < sw * sh; i++) gray[i] = decoded.samples[i * ch];
|
|
@@ -298,6 +337,18 @@ function decodeSMask(file, d, width, height) {
|
|
|
298
337
|
}
|
|
299
338
|
return { data: out };
|
|
300
339
|
}
|
|
340
|
+
/** A DCT-coded mask decoded to samples, or `undefined` past the baseline decoder. */
|
|
341
|
+
function jpegSamples(bytes) {
|
|
342
|
+
const decoded = decodeJpeg(bytes);
|
|
343
|
+
if (!decoded) return void 0;
|
|
344
|
+
return decoded.components === 1 ? {
|
|
345
|
+
color: "gray",
|
|
346
|
+
samples: decoded.samples
|
|
347
|
+
} : {
|
|
348
|
+
color: "rgb",
|
|
349
|
+
samples: decoded.samples
|
|
350
|
+
};
|
|
351
|
+
}
|
|
301
352
|
function combineAlpha(color, alpha) {
|
|
302
353
|
const px = alpha.data.length;
|
|
303
354
|
if (color.color === "gray") {
|
|
@@ -17,6 +17,12 @@ export interface PdfImage {
|
|
|
17
17
|
readonly y: number;
|
|
18
18
|
/** Enclosing marked-content id, if the placement was inside a `/Figure`. */
|
|
19
19
|
readonly mcid?: number;
|
|
20
|
+
/**
|
|
21
|
+
* §8.5.3 — where this was painted, as the chain of positions leading to it.
|
|
22
|
+
* The same key a lifted path carries, so the two can be ordered against each
|
|
23
|
+
* other: a picture drawn over a filled box has the larger key.
|
|
24
|
+
*/
|
|
25
|
+
readonly orderKey: ReadonlyArray<number>;
|
|
20
26
|
}
|
|
21
27
|
/** The images lifted off one page plus any losses for images that could not be reconstructed. */
|
|
22
28
|
export interface PageImages {
|
|
@@ -2,6 +2,7 @@ import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
|
2
2
|
import { FEATURES } from "../core/ir/features.js";
|
|
3
3
|
import { interpretContent, multiply } from "./content.js";
|
|
4
4
|
import { decodePdfImage } from "./image-decode.js";
|
|
5
|
+
import { collectPageAppearances } from "./annots.js";
|
|
5
6
|
//#region src/pdf-reader/images.ts
|
|
6
7
|
var NO_FONTS = /* @__PURE__ */ new Map();
|
|
7
8
|
var MAX_FORM_DEPTH = 12;
|
|
@@ -26,10 +27,23 @@ function collectPageImages(file, page) {
|
|
|
26
27
|
detail
|
|
27
28
|
});
|
|
28
29
|
};
|
|
29
|
-
const walk = (resources, content, baseCtm, depth, inheritedMcid) => {
|
|
30
|
+
const walk = (resources, content, baseCtm, depth, inheritedMcid, prefix) => {
|
|
30
31
|
const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
|
|
31
32
|
const xobjDict = xobjects instanceof Map ? xobjects : void 0;
|
|
32
|
-
|
|
33
|
+
const result = interpretContent(content, NO_FONTS, baseCtm);
|
|
34
|
+
const patterns = resources ? file.get(resources, "Pattern") : PDF_NULL;
|
|
35
|
+
const patternDict = patterns instanceof Map ? patterns : void 0;
|
|
36
|
+
for (const vector of result.vectors) {
|
|
37
|
+
if (vector.patternName === void 0 || depth >= MAX_FORM_DEPTH) continue;
|
|
38
|
+
const stream = patternDict ? file.resolve(patternDict.get(vector.patternName) ?? PDF_NULL) : PDF_NULL;
|
|
39
|
+
if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
|
|
40
|
+
if (file.get(stream.dict, "PatternType") !== 1) continue;
|
|
41
|
+
visiting.add(stream);
|
|
42
|
+
const patternRes = file.get(stream.dict, "Resources");
|
|
43
|
+
walk(patternRes instanceof Map ? patternRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), baseCtm), depth + 1, inheritedMcid, [...prefix, vector.order]);
|
|
44
|
+
visiting.delete(stream);
|
|
45
|
+
}
|
|
46
|
+
for (const placement of result.images) {
|
|
33
47
|
if (images.length >= MAX_IMAGES) return;
|
|
34
48
|
const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
|
|
35
49
|
if (!(stream instanceof PdfStream)) continue;
|
|
@@ -38,13 +52,16 @@ function collectPageImages(file, page) {
|
|
|
38
52
|
if (subtype === "Image") {
|
|
39
53
|
const decoded = decodePdfImage(file, stream);
|
|
40
54
|
if (decoded.ok) {
|
|
41
|
-
images.push(
|
|
55
|
+
images.push({
|
|
56
|
+
...geometry(placement.ctm, decoded, mcid),
|
|
57
|
+
orderKey: [...prefix, placement.order]
|
|
58
|
+
});
|
|
42
59
|
if (decoded.degraded) addLoss("degraded", decoded.degraded);
|
|
43
60
|
} else addLoss(decoded.severity, decoded.detail);
|
|
44
61
|
} else if (subtype === "Form" && depth < MAX_FORM_DEPTH && !visiting.has(stream)) {
|
|
45
62
|
visiting.add(stream);
|
|
46
63
|
const formRes = file.get(stream.dict, "Resources");
|
|
47
|
-
walk(formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, mcid);
|
|
64
|
+
walk(formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, mcid, [...prefix, placement.order]);
|
|
48
65
|
visiting.delete(stream);
|
|
49
66
|
}
|
|
50
67
|
}
|
|
@@ -56,7 +73,10 @@ function collectPageImages(file, page) {
|
|
|
56
73
|
1,
|
|
57
74
|
0,
|
|
58
75
|
0
|
|
59
|
-
], 0, void 0);
|
|
76
|
+
], 0, void 0, []);
|
|
77
|
+
collectPageAppearances(file, page).forEach((appearance, index) => {
|
|
78
|
+
walk(appearance.resources ?? page.resources, file.streamData(appearance.stream), appearance.ctm, 1, void 0, [Number.MAX_SAFE_INTEGER, index]);
|
|
79
|
+
});
|
|
60
80
|
return {
|
|
61
81
|
images,
|
|
62
82
|
losses: [...lossByDetail.values()]
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/** A decoded JPEG: 8-bit interleaved samples, one or three components. */
|
|
2
|
+
export interface DecodedJpeg {
|
|
3
|
+
readonly width: number;
|
|
4
|
+
readonly height: number;
|
|
5
|
+
/** 1 = grayscale, 3 = RGB (already converted from YCbCr where it applies). */
|
|
6
|
+
readonly components: 1 | 3;
|
|
7
|
+
/** `width * height * components` bytes, row-major. */
|
|
8
|
+
readonly samples: Uint8Array;
|
|
9
|
+
}
|
|
10
|
+
/**
|
|
11
|
+
* Decode a baseline JPEG to interleaved 8-bit samples.
|
|
12
|
+
*
|
|
13
|
+
* @param bytes The JPEG stream, starting at its SOI marker.
|
|
14
|
+
* @returns The decoded image, or `undefined` when the stream is not a baseline
|
|
15
|
+
* JPEG this decoder handles (progressive, arithmetic, 12-bit, CMYK) or
|
|
16
|
+
* is malformed — the caller then carries the original bytes through.
|
|
17
|
+
*/
|
|
18
|
+
export declare function decodeJpeg(bytes: Uint8Array): DecodedJpeg | undefined;
|