reamkit 1.24.0 → 1.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -11
- package/dist/esm/core/converter/facade.d.ts +3 -3
- package/dist/esm/core/converter/facade.js +12 -0
- package/dist/esm/core/converter/ream.d.ts +48 -4
- package/dist/esm/core/converter/ream.js +25 -4
- package/dist/esm/core/document-model/index.d.ts +1 -1
- package/dist/esm/core/document-model/types.d.ts +17 -0
- package/dist/esm/core/outline.d.ts +17 -0
- package/dist/esm/core/outline.js +30 -0
- package/dist/esm/core/style-cascade/resolver.js +1 -0
- package/dist/esm/core/style-cascade/types.d.ts +3 -1
- package/dist/esm/excel/sheet-to-flow.d.ts +10 -0
- package/dist/esm/excel/sheet-to-flow.js +14 -1
- package/dist/esm/html/html-writer.js +3 -2
- package/dist/esm/index.d.ts +3 -0
- package/dist/esm/index.js +2 -1
- package/dist/esm/layout/page-doc.js +1 -1
- package/dist/esm/layout/styled-layout.js +41 -15
- package/dist/esm/markdown/markdown-writer.d.ts +41 -0
- package/dist/esm/markdown/markdown-writer.js +733 -0
- package/dist/esm/pdf/styled-page-emitter.js +20 -1
- package/dist/esm/pdf-reader/annots.d.ts +24 -0
- package/dist/esm/pdf-reader/annots.js +126 -0
- package/dist/esm/pdf-reader/content.d.ts +131 -5
- package/dist/esm/pdf-reader/content.js +169 -12
- package/dist/esm/pdf-reader/display.d.ts +56 -0
- package/dist/esm/pdf-reader/display.js +162 -0
- package/dist/esm/pdf-reader/document.d.ts +36 -1
- package/dist/esm/pdf-reader/document.js +92 -25
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +31 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +94 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +61 -6
- package/dist/esm/pdf-reader/flow-build.js +128 -22
- package/dist/esm/pdf-reader/font.js +185 -4
- package/dist/esm/pdf-reader/image-decode.js +55 -4
- package/dist/esm/pdf-reader/images.d.ts +6 -0
- package/dist/esm/pdf-reader/images.js +25 -5
- package/dist/esm/pdf-reader/jpeg.d.ts +18 -0
- package/dist/esm/pdf-reader/jpeg.js +419 -0
- package/dist/esm/pdf-reader/layout.d.ts +1 -1
- package/dist/esm/pdf-reader/layout.js +221 -32
- package/dist/esm/pdf-reader/pattern-tint.d.ts +17 -0
- package/dist/esm/pdf-reader/pattern-tint.js +181 -0
- package/dist/esm/pdf-reader/reader.d.ts +9 -1
- package/dist/esm/pdf-reader/reader.js +22 -6
- package/dist/esm/pdf-reader/shading.d.ts +14 -0
- package/dist/esm/pdf-reader/shading.js +27 -1
- package/dist/esm/pdf-reader/tagged.js +156 -17
- package/dist/esm/pdf-reader/text.d.ts +13 -1
- package/dist/esm/pdf-reader/text.js +70 -3
- package/dist/esm/pdf-reader/vector.d.ts +25 -1
- package/dist/esm/pdf-reader/vector.js +168 -12
- package/dist/esm/pptx/slide-parser.js +5 -0
- package/dist/esm/word/docx-writer.js +11 -1
- package/dist/esm/word/drawing-parser.js +7 -1
- package/package.json +1 -1
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
import { pt } from "../core/ir/units.js";
|
|
2
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
|
-
import {
|
|
3
|
+
import { displayOf, placeVectors } from "./display.js";
|
|
4
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
|
|
5
|
+
import { collectEmbeddedFonts } from "./embedded-fonts.js";
|
|
4
6
|
import { collectPageImages } from "./images.js";
|
|
5
7
|
import { extractPageText } from "./text.js";
|
|
8
|
+
import { collectPageVectors } from "./vector.js";
|
|
6
9
|
import { readStructTree } from "./struct-tree.js";
|
|
7
10
|
//#region src/pdf-reader/tagged.ts
|
|
8
11
|
var ASSUMED_CONTENT_WIDTH_PT = 468;
|
|
12
|
+
var MIN_COLUMN_PT = 6;
|
|
9
13
|
/**
|
|
10
14
|
* Reconstruct a {@link Reconstruction} from a tagged PDF's logical structure
|
|
11
15
|
* (E-PDF EP3 — the honest inverse of the tagged PDF Ream writes). Walks the
|
|
@@ -54,10 +58,19 @@ function reconstructTaggedPdf(file) {
|
|
|
54
58
|
const spans = [];
|
|
55
59
|
node.mcids.forEach(({ page, mcid }, i) => {
|
|
56
60
|
if (i > 0) spans.push({ text: " " });
|
|
57
|
-
for (const run of runsOfMcid(page, mcid)) spans.push(
|
|
61
|
+
for (const run of runsOfMcid(page, mcid)) spans.push({
|
|
58
62
|
text: run.text,
|
|
59
|
-
|
|
60
|
-
|
|
63
|
+
sizePt: run.fontSizePt,
|
|
64
|
+
...run.colorHex !== "000000" ? { colorHex: run.colorHex } : {},
|
|
65
|
+
...run.fontName !== void 0 ? { fontName: run.fontName } : {},
|
|
66
|
+
...run.outlineHex !== void 0 ? { outline: {
|
|
67
|
+
colorHex: run.outlineHex,
|
|
68
|
+
widthPt: pt(run.outlineWidthPt ?? 1)
|
|
69
|
+
} } : {},
|
|
70
|
+
...run.bold ? { bold: true } : {},
|
|
71
|
+
...run.italic ? { italic: true } : {},
|
|
72
|
+
...run.href !== void 0 ? { href: run.href } : {}
|
|
73
|
+
});
|
|
61
74
|
});
|
|
62
75
|
return spans;
|
|
63
76
|
};
|
|
@@ -87,23 +100,42 @@ function reconstructTaggedPdf(file) {
|
|
|
87
100
|
for (const child of node.children) emit(child, out);
|
|
88
101
|
}
|
|
89
102
|
function buildTable(tableNode) {
|
|
90
|
-
const
|
|
103
|
+
const raw = [];
|
|
91
104
|
const collectRows = (n) => {
|
|
92
|
-
for (const child of n.children) if (child.type === "TR")
|
|
105
|
+
for (const child of n.children) if (child.type === "TR") raw.push(buildRow(child));
|
|
93
106
|
else if (child.type === "THead" || child.type === "TBody" || child.type === "TFoot") collectRows(child);
|
|
94
107
|
};
|
|
95
108
|
collectRows(tableNode);
|
|
96
|
-
if (
|
|
97
|
-
const
|
|
98
|
-
const colWidth = pt(Math.max(1, ASSUMED_CONTENT_WIDTH_PT / numCols));
|
|
109
|
+
if (raw.length === 0) return void 0;
|
|
110
|
+
const laid = layOutColumns(raw);
|
|
99
111
|
return {
|
|
100
112
|
kind: "table",
|
|
101
113
|
table: {
|
|
102
114
|
properties: {},
|
|
103
|
-
grid:
|
|
104
|
-
rows
|
|
115
|
+
grid: laid.grid,
|
|
116
|
+
rows: laid.rows
|
|
117
|
+
}
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
/**
|
|
121
|
+
* The span of page x every glyph under a node covers — measured, since the
|
|
122
|
+
* interpreter advances the text matrix by the font's own widths (§9.4.4).
|
|
123
|
+
*/
|
|
124
|
+
function edgesOf(node) {
|
|
125
|
+
let left;
|
|
126
|
+
let right;
|
|
127
|
+
const visit = (n) => {
|
|
128
|
+
for (const { page, mcid } of n.mcids) for (const run of runsOfMcid(page, mcid)) {
|
|
129
|
+
if (left === void 0 || run.x < left) left = run.x;
|
|
130
|
+
if (right === void 0 || run.endX > right) right = run.endX;
|
|
105
131
|
}
|
|
132
|
+
for (const child of n.children) visit(child);
|
|
106
133
|
};
|
|
134
|
+
visit(node);
|
|
135
|
+
return left !== void 0 && right !== void 0 ? {
|
|
136
|
+
left,
|
|
137
|
+
right
|
|
138
|
+
} : void 0;
|
|
107
139
|
}
|
|
108
140
|
function buildRow(trNode) {
|
|
109
141
|
const cells = [];
|
|
@@ -114,20 +146,40 @@ function reconstructTaggedPdf(file) {
|
|
|
114
146
|
if (cell.type !== "TH") allHeader = false;
|
|
115
147
|
const content = [];
|
|
116
148
|
for (const child of cell.children) emit(child, content);
|
|
117
|
-
if (content.length === 0) content.push(
|
|
118
|
-
const
|
|
149
|
+
if (content.length === 0) content.push(paragraphFromRuns(spansOf(cell)));
|
|
150
|
+
const edges = edgesOf(cell);
|
|
119
151
|
cells.push({
|
|
120
|
-
|
|
121
|
-
|
|
152
|
+
content,
|
|
153
|
+
span: cell.colSpan ?? 1,
|
|
154
|
+
...edges ? {
|
|
155
|
+
x: edges.left,
|
|
156
|
+
right: edges.right
|
|
157
|
+
} : {}
|
|
122
158
|
});
|
|
123
159
|
}
|
|
124
160
|
return {
|
|
125
|
-
|
|
161
|
+
header: allHeader && cells.length > 0,
|
|
126
162
|
cells
|
|
127
163
|
};
|
|
128
164
|
}
|
|
129
165
|
const body = [];
|
|
130
166
|
emit(root, body);
|
|
167
|
+
let zOrder = -1e6;
|
|
168
|
+
pages.forEach((page, index) => {
|
|
169
|
+
const display = displayOf(page);
|
|
170
|
+
const frame = {
|
|
171
|
+
left: 0,
|
|
172
|
+
top: display.height
|
|
173
|
+
};
|
|
174
|
+
const lifted = collectPageVectors(file, page, (pageImages[index]?.images ?? []).map((img) => ({
|
|
175
|
+
minX: img.x,
|
|
176
|
+
minY: img.y,
|
|
177
|
+
maxX: img.x + img.widthPt,
|
|
178
|
+
maxY: img.y + img.heightPt
|
|
179
|
+
})));
|
|
180
|
+
imageLosses.push(...lifted.losses);
|
|
181
|
+
for (const v of placeVectors(lifted.vectors, display)) body.push(shapeBlock(v, frame, zOrder++));
|
|
182
|
+
});
|
|
131
183
|
const orphans = [];
|
|
132
184
|
pageImages.forEach((p, page) => {
|
|
133
185
|
for (const img of p.images) if (!emitted.has(img)) orphans.push({
|
|
@@ -139,10 +191,97 @@ function reconstructTaggedPdf(file) {
|
|
|
139
191
|
for (const { img } of orphans) body.push(imageBlock(img, resources));
|
|
140
192
|
if (body.length === 0) return void 0;
|
|
141
193
|
return {
|
|
142
|
-
doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages)),
|
|
194
|
+
doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages), collectEmbeddedFonts(file, pages)),
|
|
143
195
|
losses: imageLosses
|
|
144
196
|
};
|
|
145
197
|
}
|
|
198
|
+
/** Two starts within this many points are the same column. */
|
|
199
|
+
var COLUMN_TOLERANCE_PT = 2;
|
|
200
|
+
/**
|
|
201
|
+
* Lay the tagged cells onto a column grid read from the page.
|
|
202
|
+
*
|
|
203
|
+
* A structure tree states no column widths, and states `/ColSpan` only when its
|
|
204
|
+
* producer bothered: 160F-2019.pdf tags twenty-six columns and then writes rows
|
|
205
|
+
* of three plain `TD`s that visually run half the page. Believed literally,
|
|
206
|
+
* those three sat in columns 0–2 while twenty-three stood empty beside them,
|
|
207
|
+
* and 153 characters were asked to fit in 7.7pt — one page reconstructed as
|
|
208
|
+
* five, a word to a line.
|
|
209
|
+
*
|
|
210
|
+
* So the grid comes from where the cells actually START. Every distinct start
|
|
211
|
+
* across the table is a column boundary; a cell runs from its own boundary to
|
|
212
|
+
* the next cell's in its row, which is its span; and the width of a column is
|
|
213
|
+
* the distance to the next boundary. The tagged `/ColSpan` still sets a floor,
|
|
214
|
+
* so a producer that did the work is never contradicted.
|
|
215
|
+
*
|
|
216
|
+
* Falls back to the tagged spans over an equal grid when the page says too
|
|
217
|
+
* little — fewer than two distinct starts, or a table whose cells hold no text.
|
|
218
|
+
*
|
|
219
|
+
* @param raw The rows as tagged, each cell carrying its page x where it has one.
|
|
220
|
+
* @returns The rows with spans resolved, and one width per column.
|
|
221
|
+
*/
|
|
222
|
+
function layOutColumns(raw) {
|
|
223
|
+
const bounds = columnBounds(raw);
|
|
224
|
+
if (bounds.length < 2) return equalGrid(raw);
|
|
225
|
+
const indexOf = (x) => {
|
|
226
|
+
let best = 0;
|
|
227
|
+
for (let i = 0; i < bounds.length; i++) if (x >= bounds[i] - COLUMN_TOLERANCE_PT) best = i;
|
|
228
|
+
return best;
|
|
229
|
+
};
|
|
230
|
+
const rows = raw.map(({ header, cells }) => {
|
|
231
|
+
const out = [];
|
|
232
|
+
let col = 0;
|
|
233
|
+
cells.forEach((cell, i) => {
|
|
234
|
+
const start = cell.x !== void 0 ? Math.max(col, indexOf(cell.x)) : col;
|
|
235
|
+
const nextX = cells.slice(i + 1).find((c) => c.x !== void 0)?.x;
|
|
236
|
+
const end = nextX !== void 0 ? Math.max(start + 1, indexOf(nextX)) : bounds.length;
|
|
237
|
+
const span = Math.max(cell.span, end - start, 1);
|
|
238
|
+
out.push({
|
|
239
|
+
properties: span > 1 ? { colSpan: span } : {},
|
|
240
|
+
content: cell.content
|
|
241
|
+
});
|
|
242
|
+
col = start + span;
|
|
243
|
+
});
|
|
244
|
+
return {
|
|
245
|
+
properties: header ? { isHeader: true } : {},
|
|
246
|
+
cells: out
|
|
247
|
+
};
|
|
248
|
+
});
|
|
249
|
+
const widths = bounds.map((x, i) => i + 1 < bounds.length ? bounds[i + 1] - x : 0);
|
|
250
|
+
const tableRight = Math.max(...raw.flatMap((r) => r.cells.map((c) => c.right ?? 0)), bounds[bounds.length - 1]);
|
|
251
|
+
widths[widths.length - 1] = Math.max(0, tableRight - bounds[bounds.length - 1]);
|
|
252
|
+
const total = widths.reduce((a, b) => a + Math.max(MIN_COLUMN_PT, b), 0);
|
|
253
|
+
const scale = total > 0 ? ASSUMED_CONTENT_WIDTH_PT / total : 1;
|
|
254
|
+
return {
|
|
255
|
+
rows,
|
|
256
|
+
grid: widths.map((w) => pt(Math.max(1, Math.max(MIN_COLUMN_PT, w) * scale)))
|
|
257
|
+
};
|
|
258
|
+
}
|
|
259
|
+
/** Every distinct cell start across the table, ascending — the column boundaries. */
|
|
260
|
+
function columnBounds(raw) {
|
|
261
|
+
const xs = raw.flatMap((r) => r.cells.map((c) => c.x)).filter((x) => x !== void 0).sort((a, b) => a - b);
|
|
262
|
+
const bounds = [];
|
|
263
|
+
for (const x of xs) {
|
|
264
|
+
const last = bounds[bounds.length - 1];
|
|
265
|
+
if (last === void 0 || x - last > COLUMN_TOLERANCE_PT) bounds.push(x);
|
|
266
|
+
}
|
|
267
|
+
return bounds;
|
|
268
|
+
}
|
|
269
|
+
/** The old reading: the tagged spans, over columns of equal width. */
|
|
270
|
+
function equalGrid(raw) {
|
|
271
|
+
const rows = raw.map(({ header, cells }) => ({
|
|
272
|
+
properties: header ? { isHeader: true } : {},
|
|
273
|
+
cells: cells.map((c) => ({
|
|
274
|
+
properties: c.span > 1 ? { colSpan: c.span } : {},
|
|
275
|
+
content: c.content
|
|
276
|
+
}))
|
|
277
|
+
}));
|
|
278
|
+
const numCols = Math.max(1, ...rows.map((r) => r.cells.reduce((s, c) => s + (c.properties.colSpan ?? 1), 0)));
|
|
279
|
+
const w = pt(Math.max(1, ASSUMED_CONTENT_WIDTH_PT / numCols));
|
|
280
|
+
return {
|
|
281
|
+
rows,
|
|
282
|
+
grid: Array.from({ length: numCols }, () => w)
|
|
283
|
+
};
|
|
284
|
+
}
|
|
146
285
|
function squash(text) {
|
|
147
286
|
return text.replace(/\s+/g, " ").trim();
|
|
148
287
|
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { TextRun } from './content.js';
|
|
1
|
+
import { ContentFont, TextRun } from './content.js';
|
|
2
|
+
import { PdfDict } from '../pdf/objects.js';
|
|
2
3
|
import { PdfFile, PdfPage } from './document.js';
|
|
3
4
|
/**
|
|
4
5
|
* Extract a page's text as positioned {@link TextRun}s (E-PDF EP2/EP8/EP13).
|
|
@@ -12,3 +13,14 @@ import { PdfFile, PdfPage } from './document.js';
|
|
|
12
13
|
* @returns The page's runs, each carrying an `href` when it sits under a link.
|
|
13
14
|
*/
|
|
14
15
|
export declare function extractPageText(file: PdfFile, page: PdfPage): Array<TextRun>;
|
|
16
|
+
/**
|
|
17
|
+
* The `/Font` resources of one dictionary, built into interpreter fonts. Shared
|
|
18
|
+
* with the path and picture walks, which need them for one thing only: a Type 3
|
|
19
|
+
* font's glyphs are content streams (§9.6.5), and a walk that interprets with
|
|
20
|
+
* no fonts at all cannot see that there is anything to run.
|
|
21
|
+
*
|
|
22
|
+
* @param file The owning file.
|
|
23
|
+
* @param resources The resource dictionary to read `/Font` from.
|
|
24
|
+
* @returns Resource name → font, skipping any the reader cannot build.
|
|
25
|
+
*/
|
|
26
|
+
export declare function buildFonts(file: PdfFile, resources: PdfDict | undefined): Map<string, ContentFont>;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
3
|
+
import { collectPageAppearances } from "./annots.js";
|
|
3
4
|
import { buildContentFont } from "./font.js";
|
|
5
|
+
import { patternTint } from "./pattern-tint.js";
|
|
4
6
|
//#region src/pdf-reader/text.ts
|
|
5
7
|
var MAX_FORM_DEPTH = 8;
|
|
6
8
|
/**
|
|
@@ -17,6 +19,7 @@ var MAX_FORM_DEPTH = 8;
|
|
|
17
19
|
function extractPageText(file, page) {
|
|
18
20
|
const runs = [];
|
|
19
21
|
collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs);
|
|
22
|
+
for (const appearance of collectPageAppearances(file, page)) collectRuns(file, appearance.resources ?? page.resources, file.streamData(appearance.stream), appearance.ctm, 1, new Set([appearance.stream]), runs);
|
|
20
23
|
const links = collectLinks(file, page);
|
|
21
24
|
if (links.length === 0) return runs;
|
|
22
25
|
return runs.map((run) => {
|
|
@@ -29,8 +32,15 @@ function extractPageText(file, page) {
|
|
|
29
32
|
}
|
|
30
33
|
function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
|
|
31
34
|
const result = interpretContent(content, buildFonts(file, resources), baseCtm);
|
|
32
|
-
out.push(...result.texts);
|
|
33
|
-
if (depth >= MAX_FORM_DEPTH
|
|
35
|
+
out.push(...result.texts.map((r) => withPatternColour(file, resources, r, visiting)));
|
|
36
|
+
if (depth >= MAX_FORM_DEPTH) return;
|
|
37
|
+
for (const glyph of result.glyphs) {
|
|
38
|
+
if (visiting.has(glyph.stream)) continue;
|
|
39
|
+
visiting.add(glyph.stream);
|
|
40
|
+
collectRuns(file, glyph.resources ?? resources, file.streamData(glyph.stream), glyph.ctm, depth + 1, visiting, out);
|
|
41
|
+
visiting.delete(glyph.stream);
|
|
42
|
+
}
|
|
43
|
+
if (!resources) return;
|
|
34
44
|
const xobjects = file.get(resources, "XObject");
|
|
35
45
|
if (!(xobjects instanceof Map)) return;
|
|
36
46
|
for (const placement of result.images) {
|
|
@@ -44,6 +54,63 @@ function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
|
|
|
44
54
|
visiting.delete(stream);
|
|
45
55
|
}
|
|
46
56
|
}
|
|
57
|
+
/**
|
|
58
|
+
* §8.6.6.2 — a run whose glyphs the page fills with a tiling PATTERN, given the
|
|
59
|
+
* pattern's own colour.
|
|
60
|
+
*
|
|
61
|
+
* A pattern is a content stream, not a colour, and type cannot be filled with
|
|
62
|
+
* one here: what a run carries is a single colour. Painting it in whatever was
|
|
63
|
+
* set before the pattern was is simply wrong — it comes out black where the
|
|
64
|
+
* page shows magenta. The pattern's first mark says what colour the page meant,
|
|
65
|
+
* and the glyphs take that. A documented approximation: the shape of the
|
|
66
|
+
* pattern is lost, its colour is not.
|
|
67
|
+
*
|
|
68
|
+
* @param file The owning file.
|
|
69
|
+
* @param resources The resources the run was drawn with.
|
|
70
|
+
* @param run The run, which may name a fill pattern.
|
|
71
|
+
* @param visiting The streams already being interpreted, against a cycle.
|
|
72
|
+
* @returns The run, its colour taken from the pattern where there is one.
|
|
73
|
+
*/
|
|
74
|
+
function withPatternColour(file, resources, run, visiting) {
|
|
75
|
+
const name = run.fillPatternName;
|
|
76
|
+
if (name === void 0 || !resources) return run;
|
|
77
|
+
const patterns = file.get(resources, "Pattern");
|
|
78
|
+
if (!(patterns instanceof Map)) return run;
|
|
79
|
+
const stream = file.resolve(patterns.get(name) ?? PDF_NULL);
|
|
80
|
+
if (!(stream instanceof PdfStream) || visiting.has(stream)) return run;
|
|
81
|
+
visiting.add(stream);
|
|
82
|
+
try {
|
|
83
|
+
const tint = patternTint(file, resources, name);
|
|
84
|
+
return tint ? {
|
|
85
|
+
...run,
|
|
86
|
+
colorHex: tintedHex(tint.colorHex, tint.coverage)
|
|
87
|
+
} : run;
|
|
88
|
+
} catch {
|
|
89
|
+
return run;
|
|
90
|
+
} finally {
|
|
91
|
+
visiting.delete(stream);
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
/** A colour laid over white paper at `coverage` strength, as a 6-hex string. */
|
|
95
|
+
function tintedHex(colorHex, coverage) {
|
|
96
|
+
const k = Math.min(1, Math.max(0, coverage));
|
|
97
|
+
if (k >= 1) return colorHex;
|
|
98
|
+
const channel = (at) => {
|
|
99
|
+
const c = Number.parseInt(colorHex.slice(at, at + 2), 16);
|
|
100
|
+
return Math.round(255 - (255 - (Number.isFinite(c) ? c : 0)) * k).toString(16).toUpperCase().padStart(2, "0");
|
|
101
|
+
};
|
|
102
|
+
return `${channel(0)}${channel(2)}${channel(4)}`;
|
|
103
|
+
}
|
|
104
|
+
/**
|
|
105
|
+
* The `/Font` resources of one dictionary, built into interpreter fonts. Shared
|
|
106
|
+
* with the path and picture walks, which need them for one thing only: a Type 3
|
|
107
|
+
* font's glyphs are content streams (§9.6.5), and a walk that interprets with
|
|
108
|
+
* no fonts at all cannot see that there is anything to run.
|
|
109
|
+
*
|
|
110
|
+
* @param file The owning file.
|
|
111
|
+
* @param resources The resource dictionary to read `/Font` from.
|
|
112
|
+
* @returns Resource name → font, skipping any the reader cannot build.
|
|
113
|
+
*/
|
|
47
114
|
function buildFonts(file, resources) {
|
|
48
115
|
const fonts = /* @__PURE__ */ new Map();
|
|
49
116
|
if (!resources) return fonts;
|
|
@@ -107,4 +174,4 @@ function inRect(x, y, r) {
|
|
|
107
174
|
return x >= r[0] - 1 && x <= r[2] + 1 && y >= r[1] - 2 && y <= r[3] + 2;
|
|
108
175
|
}
|
|
109
176
|
//#endregion
|
|
110
|
-
export { extractPageText };
|
|
177
|
+
export { buildFonts, extractPageText };
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { PathSeg } from './content.js';
|
|
2
2
|
import { ShapeGradient } from '../core/vector.js';
|
|
3
|
+
import { Loss } from '../core/ir/index.js';
|
|
3
4
|
import { PdfFile, PdfPage } from './document.js';
|
|
4
5
|
/**
|
|
5
6
|
* One painted vector path lifted off a page (E-PDF EP10/EP11/EP16c): the path
|
|
@@ -8,11 +9,21 @@ import { PdfFile, PdfPage } from './document.js';
|
|
|
8
9
|
* marked-content id.
|
|
9
10
|
*/
|
|
10
11
|
export interface PdfVector {
|
|
12
|
+
/**
|
|
13
|
+
* §8.5.3 — where this was painted, as the chain of positions leading to it:
|
|
14
|
+
* `[4]` is the fifth mark of the page, `[4, 2]` the third mark of the form
|
|
15
|
+
* that mark called. Compared element by element it is the page's painting
|
|
16
|
+
* order across forms and patterns alike, which is the only thing that says
|
|
17
|
+
* what covers what.
|
|
18
|
+
*/
|
|
19
|
+
readonly orderKey: ReadonlyArray<number>;
|
|
11
20
|
readonly segs: ReadonlyArray<PathSeg>;
|
|
12
21
|
/** Present iff a qualifying solid fill survived (EP10). */
|
|
13
22
|
readonly fillHex?: string;
|
|
14
23
|
/** Present iff a shading-pattern fill survived (EP16c). */
|
|
15
24
|
readonly gradient?: ShapeGradient;
|
|
25
|
+
/** §11.6.4.4 — how opaque the fill is, when the page asked for less than all. */
|
|
26
|
+
readonly alpha?: number;
|
|
16
27
|
/** Present iff a qualifying stroke survived (EP11). */
|
|
17
28
|
readonly strokeHex?: string;
|
|
18
29
|
/** Stroke width in page-space points (EP11). */
|
|
@@ -32,4 +43,17 @@ export interface PdfVector {
|
|
|
32
43
|
* genuine coloured shapes, lines and gradients without the dot / page-background
|
|
33
44
|
* clutter. Clips and the bare `sh` operator are not captured (a documented loss).
|
|
34
45
|
*/
|
|
35
|
-
export declare function collectPageVectors(file: PdfFile, page: PdfPage):
|
|
46
|
+
export declare function collectPageVectors(file: PdfFile, page: PdfPage, occupied?: ReadonlyArray<Box>): PageVectors;
|
|
47
|
+
/** The paths lifted off one page, plus a loss if the page ran past the cap. */
|
|
48
|
+
export interface PageVectors {
|
|
49
|
+
readonly vectors: Array<PdfVector>;
|
|
50
|
+
readonly losses: ReadonlyArray<Loss>;
|
|
51
|
+
}
|
|
52
|
+
/** A page-space rectangle: what something covers. */
|
|
53
|
+
interface Box {
|
|
54
|
+
readonly minX: number;
|
|
55
|
+
readonly minY: number;
|
|
56
|
+
readonly maxX: number;
|
|
57
|
+
readonly maxY: number;
|
|
58
|
+
}
|
|
59
|
+
export {};
|
|
@@ -1,11 +1,111 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
3
|
+
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
4
|
+
import { collectPageAppearances } from "./annots.js";
|
|
5
|
+
import { buildFonts } from "./text.js";
|
|
6
|
+
import { buildAlphaMap, buildShadingMap } from "./shading.js";
|
|
3
7
|
//#region src/pdf-reader/vector.ts
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
8
|
+
/**
|
|
9
|
+
* Every vector the page paints, its FORM XOBJECTS included (§8.8).
|
|
10
|
+
*
|
|
11
|
+
* A `Do` of a form is a call: its content stream draws in the caller's space
|
|
12
|
+
* through the form's own `/Matrix`. Interpreting the page stream alone reads
|
|
13
|
+
* only what the page drew directly, and a document that puts its artwork in
|
|
14
|
+
* forms — as every CAD and drawing producer does — comes back with none of it.
|
|
15
|
+
* 22060_A1_01_Plans.pdf holds eleven, and its floor plans were simply absent.
|
|
16
|
+
*
|
|
17
|
+
* The same walk `collectPageImages` already makes, with the same depth and
|
|
18
|
+
* cycle guards, collecting paths instead of pictures.
|
|
19
|
+
*/
|
|
20
|
+
function paintedVectors(file, page, shadings, alphas) {
|
|
21
|
+
const out = [];
|
|
22
|
+
const visiting = /* @__PURE__ */ new Set();
|
|
23
|
+
const walk = (resources, content, baseCtm, depth, prefix) => {
|
|
24
|
+
if (out.length >= MAX_VECTORS) return;
|
|
25
|
+
const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
|
|
26
|
+
const xobjDict = xobjects instanceof Map ? xobjects : void 0;
|
|
27
|
+
const result = interpretContent(content, buildFonts(file, resources), baseCtm, shadings, alphas);
|
|
28
|
+
const events = [
|
|
29
|
+
...result.vectors.map((vector) => ({
|
|
30
|
+
order: vector.order,
|
|
31
|
+
vector
|
|
32
|
+
})),
|
|
33
|
+
...result.images.map((xobject) => ({
|
|
34
|
+
order: xobject.order,
|
|
35
|
+
xobject
|
|
36
|
+
})),
|
|
37
|
+
...result.glyphs.map((glyph) => ({
|
|
38
|
+
order: glyph.order,
|
|
39
|
+
glyph
|
|
40
|
+
}))
|
|
41
|
+
].sort((a, b) => a.order - b.order);
|
|
42
|
+
for (const event of events) {
|
|
43
|
+
if (out.length >= MAX_VECTORS) return;
|
|
44
|
+
if (event.vector) {
|
|
45
|
+
out.push({
|
|
46
|
+
...event.vector,
|
|
47
|
+
orderKey: [...prefix, event.order]
|
|
48
|
+
});
|
|
49
|
+
continue;
|
|
50
|
+
}
|
|
51
|
+
if (event.glyph) {
|
|
52
|
+
const call = event.glyph;
|
|
53
|
+
if (depth >= MAX_FORM_DEPTH || visiting.has(call.stream)) continue;
|
|
54
|
+
visiting.add(call.stream);
|
|
55
|
+
walk(call.resources ?? resources, file.streamData(call.stream), call.ctm, depth + 1, [...prefix, call.order]);
|
|
56
|
+
visiting.delete(call.stream);
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
const placement = event.xobject;
|
|
60
|
+
if (depth >= MAX_FORM_DEPTH) continue;
|
|
61
|
+
const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
|
|
62
|
+
if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
|
|
63
|
+
const subtype = file.get(stream.dict, "Subtype");
|
|
64
|
+
if (!(subtype instanceof PdfName) || subtype.value !== "Form") continue;
|
|
65
|
+
visiting.add(stream);
|
|
66
|
+
const formRes = file.get(stream.dict, "Resources");
|
|
67
|
+
walk(formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(formMatrix(file, stream.dict), placement.ctm), depth + 1, [...prefix, placement.order]);
|
|
68
|
+
visiting.delete(stream);
|
|
69
|
+
}
|
|
70
|
+
};
|
|
71
|
+
walk(page.resources, file.pageContent(page), IDENTITY, 0, []);
|
|
72
|
+
collectPageAppearances(file, page).forEach((appearance, index) => {
|
|
73
|
+
walk(appearance.resources ?? page.resources, file.streamData(appearance.stream), appearance.ctm, 1, [Number.MAX_SAFE_INTEGER, index]);
|
|
74
|
+
});
|
|
75
|
+
return out;
|
|
76
|
+
}
|
|
77
|
+
/** §8.10.2 `/Matrix` — the form's own space, composed onto the placement CTM. */
|
|
78
|
+
function formMatrix(file, dict) {
|
|
79
|
+
const m = file.resolve(dict.get("Matrix") ?? PDF_NULL);
|
|
80
|
+
if (!Array.isArray(m) || m.length !== 6) return IDENTITY;
|
|
81
|
+
const n = m.map((v) => typeof file.resolve(v) === "number" ? file.resolve(v) : 0);
|
|
82
|
+
return [
|
|
83
|
+
n[0],
|
|
84
|
+
n[1],
|
|
85
|
+
n[2],
|
|
86
|
+
n[3],
|
|
87
|
+
n[4],
|
|
88
|
+
n[5]
|
|
89
|
+
];
|
|
90
|
+
}
|
|
91
|
+
var MIN_SIDE = .5;
|
|
92
|
+
var MIN_AREA = 2;
|
|
7
93
|
var MIN_STROKE_LEN = 6;
|
|
8
|
-
var
|
|
94
|
+
var MIN_RULE_LEN = 2;
|
|
95
|
+
/**
|
|
96
|
+
* How many painted paths a page may hand over, as a guard against a file built
|
|
97
|
+
* to exhaust a reader.
|
|
98
|
+
*
|
|
99
|
+
* This was two thousand, and it was not a guard, it was a silent truncation:
|
|
100
|
+
* the busiest page of Brotli-Prototype-FileA.pdf keeps 1994 paths AFTER the
|
|
101
|
+
* de-cluttering filter, so the cap cut the page off partway and took its whole
|
|
102
|
+
* title block, the vegetation of its perspective and every hatch in its legend
|
|
103
|
+
* with it — with nothing in the report to say a thing was missing. Twenty
|
|
104
|
+
* thousand leaves room for a drawing and still bounds the work (that file's
|
|
105
|
+
* twenty-five sheets read in 420ms); reaching it is now reported.
|
|
106
|
+
*/
|
|
107
|
+
var MAX_VECTORS = 2e4;
|
|
108
|
+
var MAX_FORM_DEPTH = 12;
|
|
9
109
|
/**
|
|
10
110
|
* Lift the painted vector paths off a page (E-PDF EP10/EP11/EP16c). Runs the
|
|
11
111
|
* content interpreter for its fill (EP10), stroke (EP11) and shading-pattern
|
|
@@ -15,25 +115,38 @@ var MAX_VECTORS = 2e3;
|
|
|
15
115
|
* genuine coloured shapes, lines and gradients without the dot / page-background
|
|
16
116
|
* clutter. Clips and the bare `sh` operator are not captured (a documented loss).
|
|
17
117
|
*/
|
|
18
|
-
function collectPageVectors(file, page) {
|
|
118
|
+
function collectPageVectors(file, page, occupied = []) {
|
|
19
119
|
const [px0, py0, px1, py1] = page.mediaBox;
|
|
20
120
|
const pageArea = Math.max(1, Math.abs((px1 - px0) * (py1 - py0)));
|
|
21
121
|
const shadings = buildShadingMap(file, page);
|
|
122
|
+
const alphas = buildAlphaMap(file, page);
|
|
22
123
|
const out = [];
|
|
23
|
-
|
|
124
|
+
const painted = [...occupied];
|
|
125
|
+
const raws = paintedVectors(file, page, shadings, alphas);
|
|
126
|
+
for (const raw of raws) {
|
|
24
127
|
if (out.length >= MAX_VECTORS) break;
|
|
128
|
+
const v = clipped(raw);
|
|
129
|
+
if (!v) continue;
|
|
25
130
|
const b = bbox(v.segs);
|
|
26
131
|
if (!b) continue;
|
|
27
132
|
const w = b.maxX - b.minX;
|
|
28
133
|
const h = b.maxY - b.minY;
|
|
29
134
|
const area = w * h;
|
|
30
|
-
const
|
|
31
|
-
const
|
|
32
|
-
const
|
|
135
|
+
const white = v.fillHex === "FFFFFF";
|
|
136
|
+
const solidFill = v.patternName === void 0 && v.fillHex !== void 0 && (!white || painted.some((box) => overlaps(box, b)));
|
|
137
|
+
const long = Math.max(w, h);
|
|
138
|
+
const short = Math.min(w, h);
|
|
139
|
+
const isBox = short >= MIN_SIDE && area >= MIN_AREA;
|
|
140
|
+
const isRule = long >= MIN_RULE_LEN && short > 0;
|
|
141
|
+
const filled = (v.gradient !== void 0 || solidFill) && (isBox || isRule) && area <= .85 * pageArea;
|
|
142
|
+
const stroked = v.strokeHex !== void 0 && (v.strokeHex !== "FFFFFF" || painted.some((box) => overlaps(box, b))) && Math.max(w, h) >= MIN_STROKE_LEN && area <= .85 * pageArea;
|
|
33
143
|
if (!filled && !stroked) continue;
|
|
144
|
+
painted.push(b);
|
|
34
145
|
out.push({
|
|
146
|
+
orderKey: v.orderKey,
|
|
35
147
|
segs: v.segs,
|
|
36
148
|
...filled ? v.gradient ? { gradient: v.gradient } : v.fillHex !== void 0 ? { fillHex: v.fillHex } : {} : {},
|
|
149
|
+
...filled && v.alpha !== void 0 ? { alpha: v.alpha } : {},
|
|
37
150
|
...stroked ? {
|
|
38
151
|
strokeHex: v.strokeHex,
|
|
39
152
|
...v.lineWidth !== void 0 ? { lineWidth: v.lineWidth } : {}
|
|
@@ -42,7 +155,50 @@ function collectPageVectors(file, page) {
|
|
|
42
155
|
...v.mcid !== void 0 ? { mcid: v.mcid } : {}
|
|
43
156
|
});
|
|
44
157
|
}
|
|
45
|
-
return
|
|
158
|
+
return {
|
|
159
|
+
vectors: out,
|
|
160
|
+
losses: out.length >= MAX_VECTORS || raws.length >= MAX_VECTORS ? [{
|
|
161
|
+
severity: "dropped",
|
|
162
|
+
feature: FEATURES.shapes,
|
|
163
|
+
detail: `page carries more than ${String(MAX_VECTORS)} painted paths; the rest were not read`
|
|
164
|
+
}] : []
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
/** Whether two boxes share any area at all. */
|
|
168
|
+
function overlaps(a, b) {
|
|
169
|
+
return a.minX < b.maxX && b.minX < a.maxX && a.minY < b.maxY && b.minY < a.maxY;
|
|
170
|
+
}
|
|
171
|
+
/**
|
|
172
|
+
* §8.5.4 — what a painted path actually MARKS, once its clip is honoured.
|
|
173
|
+
*
|
|
174
|
+
* The mark is the intersection of the path with the clipping region, and one
|
|
175
|
+
* observation decides it without intersecting anything: where the path covers
|
|
176
|
+
* the clip, the intersection IS the clip. That is the stencil every drawing
|
|
177
|
+
* producer writes — 22060_A1_01_Plans.pdf paints four black rectangles of
|
|
178
|
+
* 397×421pt through clips cut to its floor plans, and read as rectangles they
|
|
179
|
+
* flooded two thirds of an A3 sheet in black.
|
|
180
|
+
*
|
|
181
|
+
* So the smaller region stands: a path inside its clip is itself, a path around
|
|
182
|
+
* its clip becomes the clip in the paint's own colour. Neither is a general
|
|
183
|
+
* path intersection, and where the two merely overlap the answer is the smaller
|
|
184
|
+
* of them — bounded, never larger than the truth.
|
|
185
|
+
*
|
|
186
|
+
* @param v The painted path as the interpreter saw it.
|
|
187
|
+
* @returns The path to draw, or `undefined` when the clip leaves nothing.
|
|
188
|
+
*/
|
|
189
|
+
function clipped(v) {
|
|
190
|
+
const clip = v.clip;
|
|
191
|
+
if (!clip) return v;
|
|
192
|
+
const b = bbox(v.segs);
|
|
193
|
+
if (!b) return v;
|
|
194
|
+
if (clip.minX >= b.maxX || clip.maxX <= b.minX || clip.minY >= b.maxY || clip.maxY <= b.minY) return;
|
|
195
|
+
const pathArea = Math.max(0, b.maxX - b.minX) * Math.max(0, b.maxY - b.minY);
|
|
196
|
+
if (Math.max(0, clip.maxX - clip.minX) * Math.max(0, clip.maxY - clip.minY) >= pathArea) return v;
|
|
197
|
+
return {
|
|
198
|
+
...v,
|
|
199
|
+
clip: void 0,
|
|
200
|
+
segs: clip.segs
|
|
201
|
+
};
|
|
46
202
|
}
|
|
47
203
|
function bbox(segs) {
|
|
48
204
|
let minX = Infinity;
|
|
@@ -665,6 +665,9 @@ function parseTxBody(txBody, ph, cascade, colors, resolveLink, styleColor, theme
|
|
|
665
665
|
const autofit = bodyPr ? poChildren(bodyPr).find((c) => poIs(c, "a:normAutofit")) : void 0;
|
|
666
666
|
const shrink = autofit !== void 0 && poIntAttr(autofit, "fontScale") === void 0;
|
|
667
667
|
const warp = bodyPr ? textWarp(bodyPr) : void 0;
|
|
668
|
+
const v = bodyPr ? poAttr(bodyPr, "vert") : void 0;
|
|
669
|
+
const vertical = v === "vert270" ? "vert270" : v !== void 0 && v !== "horz" ? "vert" : void 0;
|
|
670
|
+
const upright = bodyPr ? poAttr(bodyPr, "upright") === "1" : false;
|
|
668
671
|
const anchor = a === "ctr" || a === "b" || a === "t" ? a : ph && cascade ? cascade.anchorFor(ph) : void 0;
|
|
669
672
|
return {
|
|
670
673
|
content: fitted,
|
|
@@ -673,6 +676,8 @@ function parseTxBody(txBody, ph, cascade, colors, resolveLink, styleColor, theme
|
|
|
673
676
|
...rIns !== void 0 ? { insetRight: emuToPt(rIns) } : {},
|
|
674
677
|
...bIns !== void 0 ? { insetBottom: emuToPt(bIns) } : {},
|
|
675
678
|
...anchor ? { anchor } : {},
|
|
679
|
+
...vertical ? { vertical } : {},
|
|
680
|
+
...upright ? { upright } : {},
|
|
676
681
|
...shrink ? { shrinkToFit: true } : {},
|
|
677
682
|
...warp ? { warp } : {}
|
|
678
683
|
};
|
|
@@ -527,6 +527,16 @@ function tableXml(table, losses, state, scope) {
|
|
|
527
527
|
const rows = table.rows.map((row) => rowXml(row, losses, state, scope)).join("");
|
|
528
528
|
return `<w:tbl>${tblPrXml(table.properties)}<w:tblGrid>${grid}</w:tblGrid>${rows}</w:tbl>`;
|
|
529
529
|
}
|
|
530
|
+
/**
|
|
531
|
+
* §17.4.60 — `w:tblPr`. CT_Tbl declares it `minOccurs="1"`: a table states its
|
|
532
|
+
* properties even when it has none to state, and Word refuses to open a file
|
|
533
|
+
* whose `w:tbl` opens straight into its grid. Nothing else in the wild writes
|
|
534
|
+
* one without it — 5432 tables across the LibreOffice, Word and POI corpora,
|
|
535
|
+
* every one of them with a `w:tblPr` — and it went unnoticed here because a
|
|
536
|
+
* table read from a real document always carries SOMETHING (a width, a border,
|
|
537
|
+
* an alignment). One reconstructed from a PDF's glyph positions carries none of
|
|
538
|
+
* it, so `Ream.parse(pdf).convert('docx')` wrote a document Word turned away.
|
|
539
|
+
*/
|
|
530
540
|
function tblPrXml(p) {
|
|
531
541
|
const out = [];
|
|
532
542
|
if (p.widthType !== void 0) {
|
|
@@ -538,7 +548,7 @@ function tblPrXml(p) {
|
|
|
538
548
|
if (borders) out.push(borders);
|
|
539
549
|
const margins = cellMarginsXml("w:tblCellMar", p.defaultCellMargins);
|
|
540
550
|
if (margins) out.push(margins);
|
|
541
|
-
return
|
|
551
|
+
return `<w:tblPr>${out.join("")}</w:tblPr>`;
|
|
542
552
|
}
|
|
543
553
|
function rowXml(row, losses, state, scope) {
|
|
544
554
|
const trPr = [];
|