reamkit 1.26.0 → 1.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -4
- package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
- package/dist/esm/pdf-reader/annot-draw.js +374 -0
- package/dist/esm/pdf-reader/annots.d.ts +3 -1
- package/dist/esm/pdf-reader/annots.js +18 -4
- package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
- package/dist/esm/pdf-reader/ccitt.js +70 -2
- package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
- package/dist/esm/pdf-reader/cie-color.js +112 -0
- package/dist/esm/pdf-reader/content.d.ts +47 -3
- package/dist/esm/pdf-reader/content.js +108 -13
- package/dist/esm/pdf-reader/display.d.ts +1 -1
- package/dist/esm/pdf-reader/display.js +21 -6
- package/dist/esm/pdf-reader/document.d.ts +6 -0
- package/dist/esm/pdf-reader/document.js +29 -1
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
- package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
- package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
- package/dist/esm/pdf-reader/flow-build.js +102 -8
- package/dist/esm/pdf-reader/font.js +77 -16
- package/dist/esm/pdf-reader/function.d.ts +16 -0
- package/dist/esm/pdf-reader/function.js +414 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
- package/dist/esm/pdf-reader/image-decode.js +192 -11
- package/dist/esm/pdf-reader/images.d.ts +12 -0
- package/dist/esm/pdf-reader/images.js +109 -9
- package/dist/esm/pdf-reader/jbig2.d.ts +23 -0
- package/dist/esm/pdf-reader/jbig2.js +23 -14
- package/dist/esm/pdf-reader/layout.d.ts +32 -0
- package/dist/esm/pdf-reader/layout.js +189 -40
- package/dist/esm/pdf-reader/lexer.d.ts +2 -0
- package/dist/esm/pdf-reader/lexer.js +4 -0
- package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
- package/dist/esm/pdf-reader/optional-content.js +93 -0
- package/dist/esm/pdf-reader/reader.d.ts +5 -2
- package/dist/esm/pdf-reader/reader.js +80 -7
- package/dist/esm/pdf-reader/shading.d.ts +60 -8
- package/dist/esm/pdf-reader/shading.js +124 -20
- package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
- package/dist/esm/pdf-reader/standard-metrics.js +18 -0
- package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
- package/dist/esm/pdf-reader/standard-widths.js +62 -0
- package/dist/esm/pdf-reader/tagged.js +204 -32
- package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
- package/dist/esm/pdf-reader/text-rules.js +112 -0
- package/dist/esm/pdf-reader/text.js +78 -4
- package/dist/esm/pdf-reader/vector.d.ts +5 -0
- package/dist/esm/pdf-reader/vector.js +39 -26
- package/dist/esm/word/docx-writer.js +73 -11
- package/package.json +1 -1
|
@@ -1,11 +1,14 @@
|
|
|
1
1
|
import { pt } from "../core/ir/units.js";
|
|
2
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
|
-
import {
|
|
4
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
|
|
3
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
5
4
|
import { collectEmbeddedFonts } from "./embedded-fonts.js";
|
|
6
|
-
import { collectPageImages } from "./images.js";
|
|
7
5
|
import { extractPageText } from "./text.js";
|
|
8
6
|
import { collectPageVectors } from "./vector.js";
|
|
7
|
+
import { displayOf, placeRuns, placeVectors } from "./display.js";
|
|
8
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, withMeasuredMargins } from "./flow-build.js";
|
|
9
|
+
import { collectPageImages } from "./images.js";
|
|
10
|
+
import { markDrawnRules } from "./text-rules.js";
|
|
11
|
+
import { endedParagraph } from "./layout.js";
|
|
9
12
|
import { readStructTree } from "./struct-tree.js";
|
|
10
13
|
//#region src/pdf-reader/tagged.ts
|
|
11
14
|
var ASSUMED_CONTENT_WIDTH_PT = 468;
|
|
@@ -27,9 +30,23 @@ function reconstructTaggedPdf(file) {
|
|
|
27
30
|
const root = readStructTree(file);
|
|
28
31
|
if (!root) return void 0;
|
|
29
32
|
const pages = file.pages();
|
|
30
|
-
const
|
|
33
|
+
const shown = pages.map((page) => displayOf(page));
|
|
34
|
+
const placedRuns = pages.map((page, i) => placeRuns(extractPageText(file, page), shown[i]));
|
|
35
|
+
const vectorLosses = [];
|
|
36
|
+
const pageVectors = pages.map((page, i) => {
|
|
37
|
+
const lifted = collectPageVectors(file, page, collectPageImages(file, page).images.map((img) => ({
|
|
38
|
+
minX: img.x,
|
|
39
|
+
minY: img.y,
|
|
40
|
+
maxX: img.x + img.widthPt,
|
|
41
|
+
maxY: img.y + img.heightPt
|
|
42
|
+
})));
|
|
43
|
+
vectorLosses.push(...lifted.losses);
|
|
44
|
+
return placeVectors(lifted.vectors, shown[i]);
|
|
45
|
+
});
|
|
46
|
+
const ruled = placedRuns.map((runs, i) => markDrawnRules(runs, pageVectors[i] ?? []));
|
|
47
|
+
const pageRuns = ruled.map(({ runs }) => {
|
|
31
48
|
const byMcid = /* @__PURE__ */ new Map();
|
|
32
|
-
for (const run of
|
|
49
|
+
for (const run of runs) {
|
|
33
50
|
if (run.mcid === void 0) continue;
|
|
34
51
|
const list = byMcid.get(run.mcid);
|
|
35
52
|
if (list) list.push(run);
|
|
@@ -37,7 +54,12 @@ function reconstructTaggedPdf(file) {
|
|
|
37
54
|
}
|
|
38
55
|
return byMcid;
|
|
39
56
|
});
|
|
40
|
-
const
|
|
57
|
+
const claimed = /* @__PURE__ */ new Set();
|
|
58
|
+
const runsOfMcid = (page, mcid) => {
|
|
59
|
+
const runs = pageRuns[page]?.get(mcid) ?? [];
|
|
60
|
+
for (const run of runs) claimed.add(run);
|
|
61
|
+
return runs;
|
|
62
|
+
};
|
|
41
63
|
const resources = new ResourceStore();
|
|
42
64
|
const pageImages = pages.map((page) => collectPageImages(file, page));
|
|
43
65
|
const imageLosses = dedupeLosses(pageImages.flatMap((p) => p.losses));
|
|
@@ -58,23 +80,123 @@ function reconstructTaggedPdf(file) {
|
|
|
58
80
|
const spans = [];
|
|
59
81
|
node.mcids.forEach(({ page, mcid }, i) => {
|
|
60
82
|
if (i > 0) spans.push({ text: " " });
|
|
61
|
-
for (const run of runsOfMcid(page, mcid)) spans.push(
|
|
62
|
-
text: run.text,
|
|
63
|
-
sizePt: run.fontSizePt,
|
|
64
|
-
...run.colorHex !== "000000" ? { colorHex: run.colorHex } : {},
|
|
65
|
-
...run.fontName !== void 0 ? { fontName: run.fontName } : {},
|
|
66
|
-
...run.outlineHex !== void 0 ? { outline: {
|
|
67
|
-
colorHex: run.outlineHex,
|
|
68
|
-
widthPt: pt(run.outlineWidthPt ?? 1)
|
|
69
|
-
} } : {},
|
|
70
|
-
...run.bold ? { bold: true } : {},
|
|
71
|
-
...run.italic ? { italic: true } : {},
|
|
72
|
-
...run.href !== void 0 ? { href: run.href } : {}
|
|
73
|
-
});
|
|
83
|
+
for (const run of runsOfMcid(page, mcid)) spans.push(spanOf(run));
|
|
74
84
|
});
|
|
75
85
|
return spans;
|
|
76
86
|
};
|
|
87
|
+
/** One run as the span that carries everything the page showed it with. */
|
|
88
|
+
const spanOf = (run) => ({
|
|
89
|
+
text: run.text.replaceAll("�", ""),
|
|
90
|
+
sizePt: run.fontSizePt,
|
|
91
|
+
...run.colorHex !== "000000" ? { colorHex: run.colorHex } : {},
|
|
92
|
+
...run.fontName !== void 0 ? { fontName: run.fontName } : {},
|
|
93
|
+
...run.outlineHex !== void 0 ? { outline: {
|
|
94
|
+
colorHex: run.outlineHex,
|
|
95
|
+
widthPt: pt(run.outlineWidthPt ?? 1)
|
|
96
|
+
} } : {},
|
|
97
|
+
...run.bold ? { bold: true } : {},
|
|
98
|
+
...run.italic ? { italic: true } : {},
|
|
99
|
+
...run.markup !== void 0 ? { markup: run.markup } : {},
|
|
100
|
+
...run.href !== void 0 ? { href: run.href } : {}
|
|
101
|
+
});
|
|
77
102
|
const collectText = (node) => squash([textOf(node), ...node.children.map(collectText)].join(" "));
|
|
103
|
+
const setting = /* @__PURE__ */ new Map();
|
|
104
|
+
/** The topmost and bottommost baseline under a node, and its largest face. */
|
|
105
|
+
function baselinesOf(node) {
|
|
106
|
+
let top;
|
|
107
|
+
let bottom;
|
|
108
|
+
let size = 0;
|
|
109
|
+
const visit = (n) => {
|
|
110
|
+
for (const { page, mcid } of n.mcids) for (const run of runsOfMcid(page, mcid)) {
|
|
111
|
+
if (top === void 0 || run.y > top) top = run.y;
|
|
112
|
+
if (bottom === void 0 || run.y < bottom) bottom = run.y;
|
|
113
|
+
if (run.fontSizePt > size) size = run.fontSizePt;
|
|
114
|
+
}
|
|
115
|
+
for (const child of n.children) visit(child);
|
|
116
|
+
};
|
|
117
|
+
visit(node);
|
|
118
|
+
return top !== void 0 && bottom !== void 0 && size > 0 ? {
|
|
119
|
+
top,
|
|
120
|
+
bottom,
|
|
121
|
+
size
|
|
122
|
+
} : void 0;
|
|
123
|
+
}
|
|
124
|
+
/**
|
|
125
|
+
* The element's own lines, gathered into the paragraphs they were SET as.
|
|
126
|
+
*
|
|
127
|
+
* A tree names elements, not lines, and a producer may put a whole page under
|
|
128
|
+
* one: annotation-underline.pdf marks every glyph on its page with the same
|
|
129
|
+
* id, so a heading, a blank line and a body line came back as one paragraph
|
|
130
|
+
* and re-wrapped into a single run-together line. The page still says where
|
|
131
|
+
* one ended — a line stopping well short of the measure stopped because its
|
|
132
|
+
* author stopped it — and that is the rule the heuristic reading already uses
|
|
133
|
+
* ({@link endedParagraph}). A properly tagged paragraph's inner lines all
|
|
134
|
+
* reach the measure, so nothing there changes.
|
|
135
|
+
*/
|
|
136
|
+
function settingsOf(node) {
|
|
137
|
+
const spans = spansOf(node);
|
|
138
|
+
const lines = linesOf(node);
|
|
139
|
+
if (lines.length < 2) {
|
|
140
|
+
const set = baselinesOf(node);
|
|
141
|
+
return set ? [{
|
|
142
|
+
spans,
|
|
143
|
+
set
|
|
144
|
+
}] : [];
|
|
145
|
+
}
|
|
146
|
+
const measure = {
|
|
147
|
+
left: Math.min(...lines.map((l) => l.x)),
|
|
148
|
+
right: Math.max(...lines.map((l) => l.x + l.width))
|
|
149
|
+
};
|
|
150
|
+
const groups = [];
|
|
151
|
+
let prev;
|
|
152
|
+
for (const line of lines) {
|
|
153
|
+
const gap = prev === void 0 ? 0 : prev.y - line.y;
|
|
154
|
+
const opened = prev !== void 0 && gap > line.fontSize * 1.5;
|
|
155
|
+
if (groups.length === 0 || opened || prev && endedParagraph(prev, line, measure)) groups.push([]);
|
|
156
|
+
groups[groups.length - 1].push(line);
|
|
157
|
+
prev = line;
|
|
158
|
+
}
|
|
159
|
+
return groups.map((g) => ({
|
|
160
|
+
spans: g.flatMap((l, i) => i > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
|
|
161
|
+
set: {
|
|
162
|
+
top: g[0].y,
|
|
163
|
+
bottom: g[g.length - 1].y,
|
|
164
|
+
size: Math.max(...g.map((l) => l.fontSize))
|
|
165
|
+
}
|
|
166
|
+
}));
|
|
167
|
+
}
|
|
168
|
+
/** The node's runs clustered onto the baselines they were shown on. */
|
|
169
|
+
function linesOf(node) {
|
|
170
|
+
const rows = [];
|
|
171
|
+
const visit = (n) => {
|
|
172
|
+
for (const { page, mcid } of n.mcids) for (const run of runsOfMcid(page, mcid)) {
|
|
173
|
+
if (run.angleDeg !== void 0) return;
|
|
174
|
+
const row = rows.find((r) => Math.abs(r.y - run.y) <= Math.max(r.size, run.fontSizePt) * .5);
|
|
175
|
+
if (row) {
|
|
176
|
+
row.runs.push(run);
|
|
177
|
+
row.size = Math.max(row.size, run.fontSizePt);
|
|
178
|
+
row.y = Math.min(row.y, run.y);
|
|
179
|
+
} else rows.push({
|
|
180
|
+
y: run.y,
|
|
181
|
+
size: run.fontSizePt,
|
|
182
|
+
runs: [run]
|
|
183
|
+
});
|
|
184
|
+
}
|
|
185
|
+
for (const child of n.children) visit(child);
|
|
186
|
+
};
|
|
187
|
+
visit(node);
|
|
188
|
+
return rows.sort((a, b) => b.y - a.y).map(({ y, runs }) => {
|
|
189
|
+
const ordered = [...runs].sort((a, b) => a.x - b.x);
|
|
190
|
+
const x = Math.min(...ordered.map((r) => r.x));
|
|
191
|
+
return {
|
|
192
|
+
y,
|
|
193
|
+
x,
|
|
194
|
+
width: Math.max(...ordered.map((r) => r.endX)) - x,
|
|
195
|
+
fontSize: Math.max(...ordered.map((r) => r.fontSizePt)),
|
|
196
|
+
spans: ordered.map((r) => spanOf(r))
|
|
197
|
+
};
|
|
198
|
+
}).filter((l) => l.spans.some((sp) => sp.text.trim().length > 0));
|
|
199
|
+
}
|
|
78
200
|
function emit(node, out) {
|
|
79
201
|
if (node.type === "Table") {
|
|
80
202
|
const table = buildTable(node);
|
|
@@ -94,7 +216,11 @@ function reconstructTaggedPdf(file) {
|
|
|
94
216
|
return;
|
|
95
217
|
}
|
|
96
218
|
if (node.children.length === 0) {
|
|
97
|
-
if (textOf(node).length > 0)
|
|
219
|
+
if (textOf(node).length > 0) for (const part of settingsOf(node)) {
|
|
220
|
+
const el = paragraphFromRuns(part.spans, headingLevel(node.type));
|
|
221
|
+
setting.set(el, part.set);
|
|
222
|
+
out.push(el);
|
|
223
|
+
}
|
|
98
224
|
return;
|
|
99
225
|
}
|
|
100
226
|
for (const child of node.children) emit(child, out);
|
|
@@ -164,21 +290,19 @@ function reconstructTaggedPdf(file) {
|
|
|
164
290
|
}
|
|
165
291
|
const body = [];
|
|
166
292
|
emit(root, body);
|
|
293
|
+
spaceParagraphs(body, setting, shown[0]?.height ?? 0);
|
|
167
294
|
let zOrder = -1e6;
|
|
168
|
-
|
|
169
|
-
|
|
295
|
+
imageLosses.push(...vectorLosses);
|
|
296
|
+
pages.forEach((_page, index) => {
|
|
170
297
|
const frame = {
|
|
171
298
|
left: 0,
|
|
172
|
-
top:
|
|
299
|
+
top: shown[index].height
|
|
173
300
|
};
|
|
174
|
-
const
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
})));
|
|
180
|
-
imageLosses.push(...lifted.losses);
|
|
181
|
-
for (const v of placeVectors(lifted.vectors, display)) body.push(shapeBlock(v, frame, zOrder++));
|
|
301
|
+
const taken = ruled[index]?.consumed;
|
|
302
|
+
for (const v of pageVectors[index] ?? []) {
|
|
303
|
+
if (taken?.has(v) === true) continue;
|
|
304
|
+
body.push(shapeBlock(v, frame, zOrder++, true));
|
|
305
|
+
}
|
|
182
306
|
});
|
|
183
307
|
const orphans = [];
|
|
184
308
|
pageImages.forEach((p, page) => {
|
|
@@ -190,8 +314,16 @@ function reconstructTaggedPdf(file) {
|
|
|
190
314
|
orphans.sort((a, b) => a.page - b.page || b.img.y - a.img.y);
|
|
191
315
|
for (const { img } of orphans) body.push(imageBlock(img, resources));
|
|
192
316
|
if (body.length === 0) return void 0;
|
|
317
|
+
if (placedRuns.some((page) => page.some((r) => r.text.includes("�")))) imageLosses.push({
|
|
318
|
+
severity: "dropped",
|
|
319
|
+
feature: FEATURES.text,
|
|
320
|
+
detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
|
|
321
|
+
});
|
|
322
|
+
const onPage = placedRuns.flat().reduce((n, r) => n + r.text.length, 0);
|
|
323
|
+
const reached = [...claimed].reduce((n, r) => n + r.text.length, 0);
|
|
324
|
+
if (onPage > 0 && reached * 2 < onPage) return void 0;
|
|
193
325
|
return {
|
|
194
|
-
doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages), collectEmbeddedFonts(file, pages)),
|
|
326
|
+
doc: buildFlowDoc(body, resources, withMeasuredMargins(sectionFromPdfPages(pages), shown, placedRuns), collectEmbeddedFonts(file, pages, imageLosses)),
|
|
195
327
|
losses: imageLosses
|
|
196
328
|
};
|
|
197
329
|
}
|
|
@@ -289,5 +421,45 @@ function headingLevel(type) {
|
|
|
289
421
|
const m = /^H([1-6])$/.exec(type);
|
|
290
422
|
return m ? Number(m[1]) - 1 : void 0;
|
|
291
423
|
}
|
|
424
|
+
/**
|
|
425
|
+
* §17.3.1.33 `w:spacing` — the space the SOURCE left before each paragraph.
|
|
426
|
+
*
|
|
427
|
+
* A structure tree names the words and says nothing about how far apart they
|
|
428
|
+
* stood, so every tagged PDF came back at one flat leading: on
|
|
429
|
+
* annotation-polyline-polygon-without-appearance.pdf the two labels, set a
|
|
430
|
+
* third of a page apart above their own drawings, arrived as two lines
|
|
431
|
+
* touching. The page still says it — the gap between the last baseline of one
|
|
432
|
+
* paragraph and the first of the next, less the line it would have taken
|
|
433
|
+
* anyway. This is the rule the heuristic reading already uses, applied to the
|
|
434
|
+
* paragraphs the tree named.
|
|
435
|
+
*
|
|
436
|
+
* @param body The body elements, in order; amended in place.
|
|
437
|
+
* @param setting Where each paragraph was set, for those that were.
|
|
438
|
+
* @param pageHeight The shown page's height, which bounds any one gap.
|
|
439
|
+
*/
|
|
440
|
+
function spaceParagraphs(body, setting, pageHeight) {
|
|
441
|
+
let prev;
|
|
442
|
+
body.forEach((el, i) => {
|
|
443
|
+
const here = setting.get(el);
|
|
444
|
+
if (!here) return;
|
|
445
|
+
const gap = prev !== void 0 ? prev.bottom - here.top : 0;
|
|
446
|
+
prev = here;
|
|
447
|
+
if (!(gap > 0)) return;
|
|
448
|
+
const opened = gap - here.size * 1.2;
|
|
449
|
+
if (!(opened > here.size * .3)) return;
|
|
450
|
+
if (el.kind !== "paragraph") return;
|
|
451
|
+
const most = pageHeight > 0 ? pageHeight / 3 : here.size * 3;
|
|
452
|
+
body[i] = {
|
|
453
|
+
...el,
|
|
454
|
+
paragraph: {
|
|
455
|
+
...el.paragraph,
|
|
456
|
+
properties: {
|
|
457
|
+
...el.paragraph.properties,
|
|
458
|
+
spacingBefore: pt(Math.min(opened, most))
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
};
|
|
462
|
+
});
|
|
463
|
+
}
|
|
292
464
|
//#endregion
|
|
293
465
|
export { reconstructTaggedPdf };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { TextRun } from './content.js';
|
|
2
|
+
import { PdfVector } from './vector.js';
|
|
3
|
+
/** The runs with their drawn rules read onto them, and the rules so read. */
|
|
4
|
+
export interface DrawnRules {
|
|
5
|
+
readonly runs: Array<TextRun>;
|
|
6
|
+
/** The vectors the runs took over: painting them again would double them. */
|
|
7
|
+
readonly consumed: ReadonlySet<PdfVector>;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Read the underlines and strikeouts a page DREW onto the runs they mark.
|
|
11
|
+
*
|
|
12
|
+
* @param runs The page's runs, placed on the shown page.
|
|
13
|
+
* @param vectors The page's lifted paths, placed the same way.
|
|
14
|
+
* @returns The runs, marked; and the paths that became those marks.
|
|
15
|
+
*/
|
|
16
|
+
export declare function markDrawnRules(runs: ReadonlyArray<TextRun>, vectors: ReadonlyArray<PdfVector>): DrawnRules;
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
//#region src/pdf-reader/text-rules.ts
|
|
2
|
+
/** A rule no thicker than this is a line under words, not a box. */
|
|
3
|
+
var THICKEST_PT = 2;
|
|
4
|
+
/** …and no thinner than this: below it the page drew a seam, not a mark. */
|
|
5
|
+
var THINNEST_PT = .2;
|
|
6
|
+
/** …or than this fraction of the face it underlines, for a large one. */
|
|
7
|
+
var THICKEST_EM = .09;
|
|
8
|
+
/** How much of a run's advance a rule must cover to be that run's own. */
|
|
9
|
+
var COVERS = .6;
|
|
10
|
+
/**
|
|
11
|
+
* How far past the words it may run, in points, before it is a rule about
|
|
12
|
+
* something else.
|
|
13
|
+
*
|
|
14
|
+
* An underline begins and ends with the words it underlines; a table's cell
|
|
15
|
+
* border begins at the CELL, which is the text's own edge less the padding, and
|
|
16
|
+
* ends at the cell's other edge however short the text stops. TAMReview.pdf's
|
|
17
|
+
* tables are ruled 86.2 to 509.0 across a measure whose text stops well before
|
|
18
|
+
* it, and a proportional allowance let every one of them through.
|
|
19
|
+
*/
|
|
20
|
+
var OVERHANG_PT = 2;
|
|
21
|
+
/** An underline sits within this fraction of the size below the baseline. */
|
|
22
|
+
var UNDER_EM = .35;
|
|
23
|
+
/** A strikeout crosses between these fractions of the size ABOVE it. */
|
|
24
|
+
var STRIKE_LOW = .15;
|
|
25
|
+
var STRIKE_HIGH = .5;
|
|
26
|
+
/**
|
|
27
|
+
* Read the underlines and strikeouts a page DREW onto the runs they mark.
|
|
28
|
+
*
|
|
29
|
+
* @param runs The page's runs, placed on the shown page.
|
|
30
|
+
* @param vectors The page's lifted paths, placed the same way.
|
|
31
|
+
* @returns The runs, marked; and the paths that became those marks.
|
|
32
|
+
*/
|
|
33
|
+
function markDrawnRules(runs, vectors) {
|
|
34
|
+
const marks = /* @__PURE__ */ new Map();
|
|
35
|
+
const consumed = /* @__PURE__ */ new Set();
|
|
36
|
+
for (const v of vectors) {
|
|
37
|
+
if (!isRule(v)) continue;
|
|
38
|
+
const under = coveredRuns(runs, v, "under");
|
|
39
|
+
const through = under.length > 0 ? [] : coveredRuns(runs, v, "through");
|
|
40
|
+
const hit = under.length > 0 ? under : through;
|
|
41
|
+
if (hit.length === 0) continue;
|
|
42
|
+
const left = Math.min(...hit.map((r) => Math.min(r.x, r.endX)));
|
|
43
|
+
const right = Math.max(...hit.map((r) => Math.max(r.x, r.endX)));
|
|
44
|
+
if (!(right - left > 0)) continue;
|
|
45
|
+
if (left - v.minX > OVERHANG_PT || v.maxX - right > OVERHANG_PT) continue;
|
|
46
|
+
for (const run of hit) {
|
|
47
|
+
const had = marks.get(run) ?? {};
|
|
48
|
+
marks.set(run, under.length > 0 && v.fillHex !== void 0 ? {
|
|
49
|
+
...had,
|
|
50
|
+
underline: v.fillHex
|
|
51
|
+
} : {
|
|
52
|
+
...had,
|
|
53
|
+
strike: true
|
|
54
|
+
});
|
|
55
|
+
}
|
|
56
|
+
consumed.add(v);
|
|
57
|
+
}
|
|
58
|
+
if (marks.size === 0) return {
|
|
59
|
+
runs: [...runs],
|
|
60
|
+
consumed
|
|
61
|
+
};
|
|
62
|
+
return {
|
|
63
|
+
runs: runs.map((run) => {
|
|
64
|
+
const mark = marks.get(run);
|
|
65
|
+
if (!mark) return run;
|
|
66
|
+
return {
|
|
67
|
+
...run,
|
|
68
|
+
markup: {
|
|
69
|
+
...run.markup,
|
|
70
|
+
...mark.underline !== void 0 && run.markup?.underline === void 0 ? {
|
|
71
|
+
underline: "single",
|
|
72
|
+
underlineHex: mark.underline
|
|
73
|
+
} : {},
|
|
74
|
+
...mark.strike === true ? { strike: true } : {}
|
|
75
|
+
}
|
|
76
|
+
};
|
|
77
|
+
}),
|
|
78
|
+
consumed
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
/** A thin filled bar, which is the only shape an underline is drawn as. */
|
|
82
|
+
function isRule(v) {
|
|
83
|
+
if (v.fillHex === void 0 || v.strokeHex !== void 0 || v.gradient !== void 0) return false;
|
|
84
|
+
if (v.fillHex === "FFFFFF") return false;
|
|
85
|
+
const h = v.maxY - v.minY;
|
|
86
|
+
const w = v.maxX - v.minX;
|
|
87
|
+
return h >= THINNEST_PT && h <= THICKEST_PT && w > 2;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* The runs a rule marks: on one baseline, covered along their advance, and at
|
|
91
|
+
* the offset from that baseline the mark is drawn at.
|
|
92
|
+
*/
|
|
93
|
+
function coveredRuns(runs, v, where) {
|
|
94
|
+
const out = [];
|
|
95
|
+
const mid = (v.minY + v.maxY) / 2;
|
|
96
|
+
for (const run of runs) {
|
|
97
|
+
const size = run.fontSizePt;
|
|
98
|
+
if (!(size > 0) || run.angleDeg !== void 0) continue;
|
|
99
|
+
if (v.maxY - v.minY > Math.max(THICKEST_PT, size * THICKEST_EM)) continue;
|
|
100
|
+
const below = run.y - mid;
|
|
101
|
+
if (!(where === "under" ? below > 0 && below < size * UNDER_EM : below < -size * STRIKE_LOW && below > -size * STRIKE_HIGH)) continue;
|
|
102
|
+
const left = Math.min(run.x, run.endX);
|
|
103
|
+
const right = Math.max(run.x, run.endX);
|
|
104
|
+
const advance = right - left;
|
|
105
|
+
if (!(advance > 0)) continue;
|
|
106
|
+
if (Math.min(right, v.maxX) - Math.max(left, v.minX) < advance * COVERS) continue;
|
|
107
|
+
out.push(run);
|
|
108
|
+
}
|
|
109
|
+
return out;
|
|
110
|
+
}
|
|
111
|
+
//#endregion
|
|
112
|
+
export { markDrawnRules };
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
3
|
+
import { textMarkupOf } from "./annot-draw.js";
|
|
3
4
|
import { collectPageAppearances } from "./annots.js";
|
|
5
|
+
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
4
6
|
import { buildContentFont } from "./font.js";
|
|
5
7
|
import { patternTint } from "./pattern-tint.js";
|
|
6
8
|
//#region src/pdf-reader/text.ts
|
|
@@ -21,17 +23,88 @@ function extractPageText(file, page) {
|
|
|
21
23
|
collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs);
|
|
22
24
|
for (const appearance of collectPageAppearances(file, page)) collectRuns(file, appearance.resources ?? page.resources, file.streamData(appearance.stream), appearance.ctm, 1, new Set([appearance.stream]), runs);
|
|
23
25
|
const links = collectLinks(file, page);
|
|
24
|
-
|
|
25
|
-
|
|
26
|
+
const marks = collectTextMarkup(file, page);
|
|
27
|
+
if (links.length === 0 && marks.length === 0) return runs;
|
|
28
|
+
return runs.flatMap((run) => {
|
|
26
29
|
const link = links.find((l) => inRect(run.x, run.y, l.rect));
|
|
27
|
-
|
|
30
|
+
const linked = link ? {
|
|
28
31
|
...run,
|
|
29
32
|
href: link.href
|
|
30
33
|
} : run;
|
|
34
|
+
const marked = marks.find((m) => m.quads.some((q) => touches(q, linked)));
|
|
35
|
+
if (!marked) return [linked];
|
|
36
|
+
const quad = marked.quads.find((q) => touches(q, linked));
|
|
37
|
+
return quad ? markedPieces(linked, quad, marked.mark) : [linked];
|
|
31
38
|
});
|
|
32
39
|
}
|
|
40
|
+
/**
|
|
41
|
+
* §12.5.6.10 — whether a marked quad reaches this run at all.
|
|
42
|
+
*
|
|
43
|
+
* The quad is a box round a run of text and the run is a baseline with an
|
|
44
|
+
* advance, so the test is the baseline falling inside the box's height while
|
|
45
|
+
* the two overlap horizontally by more than a hair.
|
|
46
|
+
*/
|
|
47
|
+
function touches(q, run) {
|
|
48
|
+
if (run.y < q.y0 || run.y > q.y0 + q.h) return false;
|
|
49
|
+
const left = Math.min(run.x, run.endX);
|
|
50
|
+
const right = Math.max(run.x, run.endX);
|
|
51
|
+
return Math.min(right, q.x1) - Math.max(left, q.x0) > .5;
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* The run cut at the quad's edges, so only what the quad covers is marked.
|
|
55
|
+
*
|
|
56
|
+
* A quad round ONE WORD of a line must not claim the line: highlight_popup.pdf
|
|
57
|
+
* marks "PDF.js" in "Hello PDF.js World", which is one run, and the whole line
|
|
58
|
+
* came back highlighted. The run states its advance and its text and not where
|
|
59
|
+
* each glyph fell inside it, so the cut is made by proportion of the advance —
|
|
60
|
+
* a word off by a fraction of a letter where the face is proportional, against
|
|
61
|
+
* a line marked entirely wrongly.
|
|
62
|
+
*/
|
|
63
|
+
function markedPieces(run, q, mark) {
|
|
64
|
+
const left = Math.min(run.x, run.endX);
|
|
65
|
+
const advance = Math.max(run.x, run.endX) - left;
|
|
66
|
+
const chars = [...run.text];
|
|
67
|
+
if (!(advance > 0) || chars.length === 0) return [{
|
|
68
|
+
...run,
|
|
69
|
+
markup: mark
|
|
70
|
+
}];
|
|
71
|
+
const at = (x) => Math.max(0, Math.min(chars.length, Math.round((x - left) / advance * chars.length)));
|
|
72
|
+
const from = at(q.x0);
|
|
73
|
+
const to = at(q.x1);
|
|
74
|
+
if (from <= 0 && to >= chars.length) return [{
|
|
75
|
+
...run,
|
|
76
|
+
markup: mark
|
|
77
|
+
}];
|
|
78
|
+
if (to <= from) return [run];
|
|
79
|
+
const per = advance / chars.length;
|
|
80
|
+
const piece = (a, b, marked) => ({
|
|
81
|
+
...run,
|
|
82
|
+
text: chars.slice(a, b).join(""),
|
|
83
|
+
x: left + a * per,
|
|
84
|
+
endX: left + b * per,
|
|
85
|
+
...marked ? { markup: mark } : {}
|
|
86
|
+
});
|
|
87
|
+
return [
|
|
88
|
+
...from > 0 ? [piece(0, from, false)] : [],
|
|
89
|
+
piece(from, to, true),
|
|
90
|
+
...to < chars.length ? [piece(to, chars.length, false)] : []
|
|
91
|
+
];
|
|
92
|
+
}
|
|
93
|
+
/** Every text-markup annotation on the page, with the boxes it marks. */
|
|
94
|
+
function collectTextMarkup(file, page) {
|
|
95
|
+
const annots = file.get(page.dict, "Annots");
|
|
96
|
+
if (!Array.isArray(annots)) return [];
|
|
97
|
+
const out = [];
|
|
98
|
+
for (const entry of annots) {
|
|
99
|
+
const annot = file.resolve(entry);
|
|
100
|
+
if (!(annot instanceof Map)) continue;
|
|
101
|
+
const mark = textMarkupOf(file, annot);
|
|
102
|
+
if (mark) out.push(mark);
|
|
103
|
+
}
|
|
104
|
+
return out;
|
|
105
|
+
}
|
|
33
106
|
function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
|
|
34
|
-
const result = interpretContent(content, buildFonts(file, resources), baseCtm);
|
|
107
|
+
const result = interpretContent(content, buildFonts(file, resources), baseCtm, void 0, void 0, void 0, hiddenProperties(file, resources));
|
|
35
108
|
out.push(...result.texts.map((r) => withPatternColour(file, resources, r, visiting)));
|
|
36
109
|
if (depth >= MAX_FORM_DEPTH) return;
|
|
37
110
|
for (const glyph of result.glyphs) {
|
|
@@ -46,6 +119,7 @@ function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
|
|
|
46
119
|
for (const placement of result.images) {
|
|
47
120
|
const stream = file.resolve(xobjects.get(placement.name) ?? PDF_NULL);
|
|
48
121
|
if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
|
|
122
|
+
if (hiddenXObject(file, stream)) continue;
|
|
49
123
|
const sub = file.get(stream.dict, "Subtype");
|
|
50
124
|
if (!(sub instanceof PdfName) || sub.value !== "Form") continue;
|
|
51
125
|
visiting.add(stream);
|
|
@@ -24,6 +24,11 @@ export interface PdfVector {
|
|
|
24
24
|
readonly gradient?: ShapeGradient;
|
|
25
25
|
/** §11.6.4.4 — how opaque the fill is, when the page asked for less than all. */
|
|
26
26
|
readonly alpha?: number;
|
|
27
|
+
/**
|
|
28
|
+
* §11.3.5 — the fill only DARKENS what it covers, so the marks under it read
|
|
29
|
+
* through. A highlighter is this, and nothing on a page reads it as paint.
|
|
30
|
+
*/
|
|
31
|
+
readonly darkens?: boolean;
|
|
27
32
|
/** Present iff a qualifying stroke survived (EP11). */
|
|
28
33
|
readonly strokeHex?: string;
|
|
29
34
|
/** Stroke width in page-space points (EP11). */
|
|
@@ -1,30 +1,32 @@
|
|
|
1
1
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
import { FEATURES } from "../core/ir/features.js";
|
|
3
3
|
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
4
|
+
import { buildAlphaMap, buildColorSpaceMap, buildShadingMap } from "./shading.js";
|
|
4
5
|
import { collectPageAppearances } from "./annots.js";
|
|
6
|
+
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
5
7
|
import { buildFonts } from "./text.js";
|
|
6
|
-
import { buildAlphaMap, buildColorSpaceMap, buildShadingMap } from "./shading.js";
|
|
7
8
|
//#region src/pdf-reader/vector.ts
|
|
8
|
-
|
|
9
|
-
* Every vector the page paints, its FORM XOBJECTS included (§8.8).
|
|
10
|
-
*
|
|
11
|
-
* A `Do` of a form is a call: its content stream draws in the caller's space
|
|
12
|
-
* through the form's own `/Matrix`. Interpreting the page stream alone reads
|
|
13
|
-
* only what the page drew directly, and a document that puts its artwork in
|
|
14
|
-
* forms — as every CAD and drawing producer does — comes back with none of it.
|
|
15
|
-
* 22060_A1_01_Plans.pdf holds eleven, and its floor plans were simply absent.
|
|
16
|
-
*
|
|
17
|
-
* The same walk `collectPageImages` already makes, with the same depth and
|
|
18
|
-
* cycle guards, collecting paths instead of pictures.
|
|
19
|
-
*/
|
|
20
|
-
function paintedVectors(file, page, shadings, alphas, spaces) {
|
|
9
|
+
function paintedVectors(file, page) {
|
|
21
10
|
const out = [];
|
|
22
11
|
const visiting = /* @__PURE__ */ new Set();
|
|
12
|
+
const stateCache = /* @__PURE__ */ new Map();
|
|
13
|
+
const mapsOf = (resources) => {
|
|
14
|
+
const had = stateCache.get(resources);
|
|
15
|
+
if (had) return had;
|
|
16
|
+
const made = {
|
|
17
|
+
shadings: buildShadingMap(file, resources),
|
|
18
|
+
alphas: buildAlphaMap(file, resources),
|
|
19
|
+
spaces: buildColorSpaceMap(file, resources)
|
|
20
|
+
};
|
|
21
|
+
stateCache.set(resources, made);
|
|
22
|
+
return made;
|
|
23
|
+
};
|
|
23
24
|
const walk = (resources, content, baseCtm, depth, prefix) => {
|
|
24
25
|
if (out.length >= MAX_VECTORS) return;
|
|
25
26
|
const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
|
|
26
27
|
const xobjDict = xobjects instanceof Map ? xobjects : void 0;
|
|
27
|
-
const
|
|
28
|
+
const maps = mapsOf(resources);
|
|
29
|
+
const result = interpretContent(content, buildFonts(file, resources), baseCtm, maps.shadings, maps.alphas, maps.spaces, hiddenProperties(file, resources));
|
|
28
30
|
const events = [
|
|
29
31
|
...result.vectors.map((vector) => ({
|
|
30
32
|
order: vector.order,
|
|
@@ -60,6 +62,7 @@ function paintedVectors(file, page, shadings, alphas, spaces) {
|
|
|
60
62
|
if (depth >= MAX_FORM_DEPTH) continue;
|
|
61
63
|
const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
|
|
62
64
|
if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
|
|
65
|
+
if (hiddenXObject(file, stream)) continue;
|
|
63
66
|
const subtype = file.get(stream.dict, "Subtype");
|
|
64
67
|
if (!(subtype instanceof PdfName) || subtype.value !== "Form") continue;
|
|
65
68
|
visiting.add(stream);
|
|
@@ -116,14 +119,22 @@ var MAX_FORM_DEPTH = 12;
|
|
|
116
119
|
* clutter. Clips and the bare `sh` operator are not captured (a documented loss).
|
|
117
120
|
*/
|
|
118
121
|
function collectPageVectors(file, page, occupied = []) {
|
|
119
|
-
const [px0, py0, px1, py1] = page.
|
|
122
|
+
const [px0, py0, px1, py1] = page.cropBox;
|
|
120
123
|
const pageArea = Math.max(1, Math.abs((px1 - px0) * (py1 - py0)));
|
|
121
|
-
const shadings = buildShadingMap(file, page);
|
|
122
|
-
const alphas = buildAlphaMap(file, page);
|
|
123
|
-
const spaces = buildColorSpaceMap(file, page);
|
|
124
124
|
const out = [];
|
|
125
125
|
const painted = [...occupied];
|
|
126
|
-
const raws = paintedVectors(file, page
|
|
126
|
+
const raws = paintedVectors(file, page);
|
|
127
|
+
const losses = [];
|
|
128
|
+
const asked = /* @__PURE__ */ new Set();
|
|
129
|
+
for (const raw of raws) {
|
|
130
|
+
if (raw.masked === true) asked.add("PDF soft mask (/SMask in the graphics state) is not applied; the shape is drawn at full opacity throughout");
|
|
131
|
+
if (raw.blend !== void 0) asked.add(`PDF blend mode /${raw.blend} is not performed; the shape is drawn over what it was to blend with`);
|
|
132
|
+
}
|
|
133
|
+
for (const detail of asked) losses.push({
|
|
134
|
+
severity: "degraded",
|
|
135
|
+
feature: FEATURES.images,
|
|
136
|
+
detail
|
|
137
|
+
});
|
|
127
138
|
for (const raw of raws) {
|
|
128
139
|
if (out.length >= MAX_VECTORS) break;
|
|
129
140
|
const v = clipped(raw);
|
|
@@ -139,7 +150,7 @@ function collectPageVectors(file, page, occupied = []) {
|
|
|
139
150
|
const short = Math.min(w, h);
|
|
140
151
|
const isBox = short >= MIN_SIDE && area >= MIN_AREA;
|
|
141
152
|
const isRule = long >= MIN_RULE_LEN && short > 0;
|
|
142
|
-
const filled = (v.gradient !== void 0 || solidFill) && (isBox || isRule)
|
|
153
|
+
const filled = (v.gradient !== void 0 || solidFill) && (isBox || isRule);
|
|
143
154
|
const stroked = v.strokeHex !== void 0 && (v.strokeHex !== "FFFFFF" || painted.some((box) => overlaps(box, b))) && Math.max(w, h) >= MIN_STROKE_LEN && area <= .85 * pageArea;
|
|
144
155
|
if (!filled && !stroked) continue;
|
|
145
156
|
painted.push(b);
|
|
@@ -148,6 +159,7 @@ function collectPageVectors(file, page, occupied = []) {
|
|
|
148
159
|
segs: v.segs,
|
|
149
160
|
...filled ? v.gradient ? { gradient: v.gradient } : v.fillHex !== void 0 ? { fillHex: v.fillHex } : {} : {},
|
|
150
161
|
...filled && v.alpha !== void 0 ? { alpha: v.alpha } : {},
|
|
162
|
+
...filled && v.darkens === true ? { darkens: true } : {},
|
|
151
163
|
...stroked ? {
|
|
152
164
|
strokeHex: v.strokeHex,
|
|
153
165
|
...v.lineWidth !== void 0 ? { lineWidth: v.lineWidth } : {}
|
|
@@ -156,13 +168,14 @@ function collectPageVectors(file, page, occupied = []) {
|
|
|
156
168
|
...v.mcid !== void 0 ? { mcid: v.mcid } : {}
|
|
157
169
|
});
|
|
158
170
|
}
|
|
171
|
+
if (out.length >= MAX_VECTORS || raws.length >= MAX_VECTORS) losses.push({
|
|
172
|
+
severity: "dropped",
|
|
173
|
+
feature: FEATURES.shapes,
|
|
174
|
+
detail: `page carries more than ${String(MAX_VECTORS)} painted paths; the rest were not read`
|
|
175
|
+
});
|
|
159
176
|
return {
|
|
160
177
|
vectors: out,
|
|
161
|
-
losses
|
|
162
|
-
severity: "dropped",
|
|
163
|
-
feature: FEATURES.shapes,
|
|
164
|
-
detail: `page carries more than ${String(MAX_VECTORS)} painted paths; the rest were not read`
|
|
165
|
-
}] : []
|
|
178
|
+
losses
|
|
166
179
|
};
|
|
167
180
|
}
|
|
168
181
|
/** Whether two boxes share any area at all. */
|