reamkit 1.25.1 → 1.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -5
- package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
- package/dist/esm/pdf-reader/annot-draw.js +374 -0
- package/dist/esm/pdf-reader/annots.d.ts +3 -1
- package/dist/esm/pdf-reader/annots.js +18 -4
- package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
- package/dist/esm/pdf-reader/ccitt.js +70 -2
- package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
- package/dist/esm/pdf-reader/cie-color.js +112 -0
- package/dist/esm/pdf-reader/content.d.ts +47 -17
- package/dist/esm/pdf-reader/content.js +175 -14
- package/dist/esm/pdf-reader/display.d.ts +1 -1
- package/dist/esm/pdf-reader/display.js +21 -6
- package/dist/esm/pdf-reader/document.d.ts +6 -0
- package/dist/esm/pdf-reader/document.js +29 -1
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
- package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
- package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
- package/dist/esm/pdf-reader/flow-build.js +102 -8
- package/dist/esm/pdf-reader/font.js +97 -16
- package/dist/esm/pdf-reader/function.d.ts +16 -0
- package/dist/esm/pdf-reader/function.js +414 -0
- package/dist/esm/pdf-reader/glyph-names.d.ts +6 -0
- package/dist/esm/pdf-reader/glyph-names.js +408 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
- package/dist/esm/pdf-reader/image-decode.js +224 -12
- package/dist/esm/pdf-reader/images.d.ts +12 -0
- package/dist/esm/pdf-reader/images.js +109 -9
- package/dist/esm/pdf-reader/jbig2.d.ts +109 -0
- package/dist/esm/pdf-reader/jbig2.js +2606 -0
- package/dist/esm/pdf-reader/layout.d.ts +32 -0
- package/dist/esm/pdf-reader/layout.js +189 -40
- package/dist/esm/pdf-reader/lexer.d.ts +2 -0
- package/dist/esm/pdf-reader/lexer.js +4 -0
- package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
- package/dist/esm/pdf-reader/optional-content.js +93 -0
- package/dist/esm/pdf-reader/reader.d.ts +5 -2
- package/dist/esm/pdf-reader/reader.js +80 -7
- package/dist/esm/pdf-reader/shading.d.ts +71 -6
- package/dist/esm/pdf-reader/shading.js +185 -15
- package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
- package/dist/esm/pdf-reader/standard-metrics.js +18 -0
- package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
- package/dist/esm/pdf-reader/standard-widths.js +62 -0
- package/dist/esm/pdf-reader/tagged.js +204 -32
- package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
- package/dist/esm/pdf-reader/text-rules.js +112 -0
- package/dist/esm/pdf-reader/text.js +78 -4
- package/dist/esm/pdf-reader/vector.d.ts +5 -0
- package/dist/esm/pdf-reader/vector.js +39 -25
- package/dist/esm/word/docx-writer.js +213 -23
- package/package.json +1 -1
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { PdfFile } from './document.js';
|
|
2
2
|
import { Reconstruction } from './flow-build.js';
|
|
3
|
+
/** §9.10.2 — a glyph the face maps to no character (see `./font`). */
|
|
4
|
+
export declare const UNMAPPED = "\uFFFD";
|
|
3
5
|
/**
|
|
4
6
|
* Heuristically reconstruct an untagged PDF into a {@link Reconstruction}
|
|
5
7
|
* (E-PDF EP4). With no structure tree there is only positioned content, so
|
|
@@ -16,3 +18,33 @@ import { Reconstruction } from './flow-build.js';
|
|
|
16
18
|
* @returns The reconstructed {@link FlowDoc} plus any read-time losses.
|
|
17
19
|
*/
|
|
18
20
|
export declare function reconstructByLayout(file: PdfFile, mode?: 'flow' | 'positional'): Reconstruction;
|
|
21
|
+
/**
|
|
22
|
+
* Whether a line ENDED a paragraph, rather than wrapping into the next.
|
|
23
|
+
*
|
|
24
|
+
* Leading alone cannot tell the two apart: five labels stacked at 15pt with a
|
|
25
|
+
* 12pt face look exactly like five wrapped lines, and alphatrans.pdf's five are
|
|
26
|
+
* read as one paragraph and re-wrapped into two. But a wrapping engine pulls
|
|
27
|
+
* the next word UP — so a line that stops well short of the measure stopped
|
|
28
|
+
* because its author stopped it, and the line after it begins something new.
|
|
29
|
+
* The same rule separates two paragraphs set with no extra space between them,
|
|
30
|
+
* which used to run together for the same reason.
|
|
31
|
+
*
|
|
32
|
+
* Only where both lines start at the same edge. Where they do not, the block is
|
|
33
|
+
* placed rather than set — a centred title's every line is short of the measure
|
|
34
|
+
* and none of them ends anything.
|
|
35
|
+
*
|
|
36
|
+
* @param prev The line before: where it starts, how wide it is, its face.
|
|
37
|
+
* @param next The line after — only where it starts matters.
|
|
38
|
+
* @param column The measure both were set in, when it is known.
|
|
39
|
+
* @returns Whether the first line ended a paragraph.
|
|
40
|
+
*/
|
|
41
|
+
export declare function endedParagraph(prev: {
|
|
42
|
+
x: number;
|
|
43
|
+
width: number;
|
|
44
|
+
fontSize: number;
|
|
45
|
+
}, next: {
|
|
46
|
+
x: number;
|
|
47
|
+
}, column: {
|
|
48
|
+
left: number;
|
|
49
|
+
right: number;
|
|
50
|
+
} | undefined): boolean;
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { pt } from "../core/ir/units.js";
|
|
2
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
3
|
import { FEATURES } from "../core/ir/features.js";
|
|
4
|
-
import { displayOf, placeImages, placeRuns, placeVectors } from "./display.js";
|
|
5
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
|
|
6
|
-
import { collectEmbeddedFonts } from "./embedded-fonts.js";
|
|
7
4
|
import { isRightToLeft } from "./content.js";
|
|
8
|
-
import {
|
|
5
|
+
import { collectEmbeddedFonts } from "./embedded-fonts.js";
|
|
9
6
|
import { extractPageText } from "./text.js";
|
|
10
7
|
import { collectPageVectors } from "./vector.js";
|
|
11
|
-
|
|
8
|
+
import { displayOf, placeImages, placeRuns, placeVectors } from "./display.js";
|
|
9
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock, withMeasuredMargins } from "./flow-build.js";
|
|
10
|
+
import { collectPageImages } from "./images.js";
|
|
11
|
+
import { markDrawnRules } from "./text-rules.js";
|
|
12
12
|
/**
|
|
13
13
|
* Heuristically reconstruct an untagged PDF into a {@link Reconstruction}
|
|
14
14
|
* (E-PDF EP4). With no structure tree there is only positioned content, so
|
|
@@ -31,6 +31,11 @@ function reconstructByLayout(file, mode = "flow") {
|
|
|
31
31
|
const medianFont = median(pageRuns.flat().map((r) => r.fontSizePt).filter((s) => s > 0)) || 12;
|
|
32
32
|
const resources = new ResourceStore();
|
|
33
33
|
const losses = [];
|
|
34
|
+
if (pageRuns.some((page) => page.some((r) => r.text.includes("�")))) losses.push({
|
|
35
|
+
severity: "dropped",
|
|
36
|
+
feature: FEATURES.text,
|
|
37
|
+
detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
|
|
38
|
+
});
|
|
34
39
|
if (pageRuns.some((page) => page.some((r) => r.fillPatternName !== void 0))) losses.push({
|
|
35
40
|
severity: "degraded",
|
|
36
41
|
feature: FEATURES.text,
|
|
@@ -42,11 +47,12 @@ function reconstructByLayout(file, mode = "flow") {
|
|
|
42
47
|
const display = shown[i];
|
|
43
48
|
const pageWidth = display.width;
|
|
44
49
|
const gutter = detectGutter(runs, pageWidth);
|
|
50
|
+
const stepped = stepsBetweenWords(runs);
|
|
45
51
|
const blocks = [];
|
|
46
52
|
const addColumn = (allRuns, col) => {
|
|
47
53
|
const colRuns = mode === "positional" ? allRuns.filter((r) => r.type3 !== true && r.invisible !== true) : allRuns;
|
|
48
54
|
if (mode === "positional") {
|
|
49
|
-
for (const [angle, runs] of byAngle(colRuns)) for (const line of groupIntoLines(rotate(runs, -angle), true)) {
|
|
55
|
+
for (const [angle, runs] of byAngle(colRuns)) for (const line of groupIntoLines(rotate(runs, -angle), true, stepped)) {
|
|
50
56
|
if (line.text.length === 0) continue;
|
|
51
57
|
const box = turnedBox(line, angle, pageWidth);
|
|
52
58
|
placed.push({
|
|
@@ -58,18 +64,21 @@ function reconstructByLayout(file, mode = "flow") {
|
|
|
58
64
|
}
|
|
59
65
|
return;
|
|
60
66
|
}
|
|
61
|
-
const lines = groupIntoLines(colRuns).filter((l) => l.text.length > 0);
|
|
62
|
-
|
|
67
|
+
const lines = groupIntoLines(colRuns, false, stepped).filter((l) => l.text.length > 0);
|
|
68
|
+
const measure = lines.length > 0 ? {
|
|
69
|
+
left: Math.min(...lines.map((l) => l.x)),
|
|
70
|
+
right: Math.max(...lines.map((l) => l.x + l.width))
|
|
71
|
+
} : void 0;
|
|
72
|
+
for (const para of groupIntoParagraphs(lines, measure, display.height)) blocks.push({
|
|
63
73
|
col,
|
|
64
74
|
top: para.top,
|
|
65
|
-
el: paragraphFromRuns(para.spans, headingLevel(para.fontSize, medianFont)
|
|
75
|
+
el: paragraphFromRuns(para.spans, headingLevel(para.fontSize, medianFont), {
|
|
76
|
+
...para.alignment !== void 0 ? { alignment: para.alignment } : {},
|
|
77
|
+
...para.spacingBefore !== void 0 ? { spacingBefore: pt(para.spacingBefore) } : {}
|
|
78
|
+
})
|
|
66
79
|
});
|
|
67
80
|
};
|
|
68
81
|
const placed = [];
|
|
69
|
-
if (gutter !== void 0) {
|
|
70
|
-
addColumn(runs.filter((r) => r.x < gutter), 0);
|
|
71
|
-
addColumn(runs.filter((r) => r.x >= gutter), 1);
|
|
72
|
-
} else addColumn(runs, 0);
|
|
73
82
|
const colOf = (centerX) => gutter !== void 0 && centerX >= gutter ? 1 : 0;
|
|
74
83
|
const frame = {
|
|
75
84
|
left: 0,
|
|
@@ -88,19 +97,26 @@ function reconstructByLayout(file, mode = "flow") {
|
|
|
88
97
|
maxY: img.y + img.heightPt
|
|
89
98
|
})));
|
|
90
99
|
losses.push(...lifted.losses);
|
|
91
|
-
const
|
|
100
|
+
const placedVectors = placeVectors(lifted.vectors, display);
|
|
101
|
+
const ruled = markDrawnRules(runs, placedVectors);
|
|
102
|
+
const vectors = placedVectors.filter((v) => !ruled.consumed.has(v));
|
|
103
|
+
if (gutter !== void 0) {
|
|
104
|
+
addColumn(ruled.runs.filter((r) => r.x < gutter), 0);
|
|
105
|
+
addColumn(ruled.runs.filter((r) => r.x >= gutter), 1);
|
|
106
|
+
} else addColumn(ruled.runs, 0);
|
|
107
|
+
const under = mode !== "positional";
|
|
92
108
|
[
|
|
93
109
|
...imgs.images.map((img) => ({
|
|
94
110
|
key: img.orderKey,
|
|
95
111
|
col: colOf(img.x + img.widthPt / 2),
|
|
96
112
|
top: img.y + img.heightPt,
|
|
97
|
-
make: (z) => imageBlock(img, resources, void 0, frame, z)
|
|
113
|
+
make: (z) => imageBlock(img, resources, void 0, frame, z, under)
|
|
98
114
|
})),
|
|
99
115
|
...vectors.map((v) => ({
|
|
100
116
|
key: v.orderKey,
|
|
101
117
|
col: colOf((v.minX + v.maxX) / 2),
|
|
102
118
|
top: v.maxY,
|
|
103
|
-
make: (z) => shapeBlock(v, frame, z)
|
|
119
|
+
make: (z) => shapeBlock(v, frame, z, under)
|
|
104
120
|
})),
|
|
105
121
|
...placed
|
|
106
122
|
].sort((a, b) => compareOrder(a.key, b.key)).forEach((mark, z) => {
|
|
@@ -124,8 +140,9 @@ function reconstructByLayout(file, mode = "flow") {
|
|
|
124
140
|
});
|
|
125
141
|
for (const block of blocks) body.push(block.el);
|
|
126
142
|
});
|
|
143
|
+
const section = sectionFromPdfPages(pages);
|
|
127
144
|
return {
|
|
128
|
-
doc: buildFlowDoc(body, resources,
|
|
145
|
+
doc: buildFlowDoc(body, resources, mode === "positional" ? section : withMeasuredMargins(section, shown, pageRuns), collectEmbeddedFonts(file, pages, losses)),
|
|
129
146
|
losses: dedupeLosses(losses)
|
|
130
147
|
};
|
|
131
148
|
}
|
|
@@ -241,7 +258,7 @@ function rotation60kOf(angleDeg) {
|
|
|
241
258
|
* 0.05, and thirty at 0.05 em or more. A twentieth of an em sits in that gap.
|
|
242
259
|
*/
|
|
243
260
|
var BASELINE_STEP_EM = .05;
|
|
244
|
-
function groupIntoLines(runs, split = false) {
|
|
261
|
+
function groupIntoLines(runs, split = false, stepped = false) {
|
|
245
262
|
const sorted = [...runs].sort((a, b) => b.y - a.y || a.x - b.x);
|
|
246
263
|
const clusters = [];
|
|
247
264
|
for (const run of sorted) {
|
|
@@ -259,7 +276,7 @@ function groupIntoLines(runs, split = false) {
|
|
|
259
276
|
return clusters.flatMap((c) => {
|
|
260
277
|
const ordered = c.runs.sort((a, b) => a.x - b.x);
|
|
261
278
|
const fontSize = c.fontSize || 10;
|
|
262
|
-
if (!split) return [lineOf(ordered, c.y, fontSize)];
|
|
279
|
+
if (!split) return [lineOf(ordered, c.y, fontSize, stepped)];
|
|
263
280
|
const pieces = [[]];
|
|
264
281
|
for (const run of ordered) {
|
|
265
282
|
const prev = pieces[pieces.length - 1];
|
|
@@ -268,30 +285,96 @@ function groupIntoLines(runs, split = false) {
|
|
|
268
285
|
if (last !== void 0 && (Math.abs(run.x - last.endX) > size * SPACE_GAP_EM || Math.abs(run.y - prev[0].y) > size * BASELINE_STEP_EM || isRightToLeft(run.text) || isRightToLeft(last.text))) pieces.push([]);
|
|
269
286
|
pieces[pieces.length - 1].push(run);
|
|
270
287
|
}
|
|
271
|
-
return pieces.map((piece) => lineOf(piece, piece[0].y, Math.max(...piece.map((r) => r.fontSizePt || 0)) || fontSize));
|
|
288
|
+
return pieces.map((piece) => lineOf(piece, piece[0].y, Math.max(...piece.map((r) => r.fontSizePt || 0)) || fontSize, stepped));
|
|
272
289
|
});
|
|
273
290
|
}
|
|
291
|
+
/**
|
|
292
|
+
* Where a line's INK is, which is not how far its pen travelled.
|
|
293
|
+
*
|
|
294
|
+
* A run of spaces advances the pen and marks nothing. basicapi.pdf sets its
|
|
295
|
+
* page number as thirty-one spaces and "page 1 / 3" in ONE run, reaching 635pt
|
|
296
|
+
* across a 595pt sheet — and the measure taken off that line was wide enough
|
|
297
|
+
* that the centred title in the same column no longer looked centred.
|
|
298
|
+
*
|
|
299
|
+
* The blanks are deducted at the face's own space width, which the run carries
|
|
300
|
+
* (§9.4.4); where the face states none, at a quarter of the size.
|
|
301
|
+
*/
|
|
302
|
+
function inkSpan(runs) {
|
|
303
|
+
const marked = runs.filter((r) => r.text.trim().length > 0);
|
|
304
|
+
if (marked.length === 0) {
|
|
305
|
+
const x = runs[0].x;
|
|
306
|
+
return {
|
|
307
|
+
x,
|
|
308
|
+
width: runs[runs.length - 1].endX - x
|
|
309
|
+
};
|
|
310
|
+
}
|
|
311
|
+
const first = marked[0];
|
|
312
|
+
const last = marked[marked.length - 1];
|
|
313
|
+
const space = (r) => r.spaceWidthPt !== void 0 && r.spaceWidthPt > 0 ? r.spaceWidthPt : (r.fontSizePt || 10) * .25;
|
|
314
|
+
const lead = (/^\s*/u.exec(first.text)?.[0].length ?? 0) * space(first);
|
|
315
|
+
const trail = (/\s*$/u.exec(last.text)?.[0].length ?? 0) * space(last);
|
|
316
|
+
const x = first.x + lead;
|
|
317
|
+
return {
|
|
318
|
+
x,
|
|
319
|
+
width: Math.max(0, last.endX - trail - x)
|
|
320
|
+
};
|
|
321
|
+
}
|
|
274
322
|
/** One run of runs, left to right on a shared baseline, as a {@link Line}. */
|
|
275
|
-
function lineOf(runs, y, fontSize) {
|
|
276
|
-
const ordered = lineSpans(runs, fontSize);
|
|
323
|
+
function lineOf(runs, y, fontSize, stepped) {
|
|
324
|
+
const ordered = lineSpans(runs, fontSize, stepped);
|
|
277
325
|
const spans = ordered.every((s) => s.text.trim() === "" || isRightToLeft(s.text)) ? [...ordered].reverse() : ordered;
|
|
278
|
-
const
|
|
326
|
+
const ink = inkSpan(runs);
|
|
279
327
|
return {
|
|
280
|
-
x,
|
|
281
|
-
width:
|
|
328
|
+
x: ink.x,
|
|
329
|
+
width: ink.width,
|
|
282
330
|
y,
|
|
283
331
|
fontSize,
|
|
284
332
|
text: spans.map((s) => s.text).join("").replace(/\s+/g, " ").trim(),
|
|
285
333
|
spans
|
|
286
334
|
};
|
|
287
335
|
}
|
|
288
|
-
|
|
336
|
+
/**
|
|
337
|
+
* The gap between two runs that means a WORD SPACE stood there.
|
|
338
|
+
*
|
|
339
|
+
* A page that draws its own spaces has already said where its words divide, and
|
|
340
|
+
* a gap between two of its runs is a COLUMN or a placement — 160F-2019.pdf is
|
|
341
|
+
* ruled into fields a quarter-inch apart and a generous threshold keeps them
|
|
342
|
+
* apart. A page that draws none has said nothing, and every word boundary on it
|
|
343
|
+
* is a gap: bigboundingbox.pdf steps 0.226 em between words and never writes a
|
|
344
|
+
* space, so at a quarter em its every line ran together — "OrangeDemoInc.",
|
|
345
|
+
* "Whenpayingbycheck,pleasecompletethispaymentadvice".
|
|
346
|
+
*
|
|
347
|
+
* The two want different thresholds, and the page says which it is (see
|
|
348
|
+
* {@link stepsBetweenWords}). The tight one still clears the gaps a producer
|
|
349
|
+
* leaves INSIDE a word when it splits one for kerning, which measure eight
|
|
350
|
+
* hundredths of an em at their widest across this corpus.
|
|
351
|
+
*/
|
|
352
|
+
function spaceGap(prev, fontSize, stepped) {
|
|
353
|
+
return (prev.fontSizePt || fontSize) * (stepped ? STEPPED_SPACE_EM : DRAWN_SPACE_EM);
|
|
354
|
+
}
|
|
355
|
+
/** A page that writes its own spaces: only a wide gap means anything more. */
|
|
356
|
+
var DRAWN_SPACE_EM = .25;
|
|
357
|
+
/** A page that writes none: the step between its words is all there is. */
|
|
358
|
+
var STEPPED_SPACE_EM = .12;
|
|
359
|
+
/**
|
|
360
|
+
* Whether this page STEPS between its words rather than writing spaces.
|
|
361
|
+
*
|
|
362
|
+
* Counted rather than guessed: bigboundingbox.pdf writes a space in one run in
|
|
363
|
+
* a hundred, TAMReview.pdf in a third of them, and no page does a little of
|
|
364
|
+
* both. A page with almost no text says nothing either way and keeps the
|
|
365
|
+
* cautious reading.
|
|
366
|
+
*/
|
|
367
|
+
function stepsBetweenWords(runs) {
|
|
368
|
+
if (runs.length < 8) return false;
|
|
369
|
+
return runs.filter((r) => /\s/u.test(r.text)).length / runs.length < .05;
|
|
370
|
+
}
|
|
371
|
+
function lineSpans(runs, fontSize, stepped) {
|
|
289
372
|
const spans = [];
|
|
290
|
-
let
|
|
373
|
+
let prev;
|
|
291
374
|
for (const run of runs) {
|
|
292
|
-
if (
|
|
375
|
+
if (prev !== void 0 && run.x - prev.endX > spaceGap(prev, fontSize, stepped)) spans.push({ text: " " });
|
|
293
376
|
spans.push({
|
|
294
|
-
text: run.text,
|
|
377
|
+
text: run.text.replaceAll("�", ""),
|
|
295
378
|
sizePt: run.fontSizePt,
|
|
296
379
|
...run.colorHex !== "000000" ? { colorHex: run.colorHex } : {},
|
|
297
380
|
...run.fontName !== void 0 ? { fontName: run.fontName } : {},
|
|
@@ -301,26 +384,92 @@ function lineSpans(runs, fontSize) {
|
|
|
301
384
|
} } : {},
|
|
302
385
|
...run.bold ? { bold: true } : {},
|
|
303
386
|
...run.italic ? { italic: true } : {},
|
|
387
|
+
...run.markup !== void 0 ? { markup: run.markup } : {},
|
|
304
388
|
...run.href !== void 0 ? { href: run.href } : {}
|
|
305
389
|
});
|
|
306
|
-
|
|
390
|
+
prev = run;
|
|
307
391
|
}
|
|
308
392
|
return spans;
|
|
309
393
|
}
|
|
310
|
-
function groupIntoParagraphs(lines) {
|
|
394
|
+
function groupIntoParagraphs(lines, column, pageHeight = 0) {
|
|
311
395
|
const groups = [];
|
|
312
|
-
|
|
396
|
+
const gaps = [];
|
|
397
|
+
let prev;
|
|
313
398
|
for (const line of lines) {
|
|
314
|
-
const gap =
|
|
315
|
-
|
|
399
|
+
const gap = prev !== void 0 ? prev.y - line.y : 0;
|
|
400
|
+
const opened = prev !== void 0 && gap > line.fontSize * 1.5;
|
|
401
|
+
if (groups.length === 0 || opened || prev !== void 0 && endedParagraph(prev, line, column)) {
|
|
402
|
+
groups.push([]);
|
|
403
|
+
gaps.push(prev === void 0 ? 0 : gap);
|
|
404
|
+
}
|
|
316
405
|
groups[groups.length - 1].push(line);
|
|
317
|
-
|
|
406
|
+
prev = line;
|
|
318
407
|
}
|
|
319
|
-
return groups.map((g) =>
|
|
320
|
-
|
|
321
|
-
fontSize
|
|
322
|
-
|
|
408
|
+
return groups.map((g, i) => {
|
|
409
|
+
const first = g[0];
|
|
410
|
+
const fontSize = Math.max(...g.map((l) => l.fontSize));
|
|
411
|
+
const opened = (gaps[i] ?? 0) - fontSize * 1.2;
|
|
412
|
+
const most = pageHeight > 0 ? pageHeight / 3 : fontSize * 3;
|
|
413
|
+
const spacingBefore = opened > fontSize * .3 ? Math.min(opened, most) : void 0;
|
|
414
|
+
return {
|
|
415
|
+
spans: g.flatMap((l, k) => k > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
|
|
416
|
+
fontSize,
|
|
417
|
+
top: first.y,
|
|
418
|
+
...spacingBefore !== void 0 ? { spacingBefore } : {},
|
|
419
|
+
...alignmentOf(g, column)
|
|
420
|
+
};
|
|
421
|
+
});
|
|
422
|
+
}
|
|
423
|
+
/**
|
|
424
|
+
* Whether a line ENDED a paragraph, rather than wrapping into the next.
|
|
425
|
+
*
|
|
426
|
+
* Leading alone cannot tell the two apart: five labels stacked at 15pt with a
|
|
427
|
+
* 12pt face look exactly like five wrapped lines, and alphatrans.pdf's five are
|
|
428
|
+
* read as one paragraph and re-wrapped into two. But a wrapping engine pulls
|
|
429
|
+
* the next word UP — so a line that stops well short of the measure stopped
|
|
430
|
+
* because its author stopped it, and the line after it begins something new.
|
|
431
|
+
* The same rule separates two paragraphs set with no extra space between them,
|
|
432
|
+
* which used to run together for the same reason.
|
|
433
|
+
*
|
|
434
|
+
* Only where both lines start at the same edge. Where they do not, the block is
|
|
435
|
+
* placed rather than set — a centred title's every line is short of the measure
|
|
436
|
+
* and none of them ends anything.
|
|
437
|
+
*
|
|
438
|
+
* @param prev The line before: where it starts, how wide it is, its face.
|
|
439
|
+
* @param next The line after — only where it starts matters.
|
|
440
|
+
* @param column The measure both were set in, when it is known.
|
|
441
|
+
* @returns Whether the first line ended a paragraph.
|
|
442
|
+
*/
|
|
443
|
+
function endedParagraph(prev, next, column) {
|
|
444
|
+
if (!column) return false;
|
|
445
|
+
const width = column.right - column.left;
|
|
446
|
+
if (!(width > 0)) return false;
|
|
447
|
+
if (Math.abs(prev.x - next.x) > Math.max(prev.fontSize, 4)) return false;
|
|
448
|
+
return column.right - (prev.x + prev.width) > width * .25;
|
|
449
|
+
}
|
|
450
|
+
/**
|
|
451
|
+
* §17.3.1.13 — where a paragraph sits across its column, which is the only
|
|
452
|
+
* witness a PDF leaves of how it was set: every line is placed absolutely and
|
|
453
|
+
* nothing says "centred".
|
|
454
|
+
*
|
|
455
|
+
* A paragraph whose lines are inset by about as much on each side is centred; a
|
|
456
|
+
* one-line paragraph pushed to the right edge is right-aligned. Everything else
|
|
457
|
+
* is left alone — a justified paragraph and a ragged-right one look the same
|
|
458
|
+
* from here, and guessing between them would re-set the body of every document.
|
|
459
|
+
*/
|
|
460
|
+
function alignmentOf(lines, column) {
|
|
461
|
+
if (!column || lines.length === 0) return {};
|
|
462
|
+
const width = column.right - column.left;
|
|
463
|
+
if (!(width > 0)) return {};
|
|
464
|
+
const insets = lines.map((l) => ({
|
|
465
|
+
lead: l.x - column.left,
|
|
466
|
+
trail: column.right - (l.x + l.width)
|
|
323
467
|
}));
|
|
468
|
+
const meaningful = width * .1;
|
|
469
|
+
const even = width * .06;
|
|
470
|
+
if (insets.every((i) => Math.abs(i.lead - i.trail) <= even) && Math.max(...insets.map((i) => Math.min(i.lead, i.trail))) >= meaningful) return { alignment: "center" };
|
|
471
|
+
if (insets.every((i) => i.trail <= even) && Math.max(...insets.map((i) => i.lead)) >= meaningful) return { alignment: "right" };
|
|
472
|
+
return {};
|
|
324
473
|
}
|
|
325
474
|
function headingLevel(fontSize, medianFont) {
|
|
326
475
|
if (fontSize >= medianFont * 1.5) return 0;
|
|
@@ -342,4 +491,4 @@ function compareOrder(a, b) {
|
|
|
342
491
|
return a.length - b.length;
|
|
343
492
|
}
|
|
344
493
|
//#endregion
|
|
345
|
-
export { reconstructByLayout };
|
|
494
|
+
export { endedParagraph, reconstructByLayout };
|
|
@@ -45,6 +45,8 @@ export declare class Lexer {
|
|
|
45
45
|
constructor(buf: Uint8Array, pos?: number);
|
|
46
46
|
/** The length of the underlying byte buffer. */
|
|
47
47
|
get length(): number;
|
|
48
|
+
/** The bytes between two offsets, as a view onto the buffer. */
|
|
49
|
+
slice(from: number, to: number): Uint8Array;
|
|
48
50
|
/** The byte at index `i`, or −1 when out of range. */
|
|
49
51
|
byteAt(i: number): number;
|
|
50
52
|
/** §7.2.3 — skip whitespace and `%`-to-end-of-line comments. */
|
|
@@ -34,6 +34,10 @@ var Lexer = class {
|
|
|
34
34
|
get length() {
|
|
35
35
|
return this.buf.length;
|
|
36
36
|
}
|
|
37
|
+
/** The bytes between two offsets, as a view onto the buffer. */
|
|
38
|
+
slice(from, to) {
|
|
39
|
+
return this.buf.subarray(Math.max(0, from), Math.min(this.buf.length, Math.max(from, to)));
|
|
40
|
+
}
|
|
37
41
|
/** The byte at index `i`, or −1 when out of range. */
|
|
38
42
|
byteAt(i) {
|
|
39
43
|
return i >= 0 && i < this.buf.length ? this.buf[i] : -1;
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { PdfDict, PdfStream, PdfValue } from '../pdf/objects.js';
|
|
2
|
+
import { PdfFile } from './document.js';
|
|
3
|
+
/**
|
|
4
|
+
* The optional-content groups the file's DEFAULT configuration turns off
|
|
5
|
+
* (§8.11.4.3).
|
|
6
|
+
*
|
|
7
|
+
* `/BaseState` says what an unlisted group does — `/ON` unless the file says
|
|
8
|
+
* `/OFF` — and the `/ON` and `/OFF` arrays name the exceptions, `/OFF` last.
|
|
9
|
+
*
|
|
10
|
+
* @param file The document.
|
|
11
|
+
* @returns The resolved OCG dictionaries that are hidden.
|
|
12
|
+
*/
|
|
13
|
+
export declare function hiddenGroups(file: PdfFile): ReadonlySet<PdfValue>;
|
|
14
|
+
/**
|
|
15
|
+
* Whether an `/OC` entry names something the page does not show.
|
|
16
|
+
*
|
|
17
|
+
* The entry is either a group itself or an `/OCMD` — a membership dictionary
|
|
18
|
+
* naming several groups and a `/P` policy over them (§8.11.2.3). `AnyOn` is the
|
|
19
|
+
* default and the common case: the content shows if any of its groups does.
|
|
20
|
+
*
|
|
21
|
+
* @param file The document.
|
|
22
|
+
* @param oc The `/OC` value, unresolved.
|
|
23
|
+
* @returns `true` where the content it guards is hidden.
|
|
24
|
+
*/
|
|
25
|
+
export declare function hiddenByOc(file: PdfFile, oc: PdfValue | undefined): boolean;
|
|
26
|
+
/**
|
|
27
|
+
* The `/Properties` names a page's content may name in `/OC … BDC` that are
|
|
28
|
+
* hidden (§8.11.3.2).
|
|
29
|
+
*
|
|
30
|
+
* @param file The document.
|
|
31
|
+
* @param resources The resource dictionary in force.
|
|
32
|
+
* @returns Name → hidden, for the names that ARE hidden.
|
|
33
|
+
*/
|
|
34
|
+
export declare function hiddenProperties(file: PdfFile, resources: PdfDict | undefined): Set<string>;
|
|
35
|
+
/** Whether an XObject carries an `/OC` that hides it (§8.11.3.1). */
|
|
36
|
+
export declare function hiddenXObject(file: PdfFile, stream: PdfStream): boolean;
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import { PDF_NULL, PdfName } from "../pdf/objects.js";
|
|
2
|
+
//#region src/pdf-reader/optional-content.ts
|
|
3
|
+
/** The document's own answer to "is this group shown?", worked out once. */
|
|
4
|
+
var cache = /* @__PURE__ */ new WeakMap();
|
|
5
|
+
/** An `/OCMD` naming more groups than this is not read: it is not a document. */
|
|
6
|
+
var MAX_GROUPS = 4096;
|
|
7
|
+
/**
|
|
8
|
+
* The optional-content groups the file's DEFAULT configuration turns off
|
|
9
|
+
* (§8.11.4.3).
|
|
10
|
+
*
|
|
11
|
+
* `/BaseState` says what an unlisted group does — `/ON` unless the file says
|
|
12
|
+
* `/OFF` — and the `/ON` and `/OFF` arrays name the exceptions, `/OFF` last.
|
|
13
|
+
*
|
|
14
|
+
* @param file The document.
|
|
15
|
+
* @returns The resolved OCG dictionaries that are hidden.
|
|
16
|
+
*/
|
|
17
|
+
function hiddenGroups(file) {
|
|
18
|
+
const had = cache.get(file);
|
|
19
|
+
if (had) return had;
|
|
20
|
+
const hidden = /* @__PURE__ */ new Set();
|
|
21
|
+
const props = file.get(file.catalog, "OCProperties");
|
|
22
|
+
const config = props instanceof Map ? file.get(props, "D") : void 0;
|
|
23
|
+
if (config instanceof Map) {
|
|
24
|
+
const base = file.get(config, "BaseState");
|
|
25
|
+
if (base instanceof PdfName && base.value === "OFF") {
|
|
26
|
+
const all = file.get(props instanceof Map ? props : /* @__PURE__ */ new Map(), "OCGs");
|
|
27
|
+
for (const g of asArray(file, all)) hidden.add(g);
|
|
28
|
+
}
|
|
29
|
+
for (const g of asArray(file, file.get(config, "ON"))) hidden.delete(g);
|
|
30
|
+
for (const g of asArray(file, file.get(config, "OFF"))) hidden.add(g);
|
|
31
|
+
}
|
|
32
|
+
cache.set(file, hidden);
|
|
33
|
+
return hidden;
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Whether an `/OC` entry names something the page does not show.
|
|
37
|
+
*
|
|
38
|
+
* The entry is either a group itself or an `/OCMD` — a membership dictionary
|
|
39
|
+
* naming several groups and a `/P` policy over them (§8.11.2.3). `AnyOn` is the
|
|
40
|
+
* default and the common case: the content shows if any of its groups does.
|
|
41
|
+
*
|
|
42
|
+
* @param file The document.
|
|
43
|
+
* @param oc The `/OC` value, unresolved.
|
|
44
|
+
* @returns `true` where the content it guards is hidden.
|
|
45
|
+
*/
|
|
46
|
+
function hiddenByOc(file, oc) {
|
|
47
|
+
if (oc === void 0) return false;
|
|
48
|
+
const hidden = hiddenGroups(file);
|
|
49
|
+
const resolved = file.resolve(oc);
|
|
50
|
+
if (!(resolved instanceof Map)) return false;
|
|
51
|
+
const type = file.get(resolved, "Type");
|
|
52
|
+
if (!(type instanceof PdfName) || type.value !== "OCMD") return hidden.has(resolved);
|
|
53
|
+
const groups = asArray(file, file.get(resolved, "OCGs"));
|
|
54
|
+
const single = file.resolve(resolved.get("OCGs") ?? PDF_NULL);
|
|
55
|
+
const list = groups.length > 0 ? groups : single instanceof Map ? [single] : [];
|
|
56
|
+
if (list.length === 0) return false;
|
|
57
|
+
const on = list.filter((g) => !hidden.has(g)).length;
|
|
58
|
+
const policy = file.get(resolved, "P");
|
|
59
|
+
switch (policy instanceof PdfName ? policy.value : "AnyOn") {
|
|
60
|
+
case "AllOn": return on < list.length;
|
|
61
|
+
case "AnyOff": return on === list.length;
|
|
62
|
+
case "AllOff": return on > 0;
|
|
63
|
+
default: return on === 0;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* The `/Properties` names a page's content may name in `/OC … BDC` that are
|
|
68
|
+
* hidden (§8.11.3.2).
|
|
69
|
+
*
|
|
70
|
+
* @param file The document.
|
|
71
|
+
* @param resources The resource dictionary in force.
|
|
72
|
+
* @returns Name → hidden, for the names that ARE hidden.
|
|
73
|
+
*/
|
|
74
|
+
function hiddenProperties(file, resources) {
|
|
75
|
+
const out = /* @__PURE__ */ new Set();
|
|
76
|
+
if (!resources) return out;
|
|
77
|
+
const props = file.get(resources, "Properties");
|
|
78
|
+
if (!(props instanceof Map)) return out;
|
|
79
|
+
for (const [name, value] of props) if (hiddenByOc(file, value)) out.add(name);
|
|
80
|
+
return out;
|
|
81
|
+
}
|
|
82
|
+
/** Whether an XObject carries an `/OC` that hides it (§8.11.3.1). */
|
|
83
|
+
function hiddenXObject(file, stream) {
|
|
84
|
+
return hiddenByOc(file, stream.dict.get("OC"));
|
|
85
|
+
}
|
|
86
|
+
/** An array entry, resolved; a missing or malformed one comes back empty. */
|
|
87
|
+
function asArray(file, value) {
|
|
88
|
+
const r = value !== void 0 ? file.resolve(value) : void 0;
|
|
89
|
+
if (!Array.isArray(r)) return [];
|
|
90
|
+
return r.slice(0, MAX_GROUPS).map((v) => file.resolve(v));
|
|
91
|
+
}
|
|
92
|
+
//#endregion
|
|
93
|
+
export { hiddenProperties, hiddenXObject };
|
|
@@ -13,14 +13,17 @@ import { FlowDoc } from '../core/ir/flow.js';
|
|
|
13
13
|
* opens permissions-only encryption.
|
|
14
14
|
* @param filters Decoders for `/Filter` names this reader does not implement
|
|
15
15
|
* (§7.4); see {@link StreamFilters}.
|
|
16
|
-
* @param layout `'
|
|
16
|
+
* @param layout `'auto'` (the default) lets the FILE decide: a page that is
|
|
17
|
+
* mostly marks is reproduced, one that is mostly lines is re-set,
|
|
18
|
+
* and the reader records which it chose. `'flow'` reads a
|
|
19
|
+
* re-flowable document out of the page —
|
|
17
20
|
* paragraphs and tables in reading order, from the structure
|
|
18
21
|
* tree where there is one. `'positional'` keeps the page: every
|
|
19
22
|
* line stands where its glyphs do, beside the artwork, which is
|
|
20
23
|
* what a form or a drawing needs and what a paragraph cannot be.
|
|
21
24
|
* @returns The reconstructed FlowDoc and its accumulated {@link Loss} report.
|
|
22
25
|
*/
|
|
23
|
-
export declare function readPdf(bytes: Uint8Array, password?: string, layout?: 'flow' | 'positional', filters?: StreamFilters): ReadResult<FlowDoc>;
|
|
26
|
+
export declare function readPdf(bytes: Uint8Array, password?: string, layout?: 'flow' | 'positional' | 'auto', filters?: StreamFilters): ReadResult<FlowDoc>;
|
|
24
27
|
/**
|
|
25
28
|
* The `pdfReader` adapter: a {@link DocumentReader} that sniffs the `%PDF-`
|
|
26
29
|
* header and parses the bytes into a {@link FlowDoc} (E-PDF EP5).
|