reamkit 1.26.0 → 1.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -8
- package/dist/esm/core/converter/ream.d.ts +0 -9
- package/dist/esm/core/converter/ream.js +0 -1
- package/dist/esm/core/document-model/types.d.ts +5 -3
- package/dist/esm/core/font/index.d.ts +1 -0
- package/dist/esm/core/font/ligatures.d.ts +17 -0
- package/dist/esm/core/font/ligatures.js +49 -0
- package/dist/esm/core/font/ttf-parser.d.ts +2 -1
- package/dist/esm/core/font/ttf-parser.js +11 -2
- package/dist/esm/core/fonts/remote-fonts.d.ts +1 -1
- package/dist/esm/core/fonts/remote-fonts.js +47 -2
- package/dist/esm/core/fonts/scripts.js +10 -5
- package/dist/esm/excel/header-footer.js +53 -8
- package/dist/esm/layout/styled-layout.js +53 -3
- package/dist/esm/pdf/cid-font.js +35 -8
- package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
- package/dist/esm/pdf-reader/annot-draw.js +487 -0
- package/dist/esm/pdf-reader/annots.d.ts +0 -12
- package/dist/esm/pdf-reader/annots.js +30 -4
- package/dist/esm/pdf-reader/ccitt.d.ts +20 -3
- package/dist/esm/pdf-reader/ccitt.js +102 -6
- package/dist/esm/pdf-reader/cff-outline.d.ts +36 -0
- package/dist/esm/pdf-reader/cff-outline.js +1122 -0
- package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
- package/dist/esm/pdf-reader/cie-color.js +112 -0
- package/dist/esm/pdf-reader/cmap.js +5 -2
- package/dist/esm/pdf-reader/content.d.ts +123 -3
- package/dist/esm/pdf-reader/content.js +232 -54
- package/dist/esm/pdf-reader/dingbats.d.ts +11 -0
- package/dist/esm/pdf-reader/dingbats.js +1033 -0
- package/dist/esm/pdf-reader/display.d.ts +1 -1
- package/dist/esm/pdf-reader/display.js +61 -6
- package/dist/esm/pdf-reader/document.d.ts +6 -0
- package/dist/esm/pdf-reader/document.js +55 -10
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +23 -3
- package/dist/esm/pdf-reader/embedded-fonts.js +36 -4
- package/dist/esm/pdf-reader/encodings.d.ts +25 -0
- package/dist/esm/pdf-reader/encodings.js +110 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +19 -5
- package/dist/esm/pdf-reader/flow-build.js +111 -11
- package/dist/esm/pdf-reader/font.js +403 -21
- package/dist/esm/pdf-reader/function.d.ts +16 -0
- package/dist/esm/pdf-reader/function.js +414 -0
- package/dist/esm/pdf-reader/glyf-outline.d.ts +43 -0
- package/dist/esm/pdf-reader/glyf-outline.js +351 -0
- package/dist/esm/pdf-reader/glyph-names.js +20 -0
- package/dist/esm/pdf-reader/icc.d.ts +10 -0
- package/dist/esm/pdf-reader/icc.js +210 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +9 -4
- package/dist/esm/pdf-reader/image-decode.js +274 -72
- package/dist/esm/pdf-reader/images.d.ts +18 -0
- package/dist/esm/pdf-reader/images.js +154 -11
- package/dist/esm/pdf-reader/jbig2.d.ts +23 -0
- package/dist/esm/pdf-reader/jbig2.js +126 -32
- package/dist/esm/pdf-reader/layout.d.ts +32 -0
- package/dist/esm/pdf-reader/layout.js +1316 -64
- package/dist/esm/pdf-reader/lexer.d.ts +2 -0
- package/dist/esm/pdf-reader/lexer.js +4 -0
- package/dist/esm/pdf-reader/math-rows.d.ts +23 -0
- package/dist/esm/pdf-reader/math-rows.js +198 -0
- package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
- package/dist/esm/pdf-reader/optional-content.js +93 -0
- package/dist/esm/pdf-reader/predefined-cmap.d.ts +21 -0
- package/dist/esm/pdf-reader/predefined-cmap.js +102 -0
- package/dist/esm/pdf-reader/reader.d.ts +6 -6
- package/dist/esm/pdf-reader/reader.js +102 -16
- package/dist/esm/pdf-reader/shading.d.ts +139 -8
- package/dist/esm/pdf-reader/shading.js +309 -39
- package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
- package/dist/esm/pdf-reader/standard-metrics.js +18 -0
- package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
- package/dist/esm/pdf-reader/standard-widths.js +62 -0
- package/dist/esm/pdf-reader/stream-filters.d.ts +6 -0
- package/dist/esm/pdf-reader/stream-filters.js +67 -0
- package/dist/esm/pdf-reader/tagged.js +204 -32
- package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
- package/dist/esm/pdf-reader/text-rules.js +185 -0
- package/dist/esm/pdf-reader/text.js +157 -4
- package/dist/esm/pdf-reader/type1-outline.d.ts +21 -0
- package/dist/esm/pdf-reader/type1-outline.js +576 -0
- package/dist/esm/pdf-reader/vector.d.ts +5 -0
- package/dist/esm/pdf-reader/vector.js +79 -27
- package/dist/esm/word/document-parser.js +4 -1
- package/dist/esm/word/docx-writer.js +74 -11
- package/package.json +1 -1
|
@@ -45,6 +45,8 @@ export declare class Lexer {
|
|
|
45
45
|
constructor(buf: Uint8Array, pos?: number);
|
|
46
46
|
/** The length of the underlying byte buffer. */
|
|
47
47
|
get length(): number;
|
|
48
|
+
/** The bytes between two offsets, as a view onto the buffer. */
|
|
49
|
+
slice(from: number, to: number): Uint8Array;
|
|
48
50
|
/** The byte at index `i`, or −1 when out of range. */
|
|
49
51
|
byteAt(i: number): number;
|
|
50
52
|
/** §7.2.3 — skip whitespace and `%`-to-end-of-line comments. */
|
|
@@ -34,6 +34,10 @@ var Lexer = class {
|
|
|
34
34
|
get length() {
|
|
35
35
|
return this.buf.length;
|
|
36
36
|
}
|
|
37
|
+
/** The bytes between two offsets, as a view onto the buffer. */
|
|
38
|
+
slice(from, to) {
|
|
39
|
+
return this.buf.subarray(Math.max(0, from), Math.min(this.buf.length, Math.max(from, to)));
|
|
40
|
+
}
|
|
37
41
|
/** The byte at index `i`, or −1 when out of range. */
|
|
38
42
|
byteAt(i) {
|
|
39
43
|
return i >= 0 && i < this.buf.length ? this.buf[i] : -1;
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import { MathNode } from '../core/document-model/index.js';
|
|
2
|
+
import { TextRun } from './content.js';
|
|
3
|
+
/** A matrix found on the page: what to draw, where, and which runs it used. */
|
|
4
|
+
export interface MathBlock {
|
|
5
|
+
/** The math object — a row of delimiters and whatever stands between them. */
|
|
6
|
+
readonly math: MathNode;
|
|
7
|
+
/** The top baseline it was drawn on, for ordering the page's blocks. */
|
|
8
|
+
readonly top: number;
|
|
9
|
+
/** Where it stands across the page. */
|
|
10
|
+
readonly x: number;
|
|
11
|
+
readonly width: number;
|
|
12
|
+
/** The runs it consumed, which the prose reading must not read again. */
|
|
13
|
+
readonly used: ReadonlySet<TextRun>;
|
|
14
|
+
}
|
|
15
|
+
/**
|
|
16
|
+
* The matrices a page's rows hold (§22.1.2.68 `m:m`).
|
|
17
|
+
*
|
|
18
|
+
* @param runs The column's runs.
|
|
19
|
+
* @param fontSize The text's own size, which says how far apart the lines of
|
|
20
|
+
* PROSE stand — a display's sub-lines stand closer.
|
|
21
|
+
* @returns One entry per matrix found, in page order.
|
|
22
|
+
*/
|
|
23
|
+
export declare function matrixBlocks(runs: ReadonlyArray<TextRun>, fontSize: number): Array<MathBlock>;
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
//#region src/pdf-reader/math-rows.ts
|
|
2
|
+
/** The brackets a matrix may be written in, opening → closing. */
|
|
3
|
+
var BRACKETS = new Map([
|
|
4
|
+
["(", ")"],
|
|
5
|
+
["[", "]"],
|
|
6
|
+
["{", "}"],
|
|
7
|
+
["⟨", "⟩"],
|
|
8
|
+
["|", "|"]
|
|
9
|
+
]);
|
|
10
|
+
/**
|
|
11
|
+
* The matrices a page's rows hold (§22.1.2.68 `m:m`).
|
|
12
|
+
*
|
|
13
|
+
* @param runs The column's runs.
|
|
14
|
+
* @param fontSize The text's own size, which says how far apart the lines of
|
|
15
|
+
* PROSE stand — a display's sub-lines stand closer.
|
|
16
|
+
* @returns One entry per matrix found, in page order.
|
|
17
|
+
*/
|
|
18
|
+
function matrixBlocks(runs, fontSize) {
|
|
19
|
+
const out = [];
|
|
20
|
+
for (const cluster of clusters(baselines(runs, fontSize), fontSize)) {
|
|
21
|
+
const found = matrixFrom(cluster);
|
|
22
|
+
if (found) out.push(found);
|
|
23
|
+
}
|
|
24
|
+
return out;
|
|
25
|
+
}
|
|
26
|
+
/** The baseline a row of runs stands on. */
|
|
27
|
+
var baselineOf = (row) => Math.max(...row.map((r) => r.y));
|
|
28
|
+
/**
|
|
29
|
+
* The runs grouped by the baseline they were DRAWN on, top to bottom.
|
|
30
|
+
*
|
|
31
|
+
* Tighter than the reading's own rows, which allow a line half an em of drift
|
|
32
|
+
* so that a footnote mark stays with its word: the sub-lines of a display stand
|
|
33
|
+
* six points apart in a ten-point face, and grouped that loosely
|
|
34
|
+
* bug1997343.pdf's brackets joined the row of numbers under them.
|
|
35
|
+
*/
|
|
36
|
+
function baselines(runs, fontSize) {
|
|
37
|
+
const tol = Math.max(1, fontSize * SAME_BASELINE);
|
|
38
|
+
const out = [];
|
|
39
|
+
for (const run of [...runs].sort((a, b) => b.y - a.y || a.x - b.x)) {
|
|
40
|
+
const last = out[out.length - 1];
|
|
41
|
+
if (last && Math.abs(baselineOf(last) - run.y) <= tol) last.push(run);
|
|
42
|
+
else out.push([run]);
|
|
43
|
+
}
|
|
44
|
+
return out;
|
|
45
|
+
}
|
|
46
|
+
/** How far off a baseline, in ems, a run may sit and still be ON it. */
|
|
47
|
+
var SAME_BASELINE = .15;
|
|
48
|
+
/**
|
|
49
|
+
* The runs of a page grouped into the SUB-LINES of one display: rows standing
|
|
50
|
+
* closer together than a line of prose does, three or more of them (two rows
|
|
51
|
+
* and the brackets between).
|
|
52
|
+
*/
|
|
53
|
+
function clusters(rows, fontSize) {
|
|
54
|
+
const out = [];
|
|
55
|
+
let run = [];
|
|
56
|
+
for (const row of rows) {
|
|
57
|
+
const last = run[run.length - 1];
|
|
58
|
+
if (last && baselineOf(last) - baselineOf(row) >= fontSize * SUBLINE_GAP) {
|
|
59
|
+
if (run.length >= MIN_SUBLINES) out.push(run);
|
|
60
|
+
run = [];
|
|
61
|
+
}
|
|
62
|
+
run.push(row);
|
|
63
|
+
}
|
|
64
|
+
if (run.length >= MIN_SUBLINES) out.push(run);
|
|
65
|
+
return out;
|
|
66
|
+
}
|
|
67
|
+
/** How far apart, in ems, two baselines have to stand to be separate LINES. */
|
|
68
|
+
var SUBLINE_GAP = .9;
|
|
69
|
+
/** A matrix is a row of cells above another, and the brackets between them. */
|
|
70
|
+
var MIN_SUBLINES = 3;
|
|
71
|
+
/** Build the matrix a cluster of sub-lines holds, or nothing where it holds none. */
|
|
72
|
+
function matrixFrom(cluster) {
|
|
73
|
+
const middle = cluster.find((row) => pairsOf(row).length > 0);
|
|
74
|
+
if (!middle) return void 0;
|
|
75
|
+
const pairs = pairsOf(middle);
|
|
76
|
+
const cells = cluster.filter((row) => row !== middle);
|
|
77
|
+
if (cells.length === 0) return void 0;
|
|
78
|
+
const children = [];
|
|
79
|
+
let used = /* @__PURE__ */ new Set();
|
|
80
|
+
let at = -Infinity;
|
|
81
|
+
for (const pair of pairs) {
|
|
82
|
+
for (const run of middle) if (run.x >= at && run.endX <= pair.open.x && run.text.trim().length > 0) {
|
|
83
|
+
children.push({
|
|
84
|
+
type: "run",
|
|
85
|
+
text: spaced(run.text)
|
|
86
|
+
});
|
|
87
|
+
used.add(run);
|
|
88
|
+
}
|
|
89
|
+
const inside = matrixInside(cluster, middle, pair);
|
|
90
|
+
if (!inside) return void 0;
|
|
91
|
+
children.push({
|
|
92
|
+
type: "delimiter",
|
|
93
|
+
begChr: pair.open.text.trim(),
|
|
94
|
+
endChr: pair.close.text.trim(),
|
|
95
|
+
children: [inside.matrix]
|
|
96
|
+
});
|
|
97
|
+
used = new Set([
|
|
98
|
+
...used,
|
|
99
|
+
...inside.used,
|
|
100
|
+
pair.open,
|
|
101
|
+
pair.close
|
|
102
|
+
]);
|
|
103
|
+
at = pair.close.endX;
|
|
104
|
+
}
|
|
105
|
+
for (const run of middle) if (run.x >= at && run.text.trim().length > 0) {
|
|
106
|
+
children.push({
|
|
107
|
+
type: "run",
|
|
108
|
+
text: spaced(run.text)
|
|
109
|
+
});
|
|
110
|
+
used.add(run);
|
|
111
|
+
}
|
|
112
|
+
if (children.length === 0) return void 0;
|
|
113
|
+
for (const row of cluster) for (const run of row) if (!used.has(run)) return void 0;
|
|
114
|
+
const all = [...used];
|
|
115
|
+
return {
|
|
116
|
+
math: {
|
|
117
|
+
type: "row",
|
|
118
|
+
children
|
|
119
|
+
},
|
|
120
|
+
top: baselineOf(cells[0] ?? middle),
|
|
121
|
+
x: Math.min(...all.map((r) => r.x)),
|
|
122
|
+
width: Math.max(...all.map((r) => r.endX)) - Math.min(...all.map((r) => r.x)),
|
|
123
|
+
used
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
/**
|
|
127
|
+
* An operator standing between two matrices, with the air the page gave it.
|
|
128
|
+
* The file sets `(1 2 / 3 4)(1 1 / 0 1) = (1 3 / 3 7)` with the equals sign
|
|
129
|
+
* clear of both brackets; set tight against them it reads as one word.
|
|
130
|
+
*/
|
|
131
|
+
var spaced = (text) => ` ${text.trim()} `;
|
|
132
|
+
/** The bracket pairs a row holds, left to right and never nested. */
|
|
133
|
+
function pairsOf(row) {
|
|
134
|
+
const out = [];
|
|
135
|
+
let open;
|
|
136
|
+
for (const run of [...row].sort((a, b) => a.x - b.x)) {
|
|
137
|
+
const text = run.text.trim();
|
|
138
|
+
if (open === void 0) {
|
|
139
|
+
if (BRACKETS.has(text)) open = run;
|
|
140
|
+
continue;
|
|
141
|
+
}
|
|
142
|
+
if (text === BRACKETS.get(open.text.trim())) {
|
|
143
|
+
out.push({
|
|
144
|
+
open,
|
|
145
|
+
close: run
|
|
146
|
+
});
|
|
147
|
+
open = void 0;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return out;
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* The grid inside one bracket pair: every sub-line's runs that fall between the
|
|
154
|
+
* brackets, clustered into columns by where they stand.
|
|
155
|
+
*/
|
|
156
|
+
function matrixInside(cluster, middle, pair) {
|
|
157
|
+
const used = /* @__PURE__ */ new Set();
|
|
158
|
+
const rows = [];
|
|
159
|
+
for (const row of cluster) {
|
|
160
|
+
const inside = row.filter((run) => run !== pair.open && run !== pair.close && run.x >= pair.open.x && run.endX <= pair.close.endX && run.text.trim().length > 0);
|
|
161
|
+
if (inside.length === 0) continue;
|
|
162
|
+
if (row === middle && inside.some((run) => BRACKETS.has(run.text.trim()))) return void 0;
|
|
163
|
+
inside.sort((a, b) => a.x - b.x);
|
|
164
|
+
for (const run of inside) used.add(run);
|
|
165
|
+
rows.push(inside);
|
|
166
|
+
}
|
|
167
|
+
if (rows.length < 2) return void 0;
|
|
168
|
+
const size = Math.max(...rows.flat().map((r) => r.fontSizePt || 10));
|
|
169
|
+
const centres = [];
|
|
170
|
+
const columnOf = (run) => {
|
|
171
|
+
const centre = (run.x + run.endX) / 2;
|
|
172
|
+
const found = centres.findIndex((c) => Math.abs(c - centre) <= size / 2);
|
|
173
|
+
if (found >= 0) return found;
|
|
174
|
+
centres.push(centre);
|
|
175
|
+
return centres.length - 1;
|
|
176
|
+
};
|
|
177
|
+
const grid = rows.map((row) => {
|
|
178
|
+
const cells = [];
|
|
179
|
+
for (const run of row) {
|
|
180
|
+
const at = columnOf(run);
|
|
181
|
+
(cells[at] ??= []).push(run.text.trim());
|
|
182
|
+
}
|
|
183
|
+
return cells;
|
|
184
|
+
});
|
|
185
|
+
const width = Math.max(...grid.map((row) => row.length));
|
|
186
|
+
return {
|
|
187
|
+
matrix: {
|
|
188
|
+
type: "matrix",
|
|
189
|
+
rows: grid.map((row) => Array.from({ length: width }, (_, i) => ({
|
|
190
|
+
type: "run",
|
|
191
|
+
text: (row[i] ?? []).join("")
|
|
192
|
+
})))
|
|
193
|
+
},
|
|
194
|
+
used
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
//#endregion
|
|
198
|
+
export { matrixBlocks };
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { PdfDict, PdfStream, PdfValue } from '../pdf/objects.js';
|
|
2
|
+
import { PdfFile } from './document.js';
|
|
3
|
+
/**
|
|
4
|
+
* The optional-content groups the file's DEFAULT configuration turns off
|
|
5
|
+
* (§8.11.4.3).
|
|
6
|
+
*
|
|
7
|
+
* `/BaseState` says what an unlisted group does — `/ON` unless the file says
|
|
8
|
+
* `/OFF` — and the `/ON` and `/OFF` arrays name the exceptions, `/OFF` last.
|
|
9
|
+
*
|
|
10
|
+
* @param file The document.
|
|
11
|
+
* @returns The resolved OCG dictionaries that are hidden.
|
|
12
|
+
*/
|
|
13
|
+
export declare function hiddenGroups(file: PdfFile): ReadonlySet<PdfValue>;
|
|
14
|
+
/**
|
|
15
|
+
* Whether an `/OC` entry names something the page does not show.
|
|
16
|
+
*
|
|
17
|
+
* The entry is either a group itself or an `/OCMD` — a membership dictionary
|
|
18
|
+
* naming several groups and a `/P` policy over them (§8.11.2.3). `AnyOn` is the
|
|
19
|
+
* default and the common case: the content shows if any of its groups does.
|
|
20
|
+
*
|
|
21
|
+
* @param file The document.
|
|
22
|
+
* @param oc The `/OC` value, unresolved.
|
|
23
|
+
* @returns `true` where the content it guards is hidden.
|
|
24
|
+
*/
|
|
25
|
+
export declare function hiddenByOc(file: PdfFile, oc: PdfValue | undefined): boolean;
|
|
26
|
+
/**
|
|
27
|
+
* The `/Properties` names a page's content may name in `/OC … BDC` that are
|
|
28
|
+
* hidden (§8.11.3.2).
|
|
29
|
+
*
|
|
30
|
+
* @param file The document.
|
|
31
|
+
* @param resources The resource dictionary in force.
|
|
32
|
+
* @returns Name → hidden, for the names that ARE hidden.
|
|
33
|
+
*/
|
|
34
|
+
export declare function hiddenProperties(file: PdfFile, resources: PdfDict | undefined): Set<string>;
|
|
35
|
+
/** Whether an XObject carries an `/OC` that hides it (§8.11.3.1). */
|
|
36
|
+
export declare function hiddenXObject(file: PdfFile, stream: PdfStream): boolean;
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import { PDF_NULL, PdfName } from "../pdf/objects.js";
|
|
2
|
+
//#region src/pdf-reader/optional-content.ts
|
|
3
|
+
/** The document's own answer to "is this group shown?", worked out once. */
|
|
4
|
+
var cache = /* @__PURE__ */ new WeakMap();
|
|
5
|
+
/** An `/OCMD` naming more groups than this is not read: it is not a document. */
|
|
6
|
+
var MAX_GROUPS = 4096;
|
|
7
|
+
/**
|
|
8
|
+
* The optional-content groups the file's DEFAULT configuration turns off
|
|
9
|
+
* (§8.11.4.3).
|
|
10
|
+
*
|
|
11
|
+
* `/BaseState` says what an unlisted group does — `/ON` unless the file says
|
|
12
|
+
* `/OFF` — and the `/ON` and `/OFF` arrays name the exceptions, `/OFF` last.
|
|
13
|
+
*
|
|
14
|
+
* @param file The document.
|
|
15
|
+
* @returns The resolved OCG dictionaries that are hidden.
|
|
16
|
+
*/
|
|
17
|
+
function hiddenGroups(file) {
|
|
18
|
+
const had = cache.get(file);
|
|
19
|
+
if (had) return had;
|
|
20
|
+
const hidden = /* @__PURE__ */ new Set();
|
|
21
|
+
const props = file.get(file.catalog, "OCProperties");
|
|
22
|
+
const config = props instanceof Map ? file.get(props, "D") : void 0;
|
|
23
|
+
if (config instanceof Map) {
|
|
24
|
+
const base = file.get(config, "BaseState");
|
|
25
|
+
if (base instanceof PdfName && base.value === "OFF") {
|
|
26
|
+
const all = file.get(props instanceof Map ? props : /* @__PURE__ */ new Map(), "OCGs");
|
|
27
|
+
for (const g of asArray(file, all)) hidden.add(g);
|
|
28
|
+
}
|
|
29
|
+
for (const g of asArray(file, file.get(config, "ON"))) hidden.delete(g);
|
|
30
|
+
for (const g of asArray(file, file.get(config, "OFF"))) hidden.add(g);
|
|
31
|
+
}
|
|
32
|
+
cache.set(file, hidden);
|
|
33
|
+
return hidden;
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Whether an `/OC` entry names something the page does not show.
|
|
37
|
+
*
|
|
38
|
+
* The entry is either a group itself or an `/OCMD` — a membership dictionary
|
|
39
|
+
* naming several groups and a `/P` policy over them (§8.11.2.3). `AnyOn` is the
|
|
40
|
+
* default and the common case: the content shows if any of its groups does.
|
|
41
|
+
*
|
|
42
|
+
* @param file The document.
|
|
43
|
+
* @param oc The `/OC` value, unresolved.
|
|
44
|
+
* @returns `true` where the content it guards is hidden.
|
|
45
|
+
*/
|
|
46
|
+
function hiddenByOc(file, oc) {
|
|
47
|
+
if (oc === void 0) return false;
|
|
48
|
+
const hidden = hiddenGroups(file);
|
|
49
|
+
const resolved = file.resolve(oc);
|
|
50
|
+
if (!(resolved instanceof Map)) return false;
|
|
51
|
+
const type = file.get(resolved, "Type");
|
|
52
|
+
if (!(type instanceof PdfName) || type.value !== "OCMD") return hidden.has(resolved);
|
|
53
|
+
const groups = asArray(file, file.get(resolved, "OCGs"));
|
|
54
|
+
const single = file.resolve(resolved.get("OCGs") ?? PDF_NULL);
|
|
55
|
+
const list = groups.length > 0 ? groups : single instanceof Map ? [single] : [];
|
|
56
|
+
if (list.length === 0) return false;
|
|
57
|
+
const on = list.filter((g) => !hidden.has(g)).length;
|
|
58
|
+
const policy = file.get(resolved, "P");
|
|
59
|
+
switch (policy instanceof PdfName ? policy.value : "AnyOn") {
|
|
60
|
+
case "AllOn": return on < list.length;
|
|
61
|
+
case "AnyOff": return on === list.length;
|
|
62
|
+
case "AllOff": return on > 0;
|
|
63
|
+
default: return on === 0;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* The `/Properties` names a page's content may name in `/OC … BDC` that are
|
|
68
|
+
* hidden (§8.11.3.2).
|
|
69
|
+
*
|
|
70
|
+
* @param file The document.
|
|
71
|
+
* @param resources The resource dictionary in force.
|
|
72
|
+
* @returns Name → hidden, for the names that ARE hidden.
|
|
73
|
+
*/
|
|
74
|
+
function hiddenProperties(file, resources) {
|
|
75
|
+
const out = /* @__PURE__ */ new Set();
|
|
76
|
+
if (!resources) return out;
|
|
77
|
+
const props = file.get(resources, "Properties");
|
|
78
|
+
if (!(props instanceof Map)) return out;
|
|
79
|
+
for (const [name, value] of props) if (hiddenByOc(file, value)) out.add(name);
|
|
80
|
+
return out;
|
|
81
|
+
}
|
|
82
|
+
/** Whether an XObject carries an `/OC` that hides it (§8.11.3.1). */
|
|
83
|
+
function hiddenXObject(file, stream) {
|
|
84
|
+
return hiddenByOc(file, stream.dict.get("OC"));
|
|
85
|
+
}
|
|
86
|
+
/** An array entry, resolved; a missing or malformed one comes back empty. */
|
|
87
|
+
function asArray(file, value) {
|
|
88
|
+
const r = value !== void 0 ? file.resolve(value) : void 0;
|
|
89
|
+
if (!Array.isArray(r)) return [];
|
|
90
|
+
return r.slice(0, MAX_GROUPS).map((v) => file.resolve(v));
|
|
91
|
+
}
|
|
92
|
+
//#endregion
|
|
93
|
+
export { hiddenProperties, hiddenXObject };
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/** How a named CMap breaks a string into codes, and what those codes mean. */
|
|
2
|
+
export interface PredefinedCMap {
|
|
3
|
+
/** The `TextDecoder` label the code bytes are in. */
|
|
4
|
+
readonly charset: string;
|
|
5
|
+
/** Whether a byte begins a TWO-byte code; anything else stands alone. */
|
|
6
|
+
readonly leadsPair: (byte: number) => boolean;
|
|
7
|
+
/** §9.7.5.2 — a `…-V` CMap sets its text down the page. */
|
|
8
|
+
readonly vertical: boolean;
|
|
9
|
+
}
|
|
10
|
+
/**
|
|
11
|
+
* What a predefined CMap name comes to.
|
|
12
|
+
*
|
|
13
|
+
* @param name The `/Encoding` name, as the font states it.
|
|
14
|
+
* @returns Its encoding and code structure, or `undefined` for `Identity` and
|
|
15
|
+
* for any name whose encoding this cannot decode.
|
|
16
|
+
*/
|
|
17
|
+
export declare function predefinedCMap(name: string): PredefinedCMap | undefined;
|
|
18
|
+
/** Split a shown string the way this CMap's codespace says (§9.7.6.2). */
|
|
19
|
+
export declare function splitPredefined(cmap: PredefinedCMap, bytes: Uint8Array): Array<number>;
|
|
20
|
+
/** The character one code stands for: its own bytes, read in the CMap's encoding. */
|
|
21
|
+
export declare function decodePredefined(cmap: PredefinedCMap, code: number): string;
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
//#region src/pdf-reader/predefined-cmap.ts
|
|
2
|
+
/** Shift-JIS: 81–9F and E0–FC lead a pair; 00–80 and A0–DF stand alone. */
|
|
3
|
+
var shiftJis = (b) => b >= 129 && b <= 159 || b >= 224 && b <= 252;
|
|
4
|
+
/** The EUC family: A1–FE lead a pair, and 8E/8F introduce one as well. */
|
|
5
|
+
var euc = (b) => b >= 161 || b === 142 || b === 143;
|
|
6
|
+
/** Big5, GBK and UHC: everything from 81 up leads a pair. */
|
|
7
|
+
var highLeads = (b) => b >= 129;
|
|
8
|
+
/** Two bytes always, big-endian — the UCS-2 and UTF-16 CMaps. */
|
|
9
|
+
var always = () => true;
|
|
10
|
+
/** The families, longest name first so `90ms-RKSJ` is not read as `RKSJ`. */
|
|
11
|
+
var FAMILIES = [
|
|
12
|
+
{
|
|
13
|
+
match: /RKSJ/u,
|
|
14
|
+
charset: "shift_jis",
|
|
15
|
+
leadsPair: shiftJis
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
match: /-EUC(-|$)|^EUC-/u,
|
|
19
|
+
charset: "euc-jp",
|
|
20
|
+
leadsPair: euc
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
match: /^(ETen|ETenms|B5pc|HKscs-B5|CNS-EUC)/u,
|
|
24
|
+
charset: "big5",
|
|
25
|
+
leadsPair: highLeads
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
match: /^(GBK|GBpc|GBK2K|GB-EUC|GBKp)/u,
|
|
29
|
+
charset: "gbk",
|
|
30
|
+
leadsPair: highLeads
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
match: /^(KSC|KSCms|KSCpc)/u,
|
|
34
|
+
charset: "euc-kr",
|
|
35
|
+
leadsPair: highLeads
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
match: /UCS2|UTF16/u,
|
|
39
|
+
charset: "utf-16be",
|
|
40
|
+
leadsPair: always
|
|
41
|
+
}
|
|
42
|
+
];
|
|
43
|
+
/**
|
|
44
|
+
* What a predefined CMap name comes to.
|
|
45
|
+
*
|
|
46
|
+
* @param name The `/Encoding` name, as the font states it.
|
|
47
|
+
* @returns Its encoding and code structure, or `undefined` for `Identity` and
|
|
48
|
+
* for any name whose encoding this cannot decode.
|
|
49
|
+
*/
|
|
50
|
+
function predefinedCMap(name) {
|
|
51
|
+
if (name.startsWith("Identity")) return void 0;
|
|
52
|
+
const vertical = name.endsWith("-V");
|
|
53
|
+
const family = FAMILIES.find((f) => f.match.test(name));
|
|
54
|
+
if (!family) return void 0;
|
|
55
|
+
if (!canDecode(family.charset)) return void 0;
|
|
56
|
+
return {
|
|
57
|
+
charset: family.charset,
|
|
58
|
+
leadsPair: family.leadsPair,
|
|
59
|
+
vertical
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
/** Split a shown string the way this CMap's codespace says (§9.7.6.2). */
|
|
63
|
+
function splitPredefined(cmap, bytes) {
|
|
64
|
+
const out = [];
|
|
65
|
+
for (let i = 0; i < bytes.length;) {
|
|
66
|
+
const b = bytes[i];
|
|
67
|
+
if (cmap.leadsPair(b) && i + 1 < bytes.length) {
|
|
68
|
+
out.push(b << 8 | bytes[i + 1]);
|
|
69
|
+
i += 2;
|
|
70
|
+
} else {
|
|
71
|
+
out.push(b);
|
|
72
|
+
i++;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return out;
|
|
76
|
+
}
|
|
77
|
+
/** The character one code stands for: its own bytes, read in the CMap's encoding. */
|
|
78
|
+
function decodePredefined(cmap, code) {
|
|
79
|
+
const bytes = code > 255 ? Uint8Array.from([code >> 8, code & 255]) : Uint8Array.from([code]);
|
|
80
|
+
try {
|
|
81
|
+
return new TextDecoder(cmap.charset, { fatal: false }).decode(bytes);
|
|
82
|
+
} catch {
|
|
83
|
+
return "";
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
/** Whether this runtime's `TextDecoder` carries the legacy encoding. */
|
|
87
|
+
function canDecode(charset) {
|
|
88
|
+
const had = supported.get(charset);
|
|
89
|
+
if (had !== void 0) return had;
|
|
90
|
+
let ok = false;
|
|
91
|
+
try {
|
|
92
|
+
new TextDecoder(charset);
|
|
93
|
+
ok = true;
|
|
94
|
+
} catch {
|
|
95
|
+
ok = false;
|
|
96
|
+
}
|
|
97
|
+
supported.set(charset, ok);
|
|
98
|
+
return ok;
|
|
99
|
+
}
|
|
100
|
+
var supported = /* @__PURE__ */ new Map();
|
|
101
|
+
//#endregion
|
|
102
|
+
export { decodePredefined, predefinedCMap, splitPredefined };
|
|
@@ -13,14 +13,14 @@ import { FlowDoc } from '../core/ir/flow.js';
|
|
|
13
13
|
* opens permissions-only encryption.
|
|
14
14
|
* @param filters Decoders for `/Filter` names this reader does not implement
|
|
15
15
|
* (§7.4); see {@link StreamFilters}.
|
|
16
|
-
* @param layout `'flow'` reads a re-flowable document out of the page —
|
|
17
|
-
* paragraphs and tables in reading order, from the structure
|
|
18
|
-
* tree where there is one. `'positional'` keeps the page: every
|
|
19
|
-
* line stands where its glyphs do, beside the artwork, which is
|
|
20
|
-
* what a form or a drawing needs and what a paragraph cannot be.
|
|
21
16
|
* @returns The reconstructed FlowDoc and its accumulated {@link Loss} report.
|
|
17
|
+
*
|
|
18
|
+
* Which of the two readings a file gets is the FILE's to decide and no
|
|
19
|
+
* caller's: a page that is mostly marks is reproduced where it stands, one
|
|
20
|
+
* that is mostly lines is re-set as a document, and the reader records which
|
|
21
|
+
* it chose. There is no override — see the note on {@link readingOf}.
|
|
22
22
|
*/
|
|
23
|
-
export declare function readPdf(bytes: Uint8Array, password?: string,
|
|
23
|
+
export declare function readPdf(bytes: Uint8Array, password?: string, filters?: StreamFilters): ReadResult<FlowDoc>;
|
|
24
24
|
/**
|
|
25
25
|
* The `pdfReader` adapter: a {@link DocumentReader} that sniffs the `%PDF-`
|
|
26
26
|
* header and parses the bytes into a {@link FlowDoc} (E-PDF EP5).
|
|
@@ -1,12 +1,32 @@
|
|
|
1
1
|
import { FEATURES } from "../core/ir/features.js";
|
|
2
2
|
import { PdfFile } from "./document.js";
|
|
3
|
+
import { extractPageText } from "./text.js";
|
|
4
|
+
import { collectPageVectors } from "./vector.js";
|
|
3
5
|
import { reconstructByLayout } from "./layout.js";
|
|
4
6
|
import { reconstructTaggedPdf } from "./tagged.js";
|
|
5
7
|
//#region src/pdf-reader/reader.ts
|
|
6
8
|
function sniffPdf(bytes) {
|
|
7
9
|
const limit = Math.min(bytes.length - 5, 1024);
|
|
8
10
|
for (let i = 0; i <= limit; i++) if (bytes[i] === 37 && bytes[i + 1] === 80 && bytes[i + 2] === 68 && bytes[i + 3] === 70 && bytes[i + 4] === 45) return true;
|
|
9
|
-
return
|
|
11
|
+
return endsLikePdf(bytes);
|
|
12
|
+
}
|
|
13
|
+
/**
|
|
14
|
+
* §7.5.5 — a file with no header at all, recognised by how it ENDS.
|
|
15
|
+
*
|
|
16
|
+
* bug1606566.pdf begins with the binary comment that normally FOLLOWS the
|
|
17
|
+
* header and has no `%PDF-` anywhere: the producer wrote the second line and
|
|
18
|
+
* not the first. Every reader takes it — poppler says "May not be a PDF file
|
|
19
|
+
* (continuing anyway)" and reads its one line of text — and we refused the file
|
|
20
|
+
* outright, which is the worst answer of the three.
|
|
21
|
+
*
|
|
22
|
+
* The two tokens that close every PDF and close nothing else: the cross-
|
|
23
|
+
* reference offset and the end-of-file marker, in that order, at the end.
|
|
24
|
+
*/
|
|
25
|
+
function endsLikePdf(bytes) {
|
|
26
|
+
const tail = bytes.subarray(Math.max(0, bytes.length - 2048));
|
|
27
|
+
const text = new TextDecoder("latin1").decode(tail);
|
|
28
|
+
const eof = text.lastIndexOf("%%EOF");
|
|
29
|
+
return eof >= 0 && text.lastIndexOf("startxref") < eof;
|
|
10
30
|
}
|
|
11
31
|
/**
|
|
12
32
|
* Parse PDF bytes into a {@link FlowDoc} (E-PDF EP5): reconstruct via the tagged
|
|
@@ -20,14 +40,14 @@ function sniffPdf(bytes) {
|
|
|
20
40
|
* opens permissions-only encryption.
|
|
21
41
|
* @param filters Decoders for `/Filter` names this reader does not implement
|
|
22
42
|
* (§7.4); see {@link StreamFilters}.
|
|
23
|
-
* @param layout `'flow'` reads a re-flowable document out of the page —
|
|
24
|
-
* paragraphs and tables in reading order, from the structure
|
|
25
|
-
* tree where there is one. `'positional'` keeps the page: every
|
|
26
|
-
* line stands where its glyphs do, beside the artwork, which is
|
|
27
|
-
* what a form or a drawing needs and what a paragraph cannot be.
|
|
28
43
|
* @returns The reconstructed FlowDoc and its accumulated {@link Loss} report.
|
|
44
|
+
*
|
|
45
|
+
* Which of the two readings a file gets is the FILE's to decide and no
|
|
46
|
+
* caller's: a page that is mostly marks is reproduced where it stands, one
|
|
47
|
+
* that is mostly lines is re-set as a document, and the reader records which
|
|
48
|
+
* it chose. There is no override — see the note on {@link readingOf}.
|
|
29
49
|
*/
|
|
30
|
-
function readPdf(bytes, password = "",
|
|
50
|
+
function readPdf(bytes, password = "", filters = {}) {
|
|
31
51
|
const file = PdfFile.parse(bytes, password, filters);
|
|
32
52
|
const losses = [];
|
|
33
53
|
if (file.encryptionUnsupported) losses.push({
|
|
@@ -40,25 +60,91 @@ function readPdf(bytes, password = "", layout = "flow", filters = {}) {
|
|
|
40
60
|
feature: FEATURES.text,
|
|
41
61
|
detail: `PDF stream filter /${name} is not supported; streams using it are unreadable`
|
|
42
62
|
});
|
|
43
|
-
const
|
|
44
|
-
|
|
45
|
-
|
|
63
|
+
const reading = readingOf(file);
|
|
64
|
+
losses.push({
|
|
65
|
+
severity: "degraded",
|
|
66
|
+
feature: FEATURES.text,
|
|
67
|
+
detail: reading === "positional" ? "PDF read as a PAGE (placed): its artwork outweighs its prose, so every line stands where its glyphs stand" : "PDF read as a DOCUMENT (flowing): its prose outweighs its artwork, so the words re-set and the artwork takes its turn in reading order"
|
|
68
|
+
});
|
|
69
|
+
const tagged = reading === "positional" ? void 0 : reconstructTaggedPdf(file);
|
|
70
|
+
const reconstruction = tagged ?? reconstructByLayout(file, reading);
|
|
71
|
+
if (!tagged && reading !== "positional") losses.push({
|
|
46
72
|
severity: "degraded",
|
|
47
73
|
feature: FEATURES.text,
|
|
48
74
|
detail: "untagged PDF — text and headings reconstructed heuristically from glyph positions; structure is approximate"
|
|
49
75
|
});
|
|
50
76
|
losses.push(...reconstruction.losses);
|
|
51
|
-
losses.push({
|
|
52
|
-
severity: "dropped",
|
|
53
|
-
feature: FEATURES.images,
|
|
54
|
-
detail: "PDF bare-shading (sh) vector regions are not reconstructed"
|
|
55
|
-
});
|
|
56
77
|
return {
|
|
57
78
|
doc: reconstruction.doc,
|
|
58
79
|
losses
|
|
59
80
|
};
|
|
60
81
|
}
|
|
61
82
|
/**
|
|
83
|
+
* Which reading a PDF asks for: a DOCUMENT to re-set, or a PAGE to reproduce.
|
|
84
|
+
*
|
|
85
|
+
* The two cannot be mixed. A flowing reading moves the words, and words that
|
|
86
|
+
* move cannot agree with rules that do not — anchored artwork over reflowed
|
|
87
|
+
* text puts every label on the wrong box. A placed reading keeps both where
|
|
88
|
+
* they were drawn, and reflows nothing.
|
|
89
|
+
*
|
|
90
|
+
* The file says which it is by what is on it. A form or a drawing is mostly
|
|
91
|
+
* MARKS — 160F-2019.pdf sets 28 numbered rows in 355 ruled boxes — and a paper
|
|
92
|
+
* is mostly LINES, with a rule or two between them. So the marks are counted
|
|
93
|
+
* against the lines, on the MEDIAN page rather than the worst, since one plan
|
|
94
|
+
* folded into a report does not make the report a plan.
|
|
95
|
+
*
|
|
96
|
+
* A threshold is a guess, and this one is stated rather than hidden: the reader
|
|
97
|
+
* records which reading it took and why, and the caller can name the other.
|
|
98
|
+
*/
|
|
99
|
+
function readingOf(file) {
|
|
100
|
+
/**
|
|
101
|
+
* Below this many marks a page is prose with decoration, whatever the ratio.
|
|
102
|
+
*
|
|
103
|
+
* It stood at twenty, and calgray.pdf is a five-by-four grid of grey swatches
|
|
104
|
+
* with a label in each: twenty boxes, of which the one painted white is not a
|
|
105
|
+
* mark anybody can see. Nineteen — one short — and all three pages of it were
|
|
106
|
+
* read as prose, the labels of each row run together into a line and the
|
|
107
|
+
* sheet spilling onto a second page. Nineteen boxes in a grid are not a page
|
|
108
|
+
* of prose with a rule under its heading.
|
|
109
|
+
*/
|
|
110
|
+
const ENOUGH_MARKS = 12;
|
|
111
|
+
/** Twice as many marks as lines is a page that is drawn rather than written. */
|
|
112
|
+
const DRAWN = 2;
|
|
113
|
+
/** Below this many runs, an angle is a stamp or a watermark and not the page. */
|
|
114
|
+
const ENOUGH_TURNED = 8;
|
|
115
|
+
const ratios = [];
|
|
116
|
+
for (const page of file.pages()) {
|
|
117
|
+
let marks = 0;
|
|
118
|
+
let lines = 0;
|
|
119
|
+
let turned = 0;
|
|
120
|
+
let runs = 0;
|
|
121
|
+
try {
|
|
122
|
+
marks = collectPageVectors(file, page, []).vectors.length;
|
|
123
|
+
const ys = /* @__PURE__ */ new Set();
|
|
124
|
+
for (const run of extractPageText(file, page)) {
|
|
125
|
+
ys.add(Math.round(run.y));
|
|
126
|
+
runs++;
|
|
127
|
+
if (run.angleDeg !== void 0) turned++;
|
|
128
|
+
}
|
|
129
|
+
lines = ys.size;
|
|
130
|
+
} catch {
|
|
131
|
+
continue;
|
|
132
|
+
}
|
|
133
|
+
if (runs >= ENOUGH_TURNED && turned > runs * .5) {
|
|
134
|
+
ratios.push(DRAWN);
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
if (marks < ENOUGH_MARKS) {
|
|
138
|
+
ratios.push(0);
|
|
139
|
+
continue;
|
|
140
|
+
}
|
|
141
|
+
ratios.push(lines > 0 ? marks / lines : DRAWN);
|
|
142
|
+
}
|
|
143
|
+
if (ratios.length === 0) return "flow";
|
|
144
|
+
const sorted = [...ratios].sort((a, b) => a - b);
|
|
145
|
+
return (sorted[Math.floor(sorted.length / 2)] ?? 0) >= DRAWN ? "positional" : "flow";
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
62
148
|
* The `pdfReader` adapter: a {@link DocumentReader} that sniffs the `%PDF-`
|
|
63
149
|
* header and parses the bytes into a {@link FlowDoc} (E-PDF EP5).
|
|
64
150
|
*/
|
|
@@ -72,7 +158,7 @@ var pdfReader = {
|
|
|
72
158
|
FEATURES.images
|
|
73
159
|
]),
|
|
74
160
|
sniff: sniffPdf,
|
|
75
|
-
read: (bytes, opts) => readPdf(bytes, typeof opts?.password === "string" ? opts.password : "",
|
|
161
|
+
read: (bytes, opts) => readPdf(bytes, typeof opts?.password === "string" ? opts.password : "", isFilters(opts?.filters) ? opts.filters : {})
|
|
76
162
|
};
|
|
77
163
|
/** A caller's `filters` option, when it is the shape the reader can use. */
|
|
78
164
|
function isFilters(value) {
|