scoutline 0.19.7 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/command-invocation.d.ts.map +1 -1
- package/dist/command-invocation.js +5 -0
- package/dist/command-invocation.js.map +1 -1
- package/dist/commands/archive.d.ts +111 -0
- package/dist/commands/archive.d.ts.map +1 -0
- package/dist/commands/archive.js +426 -0
- package/dist/commands/archive.js.map +1 -0
- package/dist/commands/doctor.d.ts +8 -0
- package/dist/commands/doctor.d.ts.map +1 -1
- package/dist/commands/doctor.js +39 -4
- package/dist/commands/doctor.js.map +1 -1
- package/dist/commands/fetch.d.ts +75 -0
- package/dist/commands/fetch.d.ts.map +1 -0
- package/dist/commands/fetch.js +592 -0
- package/dist/commands/fetch.js.map +1 -0
- package/dist/commands/read.d.ts.map +1 -1
- package/dist/commands/read.js +14 -0
- package/dist/commands/read.js.map +1 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +92 -7
- package/dist/index.js.map +1 -1
- package/dist/lib/artifacts.d.ts +5 -1
- package/dist/lib/artifacts.d.ts.map +1 -1
- package/dist/lib/artifacts.js +7 -2
- package/dist/lib/artifacts.js.map +1 -1
- package/dist/lib/async-file-lock.d.ts +2 -2
- package/dist/lib/async-file-lock.d.ts.map +1 -1
- package/dist/lib/async-file-lock.js +7 -4
- package/dist/lib/async-file-lock.js.map +1 -1
- package/dist/lib/pdf.d.ts +22 -0
- package/dist/lib/pdf.d.ts.map +1 -0
- package/dist/lib/pdf.js +1213 -0
- package/dist/lib/pdf.js.map +1 -0
- package/dist/lib/tty.d.ts.map +1 -1
- package/dist/lib/tty.js +10 -1
- package/dist/lib/tty.js.map +1 -1
- package/package.json +1 -1
package/dist/lib/pdf.js
ADDED
|
@@ -0,0 +1,1213 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* PDF text extraction and structural repair module (ADR-0006).
|
|
3
|
+
*
|
|
4
|
+
* Hybrid architecture:
|
|
5
|
+
* 1. Opportunistic system delegation to `pdftotext` and `qpdf` when installed.
|
|
6
|
+
* 2. Self-contained pure-Node fallback using zlib stream inflation and
|
|
7
|
+
* PDF text operator extraction (BT/ET, Tj, TJ, Td, hex strings) so Scoutline
|
|
8
|
+
* runs without mandatory system dependencies.
|
|
9
|
+
*/
|
|
10
|
+
import * as zlib from "node:zlib";
|
|
11
|
+
import { spawn } from "node:child_process";
|
|
12
|
+
/**
|
|
13
|
+
* Check whether a buffer begins with a valid PDF header (%PDF-).
|
|
14
|
+
*/
|
|
15
|
+
export function isPdfBuffer(buffer) {
|
|
16
|
+
if (!buffer || buffer.length < 5)
|
|
17
|
+
return false;
|
|
18
|
+
const header = buffer.subarray(0, 1024).toString("latin1").trimStart();
|
|
19
|
+
return header.startsWith("%PDF-");
|
|
20
|
+
}
|
|
21
|
+
/**
|
|
22
|
+
* Decode a PDF literal string (...), handling escape sequences.
|
|
23
|
+
*/
|
|
24
|
+
function decodePdfLiteralString(str) {
|
|
25
|
+
let result = "";
|
|
26
|
+
let i = 0;
|
|
27
|
+
while (i < str.length) {
|
|
28
|
+
if (str[i] === "\\") {
|
|
29
|
+
i++;
|
|
30
|
+
if (i >= str.length)
|
|
31
|
+
break;
|
|
32
|
+
const ch = str[i];
|
|
33
|
+
if (ch === "n") {
|
|
34
|
+
result += "\n";
|
|
35
|
+
i++;
|
|
36
|
+
}
|
|
37
|
+
else if (ch === "r") {
|
|
38
|
+
result += "\r";
|
|
39
|
+
i++;
|
|
40
|
+
}
|
|
41
|
+
else if (ch === "t") {
|
|
42
|
+
result += "\t";
|
|
43
|
+
i++;
|
|
44
|
+
}
|
|
45
|
+
else if (ch === "b") {
|
|
46
|
+
result += "\b";
|
|
47
|
+
i++;
|
|
48
|
+
}
|
|
49
|
+
else if (ch === "f") {
|
|
50
|
+
result += "\f";
|
|
51
|
+
i++;
|
|
52
|
+
}
|
|
53
|
+
else if (ch === "(" || ch === ")" || ch === "\\") {
|
|
54
|
+
result += ch;
|
|
55
|
+
i++;
|
|
56
|
+
}
|
|
57
|
+
else if (/[0-7]/.test(ch)) {
|
|
58
|
+
// Octal escape \ddd (1 to 3 octal digits)
|
|
59
|
+
let octalStr = ch;
|
|
60
|
+
i++;
|
|
61
|
+
if (i < str.length && /[0-7]/.test(str[i])) {
|
|
62
|
+
octalStr += str[i];
|
|
63
|
+
i++;
|
|
64
|
+
if (i < str.length && /[0-7]/.test(str[i])) {
|
|
65
|
+
octalStr += str[i];
|
|
66
|
+
i++;
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
result += String.fromCharCode(parseInt(octalStr, 8));
|
|
70
|
+
}
|
|
71
|
+
else if (ch === "\r" || ch === "\n") {
|
|
72
|
+
// Line continuation
|
|
73
|
+
if (ch === "\r" && i + 1 < str.length && str[i + 1] === "\n")
|
|
74
|
+
i++;
|
|
75
|
+
i++;
|
|
76
|
+
}
|
|
77
|
+
else {
|
|
78
|
+
result += ch;
|
|
79
|
+
i++;
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
else {
|
|
83
|
+
result += str[i];
|
|
84
|
+
i++;
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
return result;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Decode a PDF hex string <...>. Returns the decoded text plus a
|
|
91
|
+
* `unmappable` flag: hex bytes that are neither a UTF-16BE BOM sequence
|
|
92
|
+
* nor the 2-byte Identity-H ASCII pattern cannot be authoritatively
|
|
93
|
+
* decoded without the font's ToUnicode/encoding context, so callers
|
|
94
|
+
* treat a document containing them as low-confidence and prefer the
|
|
95
|
+
* external text layer (`pdftotext`) when available.
|
|
96
|
+
*/
|
|
97
|
+
function decodePdfHexString(hex) {
|
|
98
|
+
const clean = hex.replace(/\s+/g, "");
|
|
99
|
+
const padded = clean.length % 2 !== 0 ? clean + "0" : clean;
|
|
100
|
+
const bytes = [];
|
|
101
|
+
for (let i = 0; i < padded.length; i += 2) {
|
|
102
|
+
const byte = parseInt(padded.slice(i, i + 2), 16);
|
|
103
|
+
if (!isNaN(byte)) {
|
|
104
|
+
bytes.push(byte);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
// Check if hex is UTF-16BE with BOM (\xFE\xFF)
|
|
108
|
+
if (bytes.length >= 2 && bytes[0] === 0xfe && bytes[1] === 0xff) {
|
|
109
|
+
let res = "";
|
|
110
|
+
for (let i = 2; i + 1 < bytes.length; i += 2) {
|
|
111
|
+
const code = (bytes[i] << 8) | bytes[i + 1];
|
|
112
|
+
res += String.fromCharCode(code);
|
|
113
|
+
}
|
|
114
|
+
return { text: res, unmappable: false };
|
|
115
|
+
}
|
|
116
|
+
// Check if 2-byte Identity-H ASCII (e.g. <00480069> -> "Hi")
|
|
117
|
+
if (bytes.length >= 2 && bytes.length % 2 === 0) {
|
|
118
|
+
let isTwoByteAscii = true;
|
|
119
|
+
for (let i = 0; i < bytes.length; i += 2) {
|
|
120
|
+
if (bytes[i] !== 0x00 || bytes[i + 1] < 0x20 || bytes[i + 1] > 0x7e) {
|
|
121
|
+
isTwoByteAscii = false;
|
|
122
|
+
break;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
if (isTwoByteAscii) {
|
|
126
|
+
let res = "";
|
|
127
|
+
for (let i = 0; i < bytes.length; i += 2) {
|
|
128
|
+
res += String.fromCharCode(bytes[i + 1]);
|
|
129
|
+
}
|
|
130
|
+
return { text: res, unmappable: false };
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
let result = "";
|
|
134
|
+
for (const byte of bytes) {
|
|
135
|
+
result += String.fromCharCode(byte);
|
|
136
|
+
}
|
|
137
|
+
// High bytes without a recognized 2-byte structure: likely CID or a
|
|
138
|
+
// simple-font encoding we have no table for.
|
|
139
|
+
const unmappable = bytes.some((b) => b > 0x7e);
|
|
140
|
+
return { text: result, unmappable };
|
|
141
|
+
}
|
|
142
|
+
/**
|
|
143
|
+
* Read a balanced PDF literal string starting at `start` (which must
|
|
144
|
+
* point at the opening "("). Nested parentheses are literal characters
|
|
145
|
+
* per the spec; escapes are decoded. Returns the decoded value, the
|
|
146
|
+
* index just past the closing ")", and an `unmappable` flag set when
|
|
147
|
+
* the decoded text contains high bytes — literal strings share the
|
|
148
|
+
* hex strings' limitation (no font/ToUnicode context), so callers mark
|
|
149
|
+
* the document low-confidence and prefer the external text layer.
|
|
150
|
+
*/
|
|
151
|
+
function readLiteralStringAt(src, start) {
|
|
152
|
+
let depth = 0;
|
|
153
|
+
let i = start;
|
|
154
|
+
let raw = "";
|
|
155
|
+
while (i < src.length) {
|
|
156
|
+
const ch = src[i];
|
|
157
|
+
if (ch === "\\") {
|
|
158
|
+
raw += ch + (src[i + 1] ?? "");
|
|
159
|
+
i += 2;
|
|
160
|
+
continue;
|
|
161
|
+
}
|
|
162
|
+
if (ch === "(") {
|
|
163
|
+
depth++;
|
|
164
|
+
if (depth > 1)
|
|
165
|
+
raw += ch;
|
|
166
|
+
i++;
|
|
167
|
+
continue;
|
|
168
|
+
}
|
|
169
|
+
if (ch === ")") {
|
|
170
|
+
depth--;
|
|
171
|
+
if (depth === 0) {
|
|
172
|
+
i++;
|
|
173
|
+
break;
|
|
174
|
+
}
|
|
175
|
+
raw += ch;
|
|
176
|
+
i++;
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
raw += ch;
|
|
180
|
+
i++;
|
|
181
|
+
}
|
|
182
|
+
const value = decodePdfLiteralString(raw);
|
|
183
|
+
return { value, next: i, unmappable: /[^\x00-\x7e]/.test(value) };
|
|
184
|
+
}
|
|
185
|
+
/**
|
|
186
|
+
* Read a PDF number at `start` (integer or real, optional sign, leading
|
|
187
|
+
* dot like `-.5` and trailing dot like `4.` per the PDF grammar).
|
|
188
|
+
*/
|
|
189
|
+
function readNumberAt(src, start) {
|
|
190
|
+
let i = start;
|
|
191
|
+
if (src[i] === "+" || src[i] === "-")
|
|
192
|
+
i++;
|
|
193
|
+
while (i < src.length && /[0-9.]/.test(src[i]))
|
|
194
|
+
i++;
|
|
195
|
+
const value = Number(src.slice(start, i));
|
|
196
|
+
return { value: isNaN(value) ? 0 : value, next: i };
|
|
197
|
+
}
|
|
198
|
+
/**
|
|
199
|
+
* Extract BT ... ET text blocks from a content stream, respecting literal strings
|
|
200
|
+
* and comments so that "ET" occurring inside a string does not prematurely terminate the block.
|
|
201
|
+
*/
|
|
202
|
+
function extractBtBlocks(streamStr) {
|
|
203
|
+
const blocks = [];
|
|
204
|
+
const len = streamStr.length;
|
|
205
|
+
let i = 0;
|
|
206
|
+
let outerInComment = false;
|
|
207
|
+
let outerInString = false;
|
|
208
|
+
let outerParenDepth = 0;
|
|
209
|
+
let outerInHex = false;
|
|
210
|
+
while (i < len) {
|
|
211
|
+
const ch = streamStr[i];
|
|
212
|
+
if (outerInComment) {
|
|
213
|
+
if (ch === "\r" || ch === "\n")
|
|
214
|
+
outerInComment = false;
|
|
215
|
+
i++;
|
|
216
|
+
continue;
|
|
217
|
+
}
|
|
218
|
+
if (outerInHex) {
|
|
219
|
+
if (ch === ">")
|
|
220
|
+
outerInHex = false;
|
|
221
|
+
i++;
|
|
222
|
+
continue;
|
|
223
|
+
}
|
|
224
|
+
if (outerInString) {
|
|
225
|
+
if (ch === "\\") {
|
|
226
|
+
i += 2;
|
|
227
|
+
}
|
|
228
|
+
else if (ch === "(") {
|
|
229
|
+
outerParenDepth++;
|
|
230
|
+
i++;
|
|
231
|
+
}
|
|
232
|
+
else if (ch === ")") {
|
|
233
|
+
outerParenDepth--;
|
|
234
|
+
if (outerParenDepth === 0)
|
|
235
|
+
outerInString = false;
|
|
236
|
+
i++;
|
|
237
|
+
}
|
|
238
|
+
else {
|
|
239
|
+
i++;
|
|
240
|
+
}
|
|
241
|
+
continue;
|
|
242
|
+
}
|
|
243
|
+
if (ch === "%") {
|
|
244
|
+
outerInComment = true;
|
|
245
|
+
i++;
|
|
246
|
+
continue;
|
|
247
|
+
}
|
|
248
|
+
if (ch === "<") {
|
|
249
|
+
// Consume both characters of a dictionary opener: advancing one
|
|
250
|
+
// char at a time would make the second "<" look like a lone
|
|
251
|
+
// hex-string opener and mask the dictionary body.
|
|
252
|
+
if (streamStr[i + 1] === "<") {
|
|
253
|
+
i += 2;
|
|
254
|
+
continue;
|
|
255
|
+
}
|
|
256
|
+
outerInHex = true;
|
|
257
|
+
i++;
|
|
258
|
+
continue;
|
|
259
|
+
}
|
|
260
|
+
if (ch === "(") {
|
|
261
|
+
outerInString = true;
|
|
262
|
+
outerParenDepth = 1;
|
|
263
|
+
i++;
|
|
264
|
+
continue;
|
|
265
|
+
}
|
|
266
|
+
if ((i === 0 || /[\s\[\]<>()/%]/.test(streamStr[i - 1])) &&
|
|
267
|
+
streamStr.slice(i, i + 2) === "BT" &&
|
|
268
|
+
(i + 2 === len || /[\s\[\]<>()/%]/.test(streamStr[i + 2]))) {
|
|
269
|
+
i += 2;
|
|
270
|
+
const start = i;
|
|
271
|
+
let inString = false;
|
|
272
|
+
let parenDepth = 0;
|
|
273
|
+
let inHex = false;
|
|
274
|
+
let inComment = false;
|
|
275
|
+
while (i < len) {
|
|
276
|
+
const c = streamStr[i];
|
|
277
|
+
if (inComment) {
|
|
278
|
+
if (c === "\r" || c === "\n")
|
|
279
|
+
inComment = false;
|
|
280
|
+
}
|
|
281
|
+
else if (inHex) {
|
|
282
|
+
if (c === ">")
|
|
283
|
+
inHex = false;
|
|
284
|
+
}
|
|
285
|
+
else if (inString) {
|
|
286
|
+
if (c === "\\") {
|
|
287
|
+
i++;
|
|
288
|
+
}
|
|
289
|
+
else if (c === "(") {
|
|
290
|
+
parenDepth++;
|
|
291
|
+
}
|
|
292
|
+
else if (c === ")") {
|
|
293
|
+
parenDepth--;
|
|
294
|
+
if (parenDepth === 0)
|
|
295
|
+
inString = false;
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
else {
|
|
299
|
+
if (c === "%") {
|
|
300
|
+
inComment = true;
|
|
301
|
+
}
|
|
302
|
+
else if (c === "<" && streamStr[i + 1] === "<") {
|
|
303
|
+
i++; // first "<" of "<<"; the shared tail i++ consumes the second
|
|
304
|
+
}
|
|
305
|
+
else if (c === "<") {
|
|
306
|
+
inHex = true;
|
|
307
|
+
}
|
|
308
|
+
else if (c === "(") {
|
|
309
|
+
inString = true;
|
|
310
|
+
parenDepth = 1;
|
|
311
|
+
}
|
|
312
|
+
else if ((i === 0 || /[\s\[\]<>()/%]/.test(streamStr[i - 1])) &&
|
|
313
|
+
streamStr.slice(i, i + 2) === "ET" &&
|
|
314
|
+
(i + 2 === len || /[\s\[\]<>()/%]/.test(streamStr[i + 2]))) {
|
|
315
|
+
blocks.push(streamStr.slice(start, i));
|
|
316
|
+
i += 2;
|
|
317
|
+
break;
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
i++;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
else {
|
|
324
|
+
i++;
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
return blocks;
|
|
328
|
+
}
|
|
329
|
+
/**
|
|
330
|
+
* Parse text operators from a decompressed PDF content stream using a
|
|
331
|
+
* tokenizer rather than operand regexes: balanced literal strings
|
|
332
|
+
* (arbitrary nesting), hex strings, PDF reals (signed, leading/trailing
|
|
333
|
+
* dot), TJ arrays, and the full set of text-showing operators (Tj, TJ,
|
|
334
|
+
* ', ") are recognized so valid content is not silently dropped.
|
|
335
|
+
* `ctx.sawUnmappable` is set when a hex or literal string cannot be
|
|
336
|
+
* decoded with confidence (CID/unknown encoding, high bytes) so the
|
|
337
|
+
* caller can prefer the external text layer.
|
|
338
|
+
*/
|
|
339
|
+
function parseContentStreamText(streamStr, ctx) {
|
|
340
|
+
const lines = [];
|
|
341
|
+
let currentLine = "";
|
|
342
|
+
const pushLine = () => {
|
|
343
|
+
if (currentLine.trim().length > 0) {
|
|
344
|
+
lines.push(currentLine.trim());
|
|
345
|
+
currentLine = "";
|
|
346
|
+
}
|
|
347
|
+
};
|
|
348
|
+
const appendText = (value) => {
|
|
349
|
+
if (value.length > 0)
|
|
350
|
+
currentLine += value;
|
|
351
|
+
};
|
|
352
|
+
for (const block of extractBtBlocks(streamStr)) {
|
|
353
|
+
const stack = [];
|
|
354
|
+
const readHexStringOperand = (index) => {
|
|
355
|
+
const end = block.indexOf(">", index);
|
|
356
|
+
const close = end === -1 ? block.length : end;
|
|
357
|
+
const decoded = decodePdfHexString(block.slice(index + 1, close));
|
|
358
|
+
if (decoded.unmappable)
|
|
359
|
+
ctx.sawUnmappable = true;
|
|
360
|
+
return {
|
|
361
|
+
operand: { type: "hex", value: decoded.text, unmappable: decoded.unmappable },
|
|
362
|
+
next: end === -1 ? block.length : end + 1,
|
|
363
|
+
};
|
|
364
|
+
};
|
|
365
|
+
let i = 0;
|
|
366
|
+
while (i < block.length) {
|
|
367
|
+
const ch = block[i];
|
|
368
|
+
if (/\s/.test(ch)) {
|
|
369
|
+
i++;
|
|
370
|
+
continue;
|
|
371
|
+
}
|
|
372
|
+
if (ch === "%") {
|
|
373
|
+
while (i < block.length && block[i] !== "\r" && block[i] !== "\n")
|
|
374
|
+
i++;
|
|
375
|
+
continue;
|
|
376
|
+
}
|
|
377
|
+
if (ch === "(") {
|
|
378
|
+
const { value, next, unmappable } = readLiteralStringAt(block, i);
|
|
379
|
+
if (unmappable)
|
|
380
|
+
ctx.sawUnmappable = true;
|
|
381
|
+
stack.push({ type: "str", value });
|
|
382
|
+
i = next;
|
|
383
|
+
continue;
|
|
384
|
+
}
|
|
385
|
+
if (ch === "<" && block[i + 1] === "<") {
|
|
386
|
+
// Inline dictionary: skip with bracket balance.
|
|
387
|
+
let depth = 1;
|
|
388
|
+
i += 2;
|
|
389
|
+
while (i < block.length && depth > 0) {
|
|
390
|
+
if (block.slice(i, i + 2) === "<<") {
|
|
391
|
+
depth++;
|
|
392
|
+
i += 2;
|
|
393
|
+
}
|
|
394
|
+
else if (block.slice(i, i + 2) === ">>") {
|
|
395
|
+
depth--;
|
|
396
|
+
i += 2;
|
|
397
|
+
}
|
|
398
|
+
else {
|
|
399
|
+
i++;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
continue;
|
|
403
|
+
}
|
|
404
|
+
if (ch === "<") {
|
|
405
|
+
const { operand, next } = readHexStringOperand(i);
|
|
406
|
+
stack.push(operand);
|
|
407
|
+
i = next;
|
|
408
|
+
continue;
|
|
409
|
+
}
|
|
410
|
+
if (ch === "[") {
|
|
411
|
+
// TJ array: strings and kerning numbers until the closing "]".
|
|
412
|
+
const items = [];
|
|
413
|
+
i++;
|
|
414
|
+
while (i < block.length && block[i] !== "]") {
|
|
415
|
+
const c = block[i];
|
|
416
|
+
if (/\s/.test(c)) {
|
|
417
|
+
i++;
|
|
418
|
+
}
|
|
419
|
+
else if (c === "(") {
|
|
420
|
+
const { value, next, unmappable } = readLiteralStringAt(block, i);
|
|
421
|
+
if (unmappable)
|
|
422
|
+
ctx.sawUnmappable = true;
|
|
423
|
+
items.push({ type: "str", value });
|
|
424
|
+
i = next;
|
|
425
|
+
}
|
|
426
|
+
else if (c === "<") {
|
|
427
|
+
const end = block.indexOf(">", i);
|
|
428
|
+
const close = end === -1 ? block.length : end;
|
|
429
|
+
const decoded = decodePdfHexString(block.slice(i + 1, close));
|
|
430
|
+
if (decoded.unmappable)
|
|
431
|
+
ctx.sawUnmappable = true;
|
|
432
|
+
items.push({ type: "hex", value: decoded.text, unmappable: decoded.unmappable });
|
|
433
|
+
i = end === -1 ? block.length : end + 1;
|
|
434
|
+
}
|
|
435
|
+
else if (/[0-9+.\-]/.test(c)) {
|
|
436
|
+
const { value, next } = readNumberAt(block, i);
|
|
437
|
+
items.push({ type: "num", value });
|
|
438
|
+
i = next;
|
|
439
|
+
}
|
|
440
|
+
else {
|
|
441
|
+
i++;
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
i++; // consume "]"
|
|
445
|
+
stack.push({ type: "array", items });
|
|
446
|
+
continue;
|
|
447
|
+
}
|
|
448
|
+
if (/[0-9+.\-]/.test(ch)) {
|
|
449
|
+
const { value, next } = readNumberAt(block, i);
|
|
450
|
+
stack.push({ type: "num", value });
|
|
451
|
+
i = next;
|
|
452
|
+
continue;
|
|
453
|
+
}
|
|
454
|
+
// Operator keyword (letters, apostrophe, double quote, asterisk).
|
|
455
|
+
let j = i;
|
|
456
|
+
while (j < block.length && /[A-Za-z'*"]/.test(block[j]))
|
|
457
|
+
j++;
|
|
458
|
+
const op = block.slice(i, j);
|
|
459
|
+
i = j;
|
|
460
|
+
if (op.length === 0) {
|
|
461
|
+
i++;
|
|
462
|
+
continue;
|
|
463
|
+
}
|
|
464
|
+
const top = stack[stack.length - 1];
|
|
465
|
+
switch (op) {
|
|
466
|
+
case "Tj":
|
|
467
|
+
// Successive Tj operators continue at the current text
|
|
468
|
+
// position: no injected space between operands.
|
|
469
|
+
if (top && (top.type === "str" || top.type === "hex"))
|
|
470
|
+
appendText(top.value);
|
|
471
|
+
break;
|
|
472
|
+
case "TJ":
|
|
473
|
+
if (top?.type === "array") {
|
|
474
|
+
for (const item of top.items) {
|
|
475
|
+
if (item.type === "str" || item.type === "hex")
|
|
476
|
+
appendText(item.value);
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
break;
|
|
480
|
+
case "T*":
|
|
481
|
+
case "Td":
|
|
482
|
+
case "TD":
|
|
483
|
+
pushLine();
|
|
484
|
+
break;
|
|
485
|
+
case "'":
|
|
486
|
+
case '"': {
|
|
487
|
+
// Move to the next line and show text: aw/ac operands sit
|
|
488
|
+
// below the string on the operand stack.
|
|
489
|
+
pushLine();
|
|
490
|
+
if (top && (top.type === "str" || top.type === "hex"))
|
|
491
|
+
appendText(top.value + " ");
|
|
492
|
+
break;
|
|
493
|
+
}
|
|
494
|
+
default:
|
|
495
|
+
break; // operators without direct text semantics
|
|
496
|
+
}
|
|
497
|
+
// Operators consume their operands: a hostile BT block with
|
|
498
|
+
// thousands of no-op operators must not accumulate them on the
|
|
499
|
+
// stack until the heap is exhausted.
|
|
500
|
+
stack.length = 0;
|
|
501
|
+
}
|
|
502
|
+
pushLine();
|
|
503
|
+
}
|
|
504
|
+
return lines.join("\n");
|
|
505
|
+
}
|
|
506
|
+
/**
|
|
507
|
+
* Parse a stream dictionary's /Filter entry into an ordered chain.
|
|
508
|
+
* Handles both `/Filter /Name` and `/Filter [/A /B]` forms; an absent
|
|
509
|
+
* entry yields an empty chain (unfiltered).
|
|
510
|
+
*/
|
|
511
|
+
function parseFilterChain(dictSlice) {
|
|
512
|
+
const arrayMatch = /\/Filter\s*\[([^\]]*)\]/.exec(dictSlice);
|
|
513
|
+
if (arrayMatch) {
|
|
514
|
+
return (arrayMatch[1].match(/\/[A-Za-z0-9]+/g) ?? []).map((name) => name.slice(1));
|
|
515
|
+
}
|
|
516
|
+
const single = /\/Filter\s*(\/[A-Za-z0-9]+)/.exec(dictSlice);
|
|
517
|
+
return single ? [single[1].slice(1)] : [];
|
|
518
|
+
}
|
|
519
|
+
/**
|
|
520
|
+
* Heuristic content-role check: streams whose dictionary marks them as
|
|
521
|
+
* embedded images, object/metadata containers, or font programs are
|
|
522
|
+
* not page content and must not contribute text.
|
|
523
|
+
*/
|
|
524
|
+
function isNonContentDictionary(dictSlice) {
|
|
525
|
+
return /\/Subtype\s*\/Image\b|\/Type\s*\/ObjStm\b|\/Type\s*\/Metadata\b|\/Length1\b|\/FontFile\d?\b/.test(dictSlice);
|
|
526
|
+
}
|
|
527
|
+
/**
|
|
528
|
+
* Locate a stream's own dictionary by balancing brackets backward from
|
|
529
|
+
* the ">>" nearest the keyword (within the window). Imperfect for ">>"
|
|
530
|
+
* inside dictionary strings, but /Length sits early in real stream
|
|
531
|
+
* dictionaries and resolveStreamInterval validates before trusting it.
|
|
532
|
+
*/
|
|
533
|
+
function scanBackwardDictionary(fileStr, keywordStart, window = 2000) {
|
|
534
|
+
const beforeStream = fileStr.slice(Math.max(0, keywordStart - window), keywordStart).trimEnd();
|
|
535
|
+
const lastDictEnd = beforeStream.lastIndexOf(">>");
|
|
536
|
+
if (lastDictEnd === -1)
|
|
537
|
+
return "";
|
|
538
|
+
let depth = 1;
|
|
539
|
+
let pos = lastDictEnd;
|
|
540
|
+
while (pos > 1) {
|
|
541
|
+
if (beforeStream.slice(pos - 2, pos) === ">>") {
|
|
542
|
+
depth++;
|
|
543
|
+
pos -= 2;
|
|
544
|
+
}
|
|
545
|
+
else if (beforeStream.slice(pos - 2, pos) === "<<") {
|
|
546
|
+
depth--;
|
|
547
|
+
pos -= 2;
|
|
548
|
+
if (depth === 0) {
|
|
549
|
+
return beforeStream.slice(pos, lastDictEnd + 2);
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
else {
|
|
553
|
+
pos--;
|
|
554
|
+
}
|
|
555
|
+
}
|
|
556
|
+
return "";
|
|
557
|
+
}
|
|
558
|
+
/**
|
|
559
|
+
* Find an indirect-object header (`N G obj`) with a lexical scan that
|
|
560
|
+
* skips comments, literal/hex strings, and stream bodies — a header
|
|
561
|
+
* lookalike inside stream data must never be selected. Stream bodies
|
|
562
|
+
* are skipped with length-VALIDATED boundaries (`resolving` guards
|
|
563
|
+
* against indirect-length resolution cycles).
|
|
564
|
+
*/
|
|
565
|
+
function findObjectHeaderOffset(fileStr, header, resolving = new Set()) {
|
|
566
|
+
let i = 0;
|
|
567
|
+
while (i < fileStr.length) {
|
|
568
|
+
const ch = fileStr[i];
|
|
569
|
+
if (ch === "%") {
|
|
570
|
+
while (i < fileStr.length && fileStr[i] !== "\r" && fileStr[i] !== "\n")
|
|
571
|
+
i++;
|
|
572
|
+
continue;
|
|
573
|
+
}
|
|
574
|
+
if (ch === "(") {
|
|
575
|
+
let depth = 1;
|
|
576
|
+
i++;
|
|
577
|
+
while (i < fileStr.length && depth > 0) {
|
|
578
|
+
const c = fileStr[i];
|
|
579
|
+
if (c === "\\") {
|
|
580
|
+
i += 2;
|
|
581
|
+
continue;
|
|
582
|
+
}
|
|
583
|
+
if (c === "(")
|
|
584
|
+
depth++;
|
|
585
|
+
else if (c === ")")
|
|
586
|
+
depth--;
|
|
587
|
+
i++;
|
|
588
|
+
}
|
|
589
|
+
continue;
|
|
590
|
+
}
|
|
591
|
+
if (ch === "<" && fileStr[i + 1] !== "<") {
|
|
592
|
+
const gt = fileStr.indexOf(">", i);
|
|
593
|
+
i = gt === -1 ? fileStr.length : gt + 1;
|
|
594
|
+
continue;
|
|
595
|
+
}
|
|
596
|
+
if (fileStr.startsWith("stream", i) &&
|
|
597
|
+
(i === 0 || fileStr.slice(Math.max(0, i - 3), i) !== "end") &&
|
|
598
|
+
/[\r\n]/.test(fileStr[i + 6] ?? "\n")) {
|
|
599
|
+
const interval = resolveStreamInterval(fileStr, i, maskStringsAndComments(scanBackwardDictionary(fileStr, i)), resolving);
|
|
600
|
+
i = interval.resumeAfter;
|
|
601
|
+
continue;
|
|
602
|
+
}
|
|
603
|
+
if (fileStr.startsWith(header, i)) {
|
|
604
|
+
const after = fileStr[i + header.length];
|
|
605
|
+
if (after === undefined || /[\s\r\n<>%]/.test(after))
|
|
606
|
+
return i;
|
|
607
|
+
}
|
|
608
|
+
i++;
|
|
609
|
+
}
|
|
610
|
+
return -1;
|
|
611
|
+
}
|
|
612
|
+
/**
|
|
613
|
+
* Resolve an indirect /Length (N G R) by reading the integer the
|
|
614
|
+
* referenced object holds. Returns null when the reference or object
|
|
615
|
+
* cannot be resolved.
|
|
616
|
+
*/
|
|
617
|
+
function resolveIndirectLength(fileStr, maskedDict, resolving = new Set()) {
|
|
618
|
+
const ref = /\/Length\s+(\d+)\s+(\d+)\s+R\b/.exec(maskedDict);
|
|
619
|
+
if (!ref)
|
|
620
|
+
return null;
|
|
621
|
+
// Recursion guard: a length object whose own stream references back
|
|
622
|
+
// must not loop (A -> B -> A); give up and fall back instead.
|
|
623
|
+
const refKey = `${ref[1]} ${ref[2]}`;
|
|
624
|
+
if (resolving.has(refKey))
|
|
625
|
+
return null;
|
|
626
|
+
const header = `${refKey} obj`;
|
|
627
|
+
const at = findObjectHeaderOffset(fileStr, header, new Set([...resolving, refKey]));
|
|
628
|
+
if (at === -1)
|
|
629
|
+
return null;
|
|
630
|
+
let p = at + header.length;
|
|
631
|
+
while (p < fileStr.length) {
|
|
632
|
+
if (/\s/.test(fileStr[p])) {
|
|
633
|
+
p++;
|
|
634
|
+
continue;
|
|
635
|
+
}
|
|
636
|
+
// The integer may be preceded by comments inside the object.
|
|
637
|
+
if (fileStr[p] === "%") {
|
|
638
|
+
while (p < fileStr.length && fileStr[p] !== "\r" && fileStr[p] !== "\n")
|
|
639
|
+
p++;
|
|
640
|
+
continue;
|
|
641
|
+
}
|
|
642
|
+
break;
|
|
643
|
+
}
|
|
644
|
+
const num = /^(\d+)/.exec(fileStr.slice(p, p + 16));
|
|
645
|
+
return num ? Number(num[1]) : null;
|
|
646
|
+
}
|
|
647
|
+
/**
|
|
648
|
+
* Resolve a stream body interval: [bodyStart, dataEnd) plus the index
|
|
649
|
+
* to resume scanning from (past the real endstream keyword). Primary
|
|
650
|
+
* signal is the dictionary /Length, VALIDATED against a following
|
|
651
|
+
* endstream keyword, so embedded "endstream" bytes inside stream data
|
|
652
|
+
* cannot truncate the stream or leak stream interiors into syntax
|
|
653
|
+
* scans. The textual search is only a fallback for absent or indirect
|
|
654
|
+
* (/Length 12 0 R) or non-validating lengths.
|
|
655
|
+
*/
|
|
656
|
+
function resolveStreamInterval(fileStr, keywordStart, dictSlice, resolving = new Set()) {
|
|
657
|
+
let bodyStart = keywordStart + "stream".length;
|
|
658
|
+
if (fileStr[bodyStart] === "\r")
|
|
659
|
+
bodyStart++;
|
|
660
|
+
if (fileStr[bodyStart] === "\n")
|
|
661
|
+
bodyStart++;
|
|
662
|
+
// Textual fallback: accept an `endstream` keyword only when it is
|
|
663
|
+
// followed by `endobj` (or end-of-file adjacency) — an embedded
|
|
664
|
+
// `endstream` token inside stream data is typically followed by more
|
|
665
|
+
// data, never by the object terminator, so this keeps the fallback
|
|
666
|
+
// from resuming mid-stream.
|
|
667
|
+
let keywordAt = -1;
|
|
668
|
+
for (let probe = fileStr.indexOf("endstream", bodyStart); probe !== -1; probe = fileStr.indexOf("endstream", probe + 9)) {
|
|
669
|
+
if (/^endstream[\s\S]{0,4}?endobj\b/.test(fileStr.slice(probe, probe + 60))) {
|
|
670
|
+
keywordAt = probe;
|
|
671
|
+
break;
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
let dataEnd = keywordAt === -1 ? fileStr.length : keywordAt;
|
|
675
|
+
// Direct /Length N (the lookahead avoids reading the object number of
|
|
676
|
+
// an indirect /Length N G R as a direct length); indirect lengths are
|
|
677
|
+
// resolved from their integer object so generators that use them do
|
|
678
|
+
// not fall back to a textual endstream search that can stop at an
|
|
679
|
+
// embedded "endstream" sequence inside the stream body.
|
|
680
|
+
const lengthMatch = /\/Length\s+(\d+)(?!\d)(?!\s+\d+\s+R)/.exec(dictSlice);
|
|
681
|
+
const declared = lengthMatch !== null
|
|
682
|
+
? Number(lengthMatch[1])
|
|
683
|
+
: resolveIndirectLength(fileStr, dictSlice, resolving);
|
|
684
|
+
if (declared !== null) {
|
|
685
|
+
let probe = bodyStart + declared;
|
|
686
|
+
if (fileStr[probe] === "\r")
|
|
687
|
+
probe++;
|
|
688
|
+
if (fileStr[probe] === "\n")
|
|
689
|
+
probe++;
|
|
690
|
+
if (fileStr.startsWith("endstream", probe)) {
|
|
691
|
+
dataEnd = bodyStart + declared;
|
|
692
|
+
keywordAt = probe;
|
|
693
|
+
}
|
|
694
|
+
}
|
|
695
|
+
const resumeAfter = keywordAt === -1 ? fileStr.length : keywordAt + "endstream".length;
|
|
696
|
+
return { bodyStart, dataEnd, resumeAfter };
|
|
697
|
+
}
|
|
698
|
+
/**
|
|
699
|
+
* Read the dictionary that starts at/after `i` (leading whitespace
|
|
700
|
+
* tolerated), balancing << >> while skipping literal strings, hex
|
|
701
|
+
* strings, and comments so delimiters inside string data cannot
|
|
702
|
+
* unbalance the walk. Returns the dictionary text and the index just
|
|
703
|
+
* past its closing ">>" (or end of input when unterminated).
|
|
704
|
+
*/
|
|
705
|
+
function readObjectDictionaryAt(fileStr, i) {
|
|
706
|
+
let p = i;
|
|
707
|
+
while (p < fileStr.length) {
|
|
708
|
+
if (/\s/.test(fileStr[p])) {
|
|
709
|
+
p++;
|
|
710
|
+
continue;
|
|
711
|
+
}
|
|
712
|
+
// Object headers may carry comments before the dictionary opener.
|
|
713
|
+
if (fileStr[p] === "%") {
|
|
714
|
+
while (p < fileStr.length && fileStr[p] !== "\r" && fileStr[p] !== "\n")
|
|
715
|
+
p++;
|
|
716
|
+
continue;
|
|
717
|
+
}
|
|
718
|
+
break;
|
|
719
|
+
}
|
|
720
|
+
if (fileStr.slice(p, p + 2) !== "<<")
|
|
721
|
+
return { text: "", next: i };
|
|
722
|
+
const start = p;
|
|
723
|
+
let depth = 1;
|
|
724
|
+
p += 2;
|
|
725
|
+
while (p < fileStr.length && depth > 0) {
|
|
726
|
+
const ch = fileStr[p];
|
|
727
|
+
if (ch === "%") {
|
|
728
|
+
while (p < fileStr.length && fileStr[p] !== "\r" && fileStr[p] !== "\n")
|
|
729
|
+
p++;
|
|
730
|
+
}
|
|
731
|
+
else if (ch === "(") {
|
|
732
|
+
let sDepth = 1;
|
|
733
|
+
p++;
|
|
734
|
+
while (p < fileStr.length && sDepth > 0) {
|
|
735
|
+
const c = fileStr[p];
|
|
736
|
+
if (c === "\\") {
|
|
737
|
+
p += 2;
|
|
738
|
+
continue;
|
|
739
|
+
}
|
|
740
|
+
if (c === "(")
|
|
741
|
+
sDepth++;
|
|
742
|
+
else if (c === ")")
|
|
743
|
+
sDepth--;
|
|
744
|
+
p++;
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
else if (ch === "<" && fileStr[p + 1] !== "<") {
|
|
748
|
+
const gt = fileStr.indexOf(">", p);
|
|
749
|
+
p = gt === -1 ? fileStr.length : gt + 1;
|
|
750
|
+
}
|
|
751
|
+
else if (fileStr.slice(p, p + 2) === "<<") {
|
|
752
|
+
depth++;
|
|
753
|
+
p += 2;
|
|
754
|
+
}
|
|
755
|
+
else if (fileStr.slice(p, p + 2) === ">>") {
|
|
756
|
+
depth--;
|
|
757
|
+
p += 2;
|
|
758
|
+
}
|
|
759
|
+
else {
|
|
760
|
+
p++;
|
|
761
|
+
}
|
|
762
|
+
}
|
|
763
|
+
return { text: fileStr.slice(start, p), next: p };
|
|
764
|
+
}
|
|
765
|
+
/**
|
|
766
|
+
* Built-in pure-Node text extraction by scanning PDF streams. Returns
|
|
767
|
+
* the text plus a low-confidence flag set when any hex string could not
|
|
768
|
+
* be decoded with font context (CID/unknown encoding) — such text is a
|
|
769
|
+
* Latin-1 best effort, and callers should prefer the external layer.
|
|
770
|
+
*/
|
|
771
|
+
function extractPureNodePdfText(buffer) {
|
|
772
|
+
const fileStr = buffer.toString("latin1");
|
|
773
|
+
const extractedSections = [];
|
|
774
|
+
const ctx = { sawUnmappable: false };
|
|
775
|
+
const MAX_TOTAL_DECOMPRESSED_BYTES = 50 * 1024 * 1024; // 50MB across document
|
|
776
|
+
let totalDecompressedBytes = 0;
|
|
777
|
+
// Stream keyword scan; body intervals come from the dictionary
|
|
778
|
+
// /Length via resolveStreamInterval (validated against the real
|
|
779
|
+
// endstream keyword), so embedded "endstream" bytes inside stream
|
|
780
|
+
// data cannot truncate a stream. The "end" prefix guard keeps the
|
|
781
|
+
// tail of "endstream" keywords from matching as stream starts.
|
|
782
|
+
// Lexical stream scan: literal strings and comments are skipped
|
|
783
|
+
// while walking, so a `stream` token inside string or comment data
|
|
784
|
+
// can never corrupt the next stream interval.
|
|
785
|
+
let cursor = 0;
|
|
786
|
+
while (true) {
|
|
787
|
+
// Advance to the next stream keyword outside strings/comments.
|
|
788
|
+
let keywordStart = -1;
|
|
789
|
+
let i = cursor;
|
|
790
|
+
while (i < fileStr.length) {
|
|
791
|
+
const ch = fileStr[i];
|
|
792
|
+
if (ch === "%") {
|
|
793
|
+
while (i < fileStr.length && fileStr[i] !== "\r" && fileStr[i] !== "\n")
|
|
794
|
+
i++;
|
|
795
|
+
continue;
|
|
796
|
+
}
|
|
797
|
+
if (ch === "(") {
|
|
798
|
+
let depth = 1;
|
|
799
|
+
i++;
|
|
800
|
+
while (i < fileStr.length && depth > 0) {
|
|
801
|
+
const c = fileStr[i];
|
|
802
|
+
if (c === "\\") {
|
|
803
|
+
i += 2;
|
|
804
|
+
continue;
|
|
805
|
+
}
|
|
806
|
+
if (c === "(")
|
|
807
|
+
depth++;
|
|
808
|
+
else if (c === ")")
|
|
809
|
+
depth--;
|
|
810
|
+
i++;
|
|
811
|
+
}
|
|
812
|
+
continue;
|
|
813
|
+
}
|
|
814
|
+
if (fileStr.startsWith("stream", i) &&
|
|
815
|
+
(i === 0 || fileStr.slice(Math.max(0, i - 3), i) !== "end") &&
|
|
816
|
+
/[\r\n]/.test(fileStr[i + 6] ?? "\n")) {
|
|
817
|
+
keywordStart = i;
|
|
818
|
+
break;
|
|
819
|
+
}
|
|
820
|
+
i++;
|
|
821
|
+
}
|
|
822
|
+
if (keywordStart === -1)
|
|
823
|
+
break;
|
|
824
|
+
const dictSlice = maskStringsAndComments(scanBackwardDictionary(fileStr, keywordStart));
|
|
825
|
+
const interval = resolveStreamInterval(fileStr, keywordStart, dictSlice);
|
|
826
|
+
cursor = interval.resumeAfter;
|
|
827
|
+
const rawStreamSlice = buffer.subarray(interval.bodyStart, interval.dataEnd);
|
|
828
|
+
// Filter-chain classification (R6-C): the pure path can only
|
|
829
|
+
// decode an unfiltered stream or a single FlateDecode. Anything
|
|
830
|
+
// else (e.g. [/ASCII85Decode /FlateDecode]) marks the document
|
|
831
|
+
// low-confidence instead of silently skipping the stream while a
|
|
832
|
+
// nonempty partial result suppresses pdftotext. The dictionary is
|
|
833
|
+
// masked so /Filter or marker text inside strings/comments cannot
|
|
834
|
+
// impersonate the real entries.
|
|
835
|
+
const filterChain = parseFilterChain(dictSlice);
|
|
836
|
+
const flateOnly = filterChain.length === 1 && filterChain[0] === "FlateDecode";
|
|
837
|
+
// Non-content streams (embedded images, object/metadata containers,
|
|
838
|
+
// font programs) must not contribute text even when their bytes
|
|
839
|
+
// look like content syntax. Full page-resource-graph resolution is
|
|
840
|
+
// out of scope for the pure path; dictionary markers cover the
|
|
841
|
+
// common non-content families.
|
|
842
|
+
if (isNonContentDictionary(dictSlice)) {
|
|
843
|
+
continue;
|
|
844
|
+
}
|
|
845
|
+
let streamBytes = null;
|
|
846
|
+
if (flateOnly) {
|
|
847
|
+
if (totalDecompressedBytes >= MAX_TOTAL_DECOMPRESSED_BYTES) {
|
|
848
|
+
break; // Document-wide decompression budget reached
|
|
849
|
+
}
|
|
850
|
+
const maxStreamOutput = Math.min(20 * 1024 * 1024, MAX_TOTAL_DECOMPRESSED_BYTES - totalDecompressedBytes);
|
|
851
|
+
try {
|
|
852
|
+
streamBytes = zlib.inflateSync(rawStreamSlice, { maxOutputLength: maxStreamOutput });
|
|
853
|
+
totalDecompressedBytes += streamBytes.length;
|
|
854
|
+
}
|
|
855
|
+
catch {
|
|
856
|
+
try {
|
|
857
|
+
streamBytes = zlib.inflateRawSync(rawStreamSlice, { maxOutputLength: maxStreamOutput });
|
|
858
|
+
totalDecompressedBytes += streamBytes.length;
|
|
859
|
+
}
|
|
860
|
+
catch {
|
|
861
|
+
// Stream may be raw or uncompressed
|
|
862
|
+
}
|
|
863
|
+
}
|
|
864
|
+
}
|
|
865
|
+
else if (filterChain.length === 0) {
|
|
866
|
+
streamBytes = rawStreamSlice;
|
|
867
|
+
}
|
|
868
|
+
else {
|
|
869
|
+
// Unsupported filter chain: do not attempt to inflate partially
|
|
870
|
+
// encoded bytes; defer to the external text layer.
|
|
871
|
+
ctx.sawUnmappable = true;
|
|
872
|
+
}
|
|
873
|
+
if (streamBytes) {
|
|
874
|
+
const decodedStr = streamBytes.toString("latin1");
|
|
875
|
+
const text = parseContentStreamText(decodedStr, ctx);
|
|
876
|
+
if (text.trim().length > 0) {
|
|
877
|
+
extractedSections.push(text.trim());
|
|
878
|
+
}
|
|
879
|
+
}
|
|
880
|
+
}
|
|
881
|
+
return { text: extractedSections.join("\n\n").trim(), lowConfidence: ctx.sawUnmappable };
|
|
882
|
+
}
|
|
883
|
+
// External tools process untrusted input; cap accumulated stdout so a
|
|
884
|
+
// hostile PDF cannot exhaust the heap through pdftotext/qpdf output.
|
|
885
|
+
const MAX_EXTERNAL_TOOL_OUTPUT_BYTES = 50 * 1024 * 1024;
|
|
886
|
+
function runExternalTextTool(cmd, args, input, timeoutMs = 5000) {
|
|
887
|
+
return new Promise((resolve) => {
|
|
888
|
+
let finished = false;
|
|
889
|
+
let timer = null;
|
|
890
|
+
const finish = (val) => {
|
|
891
|
+
if (!finished) {
|
|
892
|
+
finished = true;
|
|
893
|
+
if (timer)
|
|
894
|
+
clearTimeout(timer);
|
|
895
|
+
resolve(val);
|
|
896
|
+
}
|
|
897
|
+
};
|
|
898
|
+
try {
|
|
899
|
+
const child = spawn(cmd, args, { stdio: ["pipe", "pipe", "ignore"] });
|
|
900
|
+
timer = setTimeout(() => {
|
|
901
|
+
child.kill("SIGKILL");
|
|
902
|
+
finish(null);
|
|
903
|
+
}, timeoutMs);
|
|
904
|
+
let stdout = "";
|
|
905
|
+
let stdoutBytes = 0;
|
|
906
|
+
child.stdout.setEncoding("utf8");
|
|
907
|
+
child.stdout.on("data", (chunk) => {
|
|
908
|
+
stdoutBytes += Buffer.byteLength(chunk);
|
|
909
|
+
if (stdoutBytes > MAX_EXTERNAL_TOOL_OUTPUT_BYTES) {
|
|
910
|
+
child.kill("SIGKILL");
|
|
911
|
+
finish(null);
|
|
912
|
+
return;
|
|
913
|
+
}
|
|
914
|
+
stdout += chunk;
|
|
915
|
+
});
|
|
916
|
+
child.on("error", () => finish(null));
|
|
917
|
+
child.on("close", (code) => {
|
|
918
|
+
if (code === 0 && stdout.trim().length > 0) {
|
|
919
|
+
finish(stdout.trim());
|
|
920
|
+
}
|
|
921
|
+
else {
|
|
922
|
+
finish(null);
|
|
923
|
+
}
|
|
924
|
+
});
|
|
925
|
+
child.stdin.on("error", () => finish(null));
|
|
926
|
+
child.stdin.write(input);
|
|
927
|
+
child.stdin.end();
|
|
928
|
+
}
|
|
929
|
+
catch {
|
|
930
|
+
finish(null);
|
|
931
|
+
}
|
|
932
|
+
});
|
|
933
|
+
}
|
|
934
|
+
function runExternalBinaryTool(cmd, args, input, timeoutMs = 5000) {
|
|
935
|
+
return new Promise((resolve) => {
|
|
936
|
+
let finished = false;
|
|
937
|
+
let timer = null;
|
|
938
|
+
const finish = (val) => {
|
|
939
|
+
if (!finished) {
|
|
940
|
+
finished = true;
|
|
941
|
+
if (timer)
|
|
942
|
+
clearTimeout(timer);
|
|
943
|
+
resolve(val);
|
|
944
|
+
}
|
|
945
|
+
};
|
|
946
|
+
try {
|
|
947
|
+
const child = spawn(cmd, args, { stdio: ["pipe", "pipe", "ignore"] });
|
|
948
|
+
timer = setTimeout(() => {
|
|
949
|
+
child.kill("SIGKILL");
|
|
950
|
+
finish(null);
|
|
951
|
+
}, timeoutMs);
|
|
952
|
+
const chunks = [];
|
|
953
|
+
let stdoutBytes = 0;
|
|
954
|
+
child.stdout.on("data", (chunk) => {
|
|
955
|
+
stdoutBytes += chunk.length;
|
|
956
|
+
if (stdoutBytes > MAX_EXTERNAL_TOOL_OUTPUT_BYTES) {
|
|
957
|
+
child.kill("SIGKILL");
|
|
958
|
+
finish(null);
|
|
959
|
+
return;
|
|
960
|
+
}
|
|
961
|
+
chunks.push(chunk);
|
|
962
|
+
});
|
|
963
|
+
child.on("error", () => finish(null));
|
|
964
|
+
child.on("close", (code) => {
|
|
965
|
+
if (code === 0 && chunks.length > 0) {
|
|
966
|
+
finish(Buffer.concat(chunks));
|
|
967
|
+
}
|
|
968
|
+
else {
|
|
969
|
+
finish(null);
|
|
970
|
+
}
|
|
971
|
+
});
|
|
972
|
+
child.stdin.on("error", () => finish(null));
|
|
973
|
+
child.stdin.write(input);
|
|
974
|
+
child.stdin.end();
|
|
975
|
+
}
|
|
976
|
+
catch {
|
|
977
|
+
finish(null);
|
|
978
|
+
}
|
|
979
|
+
});
|
|
980
|
+
}
|
|
981
|
+
/**
|
|
982
|
+
* Extract text from a PDF buffer using opportunistic delegation or pure-Node fallback.
|
|
983
|
+
*/
|
|
984
|
+
export async function extractPdfText(buffer, timeoutMs = 5000) {
|
|
985
|
+
// 1. Pure Node fast path. Low-confidence results (hex strings the
|
|
986
|
+
// pure path cannot decode with font context) defer to the external
|
|
987
|
+
// text layer when one is installed.
|
|
988
|
+
const pure = extractPureNodePdfText(buffer);
|
|
989
|
+
if (pure.text.length > 0 && !pure.lowConfidence) {
|
|
990
|
+
return pure.text;
|
|
991
|
+
}
|
|
992
|
+
// 2. Opportunistic pdftotext if installed
|
|
993
|
+
const externalText = await runExternalTextTool("pdftotext", ["-", "-"], buffer, timeoutMs);
|
|
994
|
+
if (externalText) {
|
|
995
|
+
return externalText;
|
|
996
|
+
}
|
|
997
|
+
// 3. No external tool available: serve the best-effort pure text.
|
|
998
|
+
if (pure.text.length > 0) {
|
|
999
|
+
return pure.text;
|
|
1000
|
+
}
|
|
1001
|
+
return "[PDF document with no extractable text layer or scanned raster pages]";
|
|
1002
|
+
}
|
|
1003
|
+
/**
|
|
1004
|
+
* Blank out literal strings, hex strings, and comments so textual
|
|
1005
|
+
* lookalikes inside string data cannot satisfy dictionary probes.
|
|
1006
|
+
*/
|
|
1007
|
+
function maskStringsAndComments(src) {
|
|
1008
|
+
let out = "";
|
|
1009
|
+
let i = 0;
|
|
1010
|
+
while (i < src.length) {
|
|
1011
|
+
const ch = src[i];
|
|
1012
|
+
if (ch === "%") {
|
|
1013
|
+
let j = i;
|
|
1014
|
+
while (j < src.length && src[j] !== "\r" && src[j] !== "\n")
|
|
1015
|
+
j++;
|
|
1016
|
+
out += " ".repeat(j - i);
|
|
1017
|
+
i = j;
|
|
1018
|
+
continue;
|
|
1019
|
+
}
|
|
1020
|
+
if (ch === "(") {
|
|
1021
|
+
let depth = 1;
|
|
1022
|
+
let j = i + 1;
|
|
1023
|
+
while (j < src.length && depth > 0) {
|
|
1024
|
+
const c = src[j];
|
|
1025
|
+
if (c === "\\") {
|
|
1026
|
+
j += 2;
|
|
1027
|
+
continue;
|
|
1028
|
+
}
|
|
1029
|
+
if (c === "(")
|
|
1030
|
+
depth++;
|
|
1031
|
+
else if (c === ")")
|
|
1032
|
+
depth--;
|
|
1033
|
+
j++;
|
|
1034
|
+
}
|
|
1035
|
+
out += " ".repeat(Math.min(j, src.length) - i);
|
|
1036
|
+
i = j;
|
|
1037
|
+
continue;
|
|
1038
|
+
}
|
|
1039
|
+
if (ch === "<") {
|
|
1040
|
+
// Dictionary opener: contents must stay visible to probes —
|
|
1041
|
+
// do NOT fall through to the hex-string mask on the second "<".
|
|
1042
|
+
if (src[i + 1] === "<") {
|
|
1043
|
+
out += "<<";
|
|
1044
|
+
i += 2;
|
|
1045
|
+
continue;
|
|
1046
|
+
}
|
|
1047
|
+
const gt = src.indexOf(">", i);
|
|
1048
|
+
const j = gt === -1 ? src.length : gt + 1;
|
|
1049
|
+
out += " ".repeat(j - i);
|
|
1050
|
+
i = j;
|
|
1051
|
+
continue;
|
|
1052
|
+
}
|
|
1053
|
+
out += ch;
|
|
1054
|
+
i++;
|
|
1055
|
+
}
|
|
1056
|
+
return out;
|
|
1057
|
+
}
|
|
1058
|
+
/**
|
|
1059
|
+
* Attempt structural repair on damaged PDFs (e.g. broken xref table).
|
|
1060
|
+
*/
|
|
1061
|
+
export async function repairPdf(buffer, timeoutMs = 5000) {
|
|
1062
|
+
// 1. Pure Node xref reconstruction
|
|
1063
|
+
const MAX_REPAIR_OBJECTS = 500_000;
|
|
1064
|
+
const fileStr = buffer.toString("latin1");
|
|
1065
|
+
// Lexically scan indirect-object headers OUTSIDE comments, literal and
|
|
1066
|
+
// hex strings, and stream bodies: object-like text inside string data
|
|
1067
|
+
// must never become an xref entry pointing into that string, and
|
|
1068
|
+
// duplicate ids must resolve to the real headers (later occurrences
|
|
1069
|
+
// win below, matching incremental-update semantics). /Type /ObjStm is
|
|
1070
|
+
// detected in an actual object dictionary, not anywhere in the file.
|
|
1071
|
+
const objects = [];
|
|
1072
|
+
let overLimit = false;
|
|
1073
|
+
let objStmDetected = false;
|
|
1074
|
+
{
|
|
1075
|
+
let i = 0;
|
|
1076
|
+
while (i < fileStr.length) {
|
|
1077
|
+
const ch = fileStr[i];
|
|
1078
|
+
if (ch === "%") {
|
|
1079
|
+
while (i < fileStr.length && fileStr[i] !== "\r" && fileStr[i] !== "\n")
|
|
1080
|
+
i++;
|
|
1081
|
+
continue;
|
|
1082
|
+
}
|
|
1083
|
+
if (ch === "(") {
|
|
1084
|
+
let depth = 1;
|
|
1085
|
+
i++;
|
|
1086
|
+
while (i < fileStr.length && depth > 0) {
|
|
1087
|
+
const c = fileStr[i];
|
|
1088
|
+
if (c === "\\") {
|
|
1089
|
+
i += 2;
|
|
1090
|
+
continue;
|
|
1091
|
+
}
|
|
1092
|
+
if (c === "(")
|
|
1093
|
+
depth++;
|
|
1094
|
+
else if (c === ")")
|
|
1095
|
+
depth--;
|
|
1096
|
+
i++;
|
|
1097
|
+
}
|
|
1098
|
+
continue;
|
|
1099
|
+
}
|
|
1100
|
+
if (ch === "<" && fileStr[i + 1] === "<") {
|
|
1101
|
+
// Dictionary: skip with string awareness — `>>` inside literal
|
|
1102
|
+
// or hex strings must not close the walk.
|
|
1103
|
+
i = readObjectDictionaryAt(fileStr, i).next;
|
|
1104
|
+
continue;
|
|
1105
|
+
}
|
|
1106
|
+
if (ch === "<") {
|
|
1107
|
+
const gt = fileStr.indexOf(">", i);
|
|
1108
|
+
i = gt === -1 ? fileStr.length : gt + 1;
|
|
1109
|
+
continue;
|
|
1110
|
+
}
|
|
1111
|
+
if (fileStr.startsWith("stream", i) &&
|
|
1112
|
+
(i === 0 || fileStr.slice(Math.max(0, i - 3), i) !== "end") &&
|
|
1113
|
+
/[\r\n]/.test(fileStr[i + 6] ?? "\n")) {
|
|
1114
|
+
// Skip the ENTIRE stream body using the dictionary /Length
|
|
1115
|
+
// (validated) so an embedded "endstream" sequence inside the
|
|
1116
|
+
// data cannot make the scanner resume mid-stream and index
|
|
1117
|
+
// object-like text living inside the stream.
|
|
1118
|
+
const interval = resolveStreamInterval(fileStr, i, maskStringsAndComments(scanBackwardDictionary(fileStr, i)));
|
|
1119
|
+
i = interval.resumeAfter;
|
|
1120
|
+
continue;
|
|
1121
|
+
}
|
|
1122
|
+
if (/[0-9]/.test(ch)) {
|
|
1123
|
+
const header = /^(\d+)\s+(\d+)\s+obj\b/.exec(fileStr.slice(i, i + 32));
|
|
1124
|
+
if (header) {
|
|
1125
|
+
const id = Number(header[1]);
|
|
1126
|
+
const generation = Number(header[2]);
|
|
1127
|
+
if (id > 0 && id <= MAX_REPAIR_OBJECTS) {
|
|
1128
|
+
objects.push({ id, generation, offset: i });
|
|
1129
|
+
if (objects.length > MAX_REPAIR_OBJECTS) {
|
|
1130
|
+
overLimit = true;
|
|
1131
|
+
break;
|
|
1132
|
+
}
|
|
1133
|
+
}
|
|
1134
|
+
i += header[0].length;
|
|
1135
|
+
if (!objStmDetected) {
|
|
1136
|
+
// Inspect the object's actual dictionary (no fixed byte
|
|
1137
|
+
// window) with strings masked: /Type /ObjStm separated from
|
|
1138
|
+
// the header by padding/comments must still be found, while
|
|
1139
|
+
// lookalikes inside string data must not match.
|
|
1140
|
+
const dict = readObjectDictionaryAt(fileStr, i);
|
|
1141
|
+
if (/\/Type\s*\/ObjStm\b/.test(maskStringsAndComments(dict.text))) {
|
|
1142
|
+
objStmDetected = true;
|
|
1143
|
+
}
|
|
1144
|
+
}
|
|
1145
|
+
continue;
|
|
1146
|
+
}
|
|
1147
|
+
}
|
|
1148
|
+
i++;
|
|
1149
|
+
}
|
|
1150
|
+
}
|
|
1151
|
+
if (objects.length > 0 && !overLimit && !objStmDetected) {
|
|
1152
|
+
objects.sort((a, b) => a.id - b.id);
|
|
1153
|
+
const maxId = objects[objects.length - 1].id;
|
|
1154
|
+
if (maxId <= MAX_REPAIR_OBJECTS) {
|
|
1155
|
+
const prefix = buffer.length > 0 && buffer[buffer.length - 1] === 0x0a ? "" : "\n";
|
|
1156
|
+
const xrefOffset = buffer.length + prefix.length;
|
|
1157
|
+
let xrefStr = `${prefix}xref\n0 ${maxId + 1}\n0000000000 65535 f \n`;
|
|
1158
|
+
const objMap = new Map(objects.map((o) => [o.id, o]));
|
|
1159
|
+
for (let i = 1; i <= maxId; i++) {
|
|
1160
|
+
const obj = objMap.get(i);
|
|
1161
|
+
if (obj !== undefined) {
|
|
1162
|
+
const genStr = String(obj.generation).padStart(5, "0");
|
|
1163
|
+
xrefStr += `${String(obj.offset).padStart(10, "0")} ${genStr} n \n`;
|
|
1164
|
+
}
|
|
1165
|
+
else {
|
|
1166
|
+
xrefStr += `0000000000 65535 f \n`;
|
|
1167
|
+
}
|
|
1168
|
+
}
|
|
1169
|
+
// Search for /Root, /Encrypt, /ID in trailer dictionaries first (last trailer wins)
|
|
1170
|
+
let rootMatch = null;
|
|
1171
|
+
let encryptMatch = null;
|
|
1172
|
+
let idMatch = null;
|
|
1173
|
+
const trailerRegex = /trailer\s*<<([\s\S]*?)>>/g;
|
|
1174
|
+
let trailerMatch;
|
|
1175
|
+
while ((trailerMatch = trailerRegex.exec(fileStr)) !== null) {
|
|
1176
|
+
const candidateRoot = trailerMatch[1]?.match(/\/Root\s+(\d+\s+\d+\s+R)/);
|
|
1177
|
+
if (candidateRoot)
|
|
1178
|
+
rootMatch = candidateRoot;
|
|
1179
|
+
const candidateEncrypt = trailerMatch[1]?.match(/\/Encrypt\s+(\d+\s+\d+\s+R)/);
|
|
1180
|
+
if (candidateEncrypt)
|
|
1181
|
+
encryptMatch = candidateEncrypt;
|
|
1182
|
+
const candidateId = trailerMatch[1]?.match(/\/ID\s*(\[[^\]]*\])/);
|
|
1183
|
+
if (candidateId)
|
|
1184
|
+
idMatch = candidateId;
|
|
1185
|
+
}
|
|
1186
|
+
if (!rootMatch) {
|
|
1187
|
+
rootMatch = fileStr.match(/\/Root\s+(\d+\s+\d+\s+R)/);
|
|
1188
|
+
}
|
|
1189
|
+
if (!encryptMatch) {
|
|
1190
|
+
encryptMatch = fileStr.match(/\/Encrypt\s+(\d+\s+\d+\s+R)/);
|
|
1191
|
+
}
|
|
1192
|
+
if (!idMatch) {
|
|
1193
|
+
idMatch = fileStr.match(/\/ID\s*(\[[^\]]*\])/);
|
|
1194
|
+
}
|
|
1195
|
+
let extraTrailerEntries = "";
|
|
1196
|
+
if (rootMatch)
|
|
1197
|
+
extraTrailerEntries += ` /Root ${rootMatch[1]}`;
|
|
1198
|
+
if (encryptMatch)
|
|
1199
|
+
extraTrailerEntries += ` /Encrypt ${encryptMatch[1]}`;
|
|
1200
|
+
if (idMatch)
|
|
1201
|
+
extraTrailerEntries += ` /ID ${idMatch[1]}`;
|
|
1202
|
+
const trailerStr = `trailer\n<< /Size ${maxId + 1}${extraTrailerEntries} >>\nstartxref\n${xrefOffset}\n%%EOF\n`;
|
|
1203
|
+
return Buffer.concat([buffer, Buffer.from(xrefStr + trailerStr, "latin1")]);
|
|
1204
|
+
}
|
|
1205
|
+
}
|
|
1206
|
+
// 2. Opportunistic qpdf if installed
|
|
1207
|
+
const externalRepaired = await runExternalBinaryTool("qpdf", ["--qdf", "-", "-"], buffer, timeoutMs);
|
|
1208
|
+
if (externalRepaired) {
|
|
1209
|
+
return externalRepaired;
|
|
1210
|
+
}
|
|
1211
|
+
return buffer;
|
|
1212
|
+
}
|
|
1213
|
+
//# sourceMappingURL=pdf.js.map
|