scoutline 0.19.7 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/dist/command-invocation.d.ts.map +1 -1
  2. package/dist/command-invocation.js +5 -0
  3. package/dist/command-invocation.js.map +1 -1
  4. package/dist/commands/archive.d.ts +111 -0
  5. package/dist/commands/archive.d.ts.map +1 -0
  6. package/dist/commands/archive.js +426 -0
  7. package/dist/commands/archive.js.map +1 -0
  8. package/dist/commands/doctor.d.ts +8 -0
  9. package/dist/commands/doctor.d.ts.map +1 -1
  10. package/dist/commands/doctor.js +39 -4
  11. package/dist/commands/doctor.js.map +1 -1
  12. package/dist/commands/fetch.d.ts +75 -0
  13. package/dist/commands/fetch.d.ts.map +1 -0
  14. package/dist/commands/fetch.js +592 -0
  15. package/dist/commands/fetch.js.map +1 -0
  16. package/dist/commands/read.d.ts.map +1 -1
  17. package/dist/commands/read.js +14 -0
  18. package/dist/commands/read.js.map +1 -1
  19. package/dist/index.d.ts +2 -1
  20. package/dist/index.d.ts.map +1 -1
  21. package/dist/index.js +92 -7
  22. package/dist/index.js.map +1 -1
  23. package/dist/lib/artifacts.d.ts +5 -1
  24. package/dist/lib/artifacts.d.ts.map +1 -1
  25. package/dist/lib/artifacts.js +7 -2
  26. package/dist/lib/artifacts.js.map +1 -1
  27. package/dist/lib/async-file-lock.d.ts +2 -2
  28. package/dist/lib/async-file-lock.d.ts.map +1 -1
  29. package/dist/lib/async-file-lock.js +7 -4
  30. package/dist/lib/async-file-lock.js.map +1 -1
  31. package/dist/lib/pdf.d.ts +22 -0
  32. package/dist/lib/pdf.d.ts.map +1 -0
  33. package/dist/lib/pdf.js +1213 -0
  34. package/dist/lib/pdf.js.map +1 -0
  35. package/dist/lib/tty.d.ts.map +1 -1
  36. package/dist/lib/tty.js +10 -1
  37. package/dist/lib/tty.js.map +1 -1
  38. package/package.json +1 -1
@@ -0,0 +1,1213 @@
1
+ /**
2
+ * PDF text extraction and structural repair module (ADR-0006).
3
+ *
4
+ * Hybrid architecture:
5
+ * 1. Opportunistic system delegation to `pdftotext` and `qpdf` when installed.
6
+ * 2. Self-contained pure-Node fallback using zlib stream inflation and
7
+ * PDF text operator extraction (BT/ET, Tj, TJ, Td, hex strings) so Scoutline
8
+ * runs without mandatory system dependencies.
9
+ */
10
+ import * as zlib from "node:zlib";
11
+ import { spawn } from "node:child_process";
12
+ /**
13
+ * Check whether a buffer begins with a valid PDF header (%PDF-).
14
+ */
15
+ export function isPdfBuffer(buffer) {
16
+ if (!buffer || buffer.length < 5)
17
+ return false;
18
+ const header = buffer.subarray(0, 1024).toString("latin1").trimStart();
19
+ return header.startsWith("%PDF-");
20
+ }
21
+ /**
22
+ * Decode a PDF literal string (...), handling escape sequences.
23
+ */
24
+ function decodePdfLiteralString(str) {
25
+ let result = "";
26
+ let i = 0;
27
+ while (i < str.length) {
28
+ if (str[i] === "\\") {
29
+ i++;
30
+ if (i >= str.length)
31
+ break;
32
+ const ch = str[i];
33
+ if (ch === "n") {
34
+ result += "\n";
35
+ i++;
36
+ }
37
+ else if (ch === "r") {
38
+ result += "\r";
39
+ i++;
40
+ }
41
+ else if (ch === "t") {
42
+ result += "\t";
43
+ i++;
44
+ }
45
+ else if (ch === "b") {
46
+ result += "\b";
47
+ i++;
48
+ }
49
+ else if (ch === "f") {
50
+ result += "\f";
51
+ i++;
52
+ }
53
+ else if (ch === "(" || ch === ")" || ch === "\\") {
54
+ result += ch;
55
+ i++;
56
+ }
57
+ else if (/[0-7]/.test(ch)) {
58
+ // Octal escape \ddd (1 to 3 octal digits)
59
+ let octalStr = ch;
60
+ i++;
61
+ if (i < str.length && /[0-7]/.test(str[i])) {
62
+ octalStr += str[i];
63
+ i++;
64
+ if (i < str.length && /[0-7]/.test(str[i])) {
65
+ octalStr += str[i];
66
+ i++;
67
+ }
68
+ }
69
+ result += String.fromCharCode(parseInt(octalStr, 8));
70
+ }
71
+ else if (ch === "\r" || ch === "\n") {
72
+ // Line continuation
73
+ if (ch === "\r" && i + 1 < str.length && str[i + 1] === "\n")
74
+ i++;
75
+ i++;
76
+ }
77
+ else {
78
+ result += ch;
79
+ i++;
80
+ }
81
+ }
82
+ else {
83
+ result += str[i];
84
+ i++;
85
+ }
86
+ }
87
+ return result;
88
+ }
89
+ /**
90
+ * Decode a PDF hex string <...>. Returns the decoded text plus a
91
+ * `unmappable` flag: hex bytes that are neither a UTF-16BE BOM sequence
92
+ * nor the 2-byte Identity-H ASCII pattern cannot be authoritatively
93
+ * decoded without the font's ToUnicode/encoding context, so callers
94
+ * treat a document containing them as low-confidence and prefer the
95
+ * external text layer (`pdftotext`) when available.
96
+ */
97
+ function decodePdfHexString(hex) {
98
+ const clean = hex.replace(/\s+/g, "");
99
+ const padded = clean.length % 2 !== 0 ? clean + "0" : clean;
100
+ const bytes = [];
101
+ for (let i = 0; i < padded.length; i += 2) {
102
+ const byte = parseInt(padded.slice(i, i + 2), 16);
103
+ if (!isNaN(byte)) {
104
+ bytes.push(byte);
105
+ }
106
+ }
107
+ // Check if hex is UTF-16BE with BOM (\xFE\xFF)
108
+ if (bytes.length >= 2 && bytes[0] === 0xfe && bytes[1] === 0xff) {
109
+ let res = "";
110
+ for (let i = 2; i + 1 < bytes.length; i += 2) {
111
+ const code = (bytes[i] << 8) | bytes[i + 1];
112
+ res += String.fromCharCode(code);
113
+ }
114
+ return { text: res, unmappable: false };
115
+ }
116
+ // Check if 2-byte Identity-H ASCII (e.g. <00480069> -> "Hi")
117
+ if (bytes.length >= 2 && bytes.length % 2 === 0) {
118
+ let isTwoByteAscii = true;
119
+ for (let i = 0; i < bytes.length; i += 2) {
120
+ if (bytes[i] !== 0x00 || bytes[i + 1] < 0x20 || bytes[i + 1] > 0x7e) {
121
+ isTwoByteAscii = false;
122
+ break;
123
+ }
124
+ }
125
+ if (isTwoByteAscii) {
126
+ let res = "";
127
+ for (let i = 0; i < bytes.length; i += 2) {
128
+ res += String.fromCharCode(bytes[i + 1]);
129
+ }
130
+ return { text: res, unmappable: false };
131
+ }
132
+ }
133
+ let result = "";
134
+ for (const byte of bytes) {
135
+ result += String.fromCharCode(byte);
136
+ }
137
+ // High bytes without a recognized 2-byte structure: likely CID or a
138
+ // simple-font encoding we have no table for.
139
+ const unmappable = bytes.some((b) => b > 0x7e);
140
+ return { text: result, unmappable };
141
+ }
142
+ /**
143
+ * Read a balanced PDF literal string starting at `start` (which must
144
+ * point at the opening "("). Nested parentheses are literal characters
145
+ * per the spec; escapes are decoded. Returns the decoded value, the
146
+ * index just past the closing ")", and an `unmappable` flag set when
147
+ * the decoded text contains high bytes — literal strings share the
148
+ * hex strings' limitation (no font/ToUnicode context), so callers mark
149
+ * the document low-confidence and prefer the external text layer.
150
+ */
151
+ function readLiteralStringAt(src, start) {
152
+ let depth = 0;
153
+ let i = start;
154
+ let raw = "";
155
+ while (i < src.length) {
156
+ const ch = src[i];
157
+ if (ch === "\\") {
158
+ raw += ch + (src[i + 1] ?? "");
159
+ i += 2;
160
+ continue;
161
+ }
162
+ if (ch === "(") {
163
+ depth++;
164
+ if (depth > 1)
165
+ raw += ch;
166
+ i++;
167
+ continue;
168
+ }
169
+ if (ch === ")") {
170
+ depth--;
171
+ if (depth === 0) {
172
+ i++;
173
+ break;
174
+ }
175
+ raw += ch;
176
+ i++;
177
+ continue;
178
+ }
179
+ raw += ch;
180
+ i++;
181
+ }
182
+ const value = decodePdfLiteralString(raw);
183
+ return { value, next: i, unmappable: /[^\x00-\x7e]/.test(value) };
184
+ }
185
+ /**
186
+ * Read a PDF number at `start` (integer or real, optional sign, leading
187
+ * dot like `-.5` and trailing dot like `4.` per the PDF grammar).
188
+ */
189
+ function readNumberAt(src, start) {
190
+ let i = start;
191
+ if (src[i] === "+" || src[i] === "-")
192
+ i++;
193
+ while (i < src.length && /[0-9.]/.test(src[i]))
194
+ i++;
195
+ const value = Number(src.slice(start, i));
196
+ return { value: isNaN(value) ? 0 : value, next: i };
197
+ }
198
+ /**
199
+ * Extract BT ... ET text blocks from a content stream, respecting literal strings
200
+ * and comments so that "ET" occurring inside a string does not prematurely terminate the block.
201
+ */
202
+ function extractBtBlocks(streamStr) {
203
+ const blocks = [];
204
+ const len = streamStr.length;
205
+ let i = 0;
206
+ let outerInComment = false;
207
+ let outerInString = false;
208
+ let outerParenDepth = 0;
209
+ let outerInHex = false;
210
+ while (i < len) {
211
+ const ch = streamStr[i];
212
+ if (outerInComment) {
213
+ if (ch === "\r" || ch === "\n")
214
+ outerInComment = false;
215
+ i++;
216
+ continue;
217
+ }
218
+ if (outerInHex) {
219
+ if (ch === ">")
220
+ outerInHex = false;
221
+ i++;
222
+ continue;
223
+ }
224
+ if (outerInString) {
225
+ if (ch === "\\") {
226
+ i += 2;
227
+ }
228
+ else if (ch === "(") {
229
+ outerParenDepth++;
230
+ i++;
231
+ }
232
+ else if (ch === ")") {
233
+ outerParenDepth--;
234
+ if (outerParenDepth === 0)
235
+ outerInString = false;
236
+ i++;
237
+ }
238
+ else {
239
+ i++;
240
+ }
241
+ continue;
242
+ }
243
+ if (ch === "%") {
244
+ outerInComment = true;
245
+ i++;
246
+ continue;
247
+ }
248
+ if (ch === "<") {
249
+ // Consume both characters of a dictionary opener: advancing one
250
+ // char at a time would make the second "<" look like a lone
251
+ // hex-string opener and mask the dictionary body.
252
+ if (streamStr[i + 1] === "<") {
253
+ i += 2;
254
+ continue;
255
+ }
256
+ outerInHex = true;
257
+ i++;
258
+ continue;
259
+ }
260
+ if (ch === "(") {
261
+ outerInString = true;
262
+ outerParenDepth = 1;
263
+ i++;
264
+ continue;
265
+ }
266
+ if ((i === 0 || /[\s\[\]<>()/%]/.test(streamStr[i - 1])) &&
267
+ streamStr.slice(i, i + 2) === "BT" &&
268
+ (i + 2 === len || /[\s\[\]<>()/%]/.test(streamStr[i + 2]))) {
269
+ i += 2;
270
+ const start = i;
271
+ let inString = false;
272
+ let parenDepth = 0;
273
+ let inHex = false;
274
+ let inComment = false;
275
+ while (i < len) {
276
+ const c = streamStr[i];
277
+ if (inComment) {
278
+ if (c === "\r" || c === "\n")
279
+ inComment = false;
280
+ }
281
+ else if (inHex) {
282
+ if (c === ">")
283
+ inHex = false;
284
+ }
285
+ else if (inString) {
286
+ if (c === "\\") {
287
+ i++;
288
+ }
289
+ else if (c === "(") {
290
+ parenDepth++;
291
+ }
292
+ else if (c === ")") {
293
+ parenDepth--;
294
+ if (parenDepth === 0)
295
+ inString = false;
296
+ }
297
+ }
298
+ else {
299
+ if (c === "%") {
300
+ inComment = true;
301
+ }
302
+ else if (c === "<" && streamStr[i + 1] === "<") {
303
+ i++; // first "<" of "<<"; the shared tail i++ consumes the second
304
+ }
305
+ else if (c === "<") {
306
+ inHex = true;
307
+ }
308
+ else if (c === "(") {
309
+ inString = true;
310
+ parenDepth = 1;
311
+ }
312
+ else if ((i === 0 || /[\s\[\]<>()/%]/.test(streamStr[i - 1])) &&
313
+ streamStr.slice(i, i + 2) === "ET" &&
314
+ (i + 2 === len || /[\s\[\]<>()/%]/.test(streamStr[i + 2]))) {
315
+ blocks.push(streamStr.slice(start, i));
316
+ i += 2;
317
+ break;
318
+ }
319
+ }
320
+ i++;
321
+ }
322
+ }
323
+ else {
324
+ i++;
325
+ }
326
+ }
327
+ return blocks;
328
+ }
329
+ /**
330
+ * Parse text operators from a decompressed PDF content stream using a
331
+ * tokenizer rather than operand regexes: balanced literal strings
332
+ * (arbitrary nesting), hex strings, PDF reals (signed, leading/trailing
333
+ * dot), TJ arrays, and the full set of text-showing operators (Tj, TJ,
334
+ * ', ") are recognized so valid content is not silently dropped.
335
+ * `ctx.sawUnmappable` is set when a hex or literal string cannot be
336
+ * decoded with confidence (CID/unknown encoding, high bytes) so the
337
+ * caller can prefer the external text layer.
338
+ */
339
+ function parseContentStreamText(streamStr, ctx) {
340
+ const lines = [];
341
+ let currentLine = "";
342
+ const pushLine = () => {
343
+ if (currentLine.trim().length > 0) {
344
+ lines.push(currentLine.trim());
345
+ currentLine = "";
346
+ }
347
+ };
348
+ const appendText = (value) => {
349
+ if (value.length > 0)
350
+ currentLine += value;
351
+ };
352
+ for (const block of extractBtBlocks(streamStr)) {
353
+ const stack = [];
354
+ const readHexStringOperand = (index) => {
355
+ const end = block.indexOf(">", index);
356
+ const close = end === -1 ? block.length : end;
357
+ const decoded = decodePdfHexString(block.slice(index + 1, close));
358
+ if (decoded.unmappable)
359
+ ctx.sawUnmappable = true;
360
+ return {
361
+ operand: { type: "hex", value: decoded.text, unmappable: decoded.unmappable },
362
+ next: end === -1 ? block.length : end + 1,
363
+ };
364
+ };
365
+ let i = 0;
366
+ while (i < block.length) {
367
+ const ch = block[i];
368
+ if (/\s/.test(ch)) {
369
+ i++;
370
+ continue;
371
+ }
372
+ if (ch === "%") {
373
+ while (i < block.length && block[i] !== "\r" && block[i] !== "\n")
374
+ i++;
375
+ continue;
376
+ }
377
+ if (ch === "(") {
378
+ const { value, next, unmappable } = readLiteralStringAt(block, i);
379
+ if (unmappable)
380
+ ctx.sawUnmappable = true;
381
+ stack.push({ type: "str", value });
382
+ i = next;
383
+ continue;
384
+ }
385
+ if (ch === "<" && block[i + 1] === "<") {
386
+ // Inline dictionary: skip with bracket balance.
387
+ let depth = 1;
388
+ i += 2;
389
+ while (i < block.length && depth > 0) {
390
+ if (block.slice(i, i + 2) === "<<") {
391
+ depth++;
392
+ i += 2;
393
+ }
394
+ else if (block.slice(i, i + 2) === ">>") {
395
+ depth--;
396
+ i += 2;
397
+ }
398
+ else {
399
+ i++;
400
+ }
401
+ }
402
+ continue;
403
+ }
404
+ if (ch === "<") {
405
+ const { operand, next } = readHexStringOperand(i);
406
+ stack.push(operand);
407
+ i = next;
408
+ continue;
409
+ }
410
+ if (ch === "[") {
411
+ // TJ array: strings and kerning numbers until the closing "]".
412
+ const items = [];
413
+ i++;
414
+ while (i < block.length && block[i] !== "]") {
415
+ const c = block[i];
416
+ if (/\s/.test(c)) {
417
+ i++;
418
+ }
419
+ else if (c === "(") {
420
+ const { value, next, unmappable } = readLiteralStringAt(block, i);
421
+ if (unmappable)
422
+ ctx.sawUnmappable = true;
423
+ items.push({ type: "str", value });
424
+ i = next;
425
+ }
426
+ else if (c === "<") {
427
+ const end = block.indexOf(">", i);
428
+ const close = end === -1 ? block.length : end;
429
+ const decoded = decodePdfHexString(block.slice(i + 1, close));
430
+ if (decoded.unmappable)
431
+ ctx.sawUnmappable = true;
432
+ items.push({ type: "hex", value: decoded.text, unmappable: decoded.unmappable });
433
+ i = end === -1 ? block.length : end + 1;
434
+ }
435
+ else if (/[0-9+.\-]/.test(c)) {
436
+ const { value, next } = readNumberAt(block, i);
437
+ items.push({ type: "num", value });
438
+ i = next;
439
+ }
440
+ else {
441
+ i++;
442
+ }
443
+ }
444
+ i++; // consume "]"
445
+ stack.push({ type: "array", items });
446
+ continue;
447
+ }
448
+ if (/[0-9+.\-]/.test(ch)) {
449
+ const { value, next } = readNumberAt(block, i);
450
+ stack.push({ type: "num", value });
451
+ i = next;
452
+ continue;
453
+ }
454
+ // Operator keyword (letters, apostrophe, double quote, asterisk).
455
+ let j = i;
456
+ while (j < block.length && /[A-Za-z'*"]/.test(block[j]))
457
+ j++;
458
+ const op = block.slice(i, j);
459
+ i = j;
460
+ if (op.length === 0) {
461
+ i++;
462
+ continue;
463
+ }
464
+ const top = stack[stack.length - 1];
465
+ switch (op) {
466
+ case "Tj":
467
+ // Successive Tj operators continue at the current text
468
+ // position: no injected space between operands.
469
+ if (top && (top.type === "str" || top.type === "hex"))
470
+ appendText(top.value);
471
+ break;
472
+ case "TJ":
473
+ if (top?.type === "array") {
474
+ for (const item of top.items) {
475
+ if (item.type === "str" || item.type === "hex")
476
+ appendText(item.value);
477
+ }
478
+ }
479
+ break;
480
+ case "T*":
481
+ case "Td":
482
+ case "TD":
483
+ pushLine();
484
+ break;
485
+ case "'":
486
+ case '"': {
487
+ // Move to the next line and show text: aw/ac operands sit
488
+ // below the string on the operand stack.
489
+ pushLine();
490
+ if (top && (top.type === "str" || top.type === "hex"))
491
+ appendText(top.value + " ");
492
+ break;
493
+ }
494
+ default:
495
+ break; // operators without direct text semantics
496
+ }
497
+ // Operators consume their operands: a hostile BT block with
498
+ // thousands of no-op operators must not accumulate them on the
499
+ // stack until the heap is exhausted.
500
+ stack.length = 0;
501
+ }
502
+ pushLine();
503
+ }
504
+ return lines.join("\n");
505
+ }
506
+ /**
507
+ * Parse a stream dictionary's /Filter entry into an ordered chain.
508
+ * Handles both `/Filter /Name` and `/Filter [/A /B]` forms; an absent
509
+ * entry yields an empty chain (unfiltered).
510
+ */
511
+ function parseFilterChain(dictSlice) {
512
+ const arrayMatch = /\/Filter\s*\[([^\]]*)\]/.exec(dictSlice);
513
+ if (arrayMatch) {
514
+ return (arrayMatch[1].match(/\/[A-Za-z0-9]+/g) ?? []).map((name) => name.slice(1));
515
+ }
516
+ const single = /\/Filter\s*(\/[A-Za-z0-9]+)/.exec(dictSlice);
517
+ return single ? [single[1].slice(1)] : [];
518
+ }
519
+ /**
520
+ * Heuristic content-role check: streams whose dictionary marks them as
521
+ * embedded images, object/metadata containers, or font programs are
522
+ * not page content and must not contribute text.
523
+ */
524
+ function isNonContentDictionary(dictSlice) {
525
+ return /\/Subtype\s*\/Image\b|\/Type\s*\/ObjStm\b|\/Type\s*\/Metadata\b|\/Length1\b|\/FontFile\d?\b/.test(dictSlice);
526
+ }
527
+ /**
528
+ * Locate a stream's own dictionary by balancing brackets backward from
529
+ * the ">>" nearest the keyword (within the window). Imperfect for ">>"
530
+ * inside dictionary strings, but /Length sits early in real stream
531
+ * dictionaries and resolveStreamInterval validates before trusting it.
532
+ */
533
+ function scanBackwardDictionary(fileStr, keywordStart, window = 2000) {
534
+ const beforeStream = fileStr.slice(Math.max(0, keywordStart - window), keywordStart).trimEnd();
535
+ const lastDictEnd = beforeStream.lastIndexOf(">>");
536
+ if (lastDictEnd === -1)
537
+ return "";
538
+ let depth = 1;
539
+ let pos = lastDictEnd;
540
+ while (pos > 1) {
541
+ if (beforeStream.slice(pos - 2, pos) === ">>") {
542
+ depth++;
543
+ pos -= 2;
544
+ }
545
+ else if (beforeStream.slice(pos - 2, pos) === "<<") {
546
+ depth--;
547
+ pos -= 2;
548
+ if (depth === 0) {
549
+ return beforeStream.slice(pos, lastDictEnd + 2);
550
+ }
551
+ }
552
+ else {
553
+ pos--;
554
+ }
555
+ }
556
+ return "";
557
+ }
558
+ /**
559
+ * Find an indirect-object header (`N G obj`) with a lexical scan that
560
+ * skips comments, literal/hex strings, and stream bodies — a header
561
+ * lookalike inside stream data must never be selected. Stream bodies
562
+ * are skipped with length-VALIDATED boundaries (`resolving` guards
563
+ * against indirect-length resolution cycles).
564
+ */
565
+ function findObjectHeaderOffset(fileStr, header, resolving = new Set()) {
566
+ let i = 0;
567
+ while (i < fileStr.length) {
568
+ const ch = fileStr[i];
569
+ if (ch === "%") {
570
+ while (i < fileStr.length && fileStr[i] !== "\r" && fileStr[i] !== "\n")
571
+ i++;
572
+ continue;
573
+ }
574
+ if (ch === "(") {
575
+ let depth = 1;
576
+ i++;
577
+ while (i < fileStr.length && depth > 0) {
578
+ const c = fileStr[i];
579
+ if (c === "\\") {
580
+ i += 2;
581
+ continue;
582
+ }
583
+ if (c === "(")
584
+ depth++;
585
+ else if (c === ")")
586
+ depth--;
587
+ i++;
588
+ }
589
+ continue;
590
+ }
591
+ if (ch === "<" && fileStr[i + 1] !== "<") {
592
+ const gt = fileStr.indexOf(">", i);
593
+ i = gt === -1 ? fileStr.length : gt + 1;
594
+ continue;
595
+ }
596
+ if (fileStr.startsWith("stream", i) &&
597
+ (i === 0 || fileStr.slice(Math.max(0, i - 3), i) !== "end") &&
598
+ /[\r\n]/.test(fileStr[i + 6] ?? "\n")) {
599
+ const interval = resolveStreamInterval(fileStr, i, maskStringsAndComments(scanBackwardDictionary(fileStr, i)), resolving);
600
+ i = interval.resumeAfter;
601
+ continue;
602
+ }
603
+ if (fileStr.startsWith(header, i)) {
604
+ const after = fileStr[i + header.length];
605
+ if (after === undefined || /[\s\r\n<>%]/.test(after))
606
+ return i;
607
+ }
608
+ i++;
609
+ }
610
+ return -1;
611
+ }
612
+ /**
613
+ * Resolve an indirect /Length (N G R) by reading the integer the
614
+ * referenced object holds. Returns null when the reference or object
615
+ * cannot be resolved.
616
+ */
617
+ function resolveIndirectLength(fileStr, maskedDict, resolving = new Set()) {
618
+ const ref = /\/Length\s+(\d+)\s+(\d+)\s+R\b/.exec(maskedDict);
619
+ if (!ref)
620
+ return null;
621
+ // Recursion guard: a length object whose own stream references back
622
+ // must not loop (A -> B -> A); give up and fall back instead.
623
+ const refKey = `${ref[1]} ${ref[2]}`;
624
+ if (resolving.has(refKey))
625
+ return null;
626
+ const header = `${refKey} obj`;
627
+ const at = findObjectHeaderOffset(fileStr, header, new Set([...resolving, refKey]));
628
+ if (at === -1)
629
+ return null;
630
+ let p = at + header.length;
631
+ while (p < fileStr.length) {
632
+ if (/\s/.test(fileStr[p])) {
633
+ p++;
634
+ continue;
635
+ }
636
+ // The integer may be preceded by comments inside the object.
637
+ if (fileStr[p] === "%") {
638
+ while (p < fileStr.length && fileStr[p] !== "\r" && fileStr[p] !== "\n")
639
+ p++;
640
+ continue;
641
+ }
642
+ break;
643
+ }
644
+ const num = /^(\d+)/.exec(fileStr.slice(p, p + 16));
645
+ return num ? Number(num[1]) : null;
646
+ }
647
+ /**
648
+ * Resolve a stream body interval: [bodyStart, dataEnd) plus the index
649
+ * to resume scanning from (past the real endstream keyword). Primary
650
+ * signal is the dictionary /Length, VALIDATED against a following
651
+ * endstream keyword, so embedded "endstream" bytes inside stream data
652
+ * cannot truncate the stream or leak stream interiors into syntax
653
+ * scans. The textual search is only a fallback for absent or indirect
654
+ * (/Length 12 0 R) or non-validating lengths.
655
+ */
656
+ function resolveStreamInterval(fileStr, keywordStart, dictSlice, resolving = new Set()) {
657
+ let bodyStart = keywordStart + "stream".length;
658
+ if (fileStr[bodyStart] === "\r")
659
+ bodyStart++;
660
+ if (fileStr[bodyStart] === "\n")
661
+ bodyStart++;
662
+ // Textual fallback: accept an `endstream` keyword only when it is
663
+ // followed by `endobj` (or end-of-file adjacency) — an embedded
664
+ // `endstream` token inside stream data is typically followed by more
665
+ // data, never by the object terminator, so this keeps the fallback
666
+ // from resuming mid-stream.
667
+ let keywordAt = -1;
668
+ for (let probe = fileStr.indexOf("endstream", bodyStart); probe !== -1; probe = fileStr.indexOf("endstream", probe + 9)) {
669
+ if (/^endstream[\s\S]{0,4}?endobj\b/.test(fileStr.slice(probe, probe + 60))) {
670
+ keywordAt = probe;
671
+ break;
672
+ }
673
+ }
674
+ let dataEnd = keywordAt === -1 ? fileStr.length : keywordAt;
675
+ // Direct /Length N (the lookahead avoids reading the object number of
676
+ // an indirect /Length N G R as a direct length); indirect lengths are
677
+ // resolved from their integer object so generators that use them do
678
+ // not fall back to a textual endstream search that can stop at an
679
+ // embedded "endstream" sequence inside the stream body.
680
+ const lengthMatch = /\/Length\s+(\d+)(?!\d)(?!\s+\d+\s+R)/.exec(dictSlice);
681
+ const declared = lengthMatch !== null
682
+ ? Number(lengthMatch[1])
683
+ : resolveIndirectLength(fileStr, dictSlice, resolving);
684
+ if (declared !== null) {
685
+ let probe = bodyStart + declared;
686
+ if (fileStr[probe] === "\r")
687
+ probe++;
688
+ if (fileStr[probe] === "\n")
689
+ probe++;
690
+ if (fileStr.startsWith("endstream", probe)) {
691
+ dataEnd = bodyStart + declared;
692
+ keywordAt = probe;
693
+ }
694
+ }
695
+ const resumeAfter = keywordAt === -1 ? fileStr.length : keywordAt + "endstream".length;
696
+ return { bodyStart, dataEnd, resumeAfter };
697
+ }
698
+ /**
699
+ * Read the dictionary that starts at/after `i` (leading whitespace
700
+ * tolerated), balancing << >> while skipping literal strings, hex
701
+ * strings, and comments so delimiters inside string data cannot
702
+ * unbalance the walk. Returns the dictionary text and the index just
703
+ * past its closing ">>" (or end of input when unterminated).
704
+ */
705
+ function readObjectDictionaryAt(fileStr, i) {
706
+ let p = i;
707
+ while (p < fileStr.length) {
708
+ if (/\s/.test(fileStr[p])) {
709
+ p++;
710
+ continue;
711
+ }
712
+ // Object headers may carry comments before the dictionary opener.
713
+ if (fileStr[p] === "%") {
714
+ while (p < fileStr.length && fileStr[p] !== "\r" && fileStr[p] !== "\n")
715
+ p++;
716
+ continue;
717
+ }
718
+ break;
719
+ }
720
+ if (fileStr.slice(p, p + 2) !== "<<")
721
+ return { text: "", next: i };
722
+ const start = p;
723
+ let depth = 1;
724
+ p += 2;
725
+ while (p < fileStr.length && depth > 0) {
726
+ const ch = fileStr[p];
727
+ if (ch === "%") {
728
+ while (p < fileStr.length && fileStr[p] !== "\r" && fileStr[p] !== "\n")
729
+ p++;
730
+ }
731
+ else if (ch === "(") {
732
+ let sDepth = 1;
733
+ p++;
734
+ while (p < fileStr.length && sDepth > 0) {
735
+ const c = fileStr[p];
736
+ if (c === "\\") {
737
+ p += 2;
738
+ continue;
739
+ }
740
+ if (c === "(")
741
+ sDepth++;
742
+ else if (c === ")")
743
+ sDepth--;
744
+ p++;
745
+ }
746
+ }
747
+ else if (ch === "<" && fileStr[p + 1] !== "<") {
748
+ const gt = fileStr.indexOf(">", p);
749
+ p = gt === -1 ? fileStr.length : gt + 1;
750
+ }
751
+ else if (fileStr.slice(p, p + 2) === "<<") {
752
+ depth++;
753
+ p += 2;
754
+ }
755
+ else if (fileStr.slice(p, p + 2) === ">>") {
756
+ depth--;
757
+ p += 2;
758
+ }
759
+ else {
760
+ p++;
761
+ }
762
+ }
763
+ return { text: fileStr.slice(start, p), next: p };
764
+ }
765
+ /**
766
+ * Built-in pure-Node text extraction by scanning PDF streams. Returns
767
+ * the text plus a low-confidence flag set when any hex string could not
768
+ * be decoded with font context (CID/unknown encoding) — such text is a
769
+ * Latin-1 best effort, and callers should prefer the external layer.
770
+ */
771
+ function extractPureNodePdfText(buffer) {
772
+ const fileStr = buffer.toString("latin1");
773
+ const extractedSections = [];
774
+ const ctx = { sawUnmappable: false };
775
+ const MAX_TOTAL_DECOMPRESSED_BYTES = 50 * 1024 * 1024; // 50MB across document
776
+ let totalDecompressedBytes = 0;
777
+ // Stream keyword scan; body intervals come from the dictionary
778
+ // /Length via resolveStreamInterval (validated against the real
779
+ // endstream keyword), so embedded "endstream" bytes inside stream
780
+ // data cannot truncate a stream. The "end" prefix guard keeps the
781
+ // tail of "endstream" keywords from matching as stream starts.
782
+ // Lexical stream scan: literal strings and comments are skipped
783
+ // while walking, so a `stream` token inside string or comment data
784
+ // can never corrupt the next stream interval.
785
+ let cursor = 0;
786
+ while (true) {
787
+ // Advance to the next stream keyword outside strings/comments.
788
+ let keywordStart = -1;
789
+ let i = cursor;
790
+ while (i < fileStr.length) {
791
+ const ch = fileStr[i];
792
+ if (ch === "%") {
793
+ while (i < fileStr.length && fileStr[i] !== "\r" && fileStr[i] !== "\n")
794
+ i++;
795
+ continue;
796
+ }
797
+ if (ch === "(") {
798
+ let depth = 1;
799
+ i++;
800
+ while (i < fileStr.length && depth > 0) {
801
+ const c = fileStr[i];
802
+ if (c === "\\") {
803
+ i += 2;
804
+ continue;
805
+ }
806
+ if (c === "(")
807
+ depth++;
808
+ else if (c === ")")
809
+ depth--;
810
+ i++;
811
+ }
812
+ continue;
813
+ }
814
+ if (fileStr.startsWith("stream", i) &&
815
+ (i === 0 || fileStr.slice(Math.max(0, i - 3), i) !== "end") &&
816
+ /[\r\n]/.test(fileStr[i + 6] ?? "\n")) {
817
+ keywordStart = i;
818
+ break;
819
+ }
820
+ i++;
821
+ }
822
+ if (keywordStart === -1)
823
+ break;
824
+ const dictSlice = maskStringsAndComments(scanBackwardDictionary(fileStr, keywordStart));
825
+ const interval = resolveStreamInterval(fileStr, keywordStart, dictSlice);
826
+ cursor = interval.resumeAfter;
827
+ const rawStreamSlice = buffer.subarray(interval.bodyStart, interval.dataEnd);
828
+ // Filter-chain classification (R6-C): the pure path can only
829
+ // decode an unfiltered stream or a single FlateDecode. Anything
830
+ // else (e.g. [/ASCII85Decode /FlateDecode]) marks the document
831
+ // low-confidence instead of silently skipping the stream while a
832
+ // nonempty partial result suppresses pdftotext. The dictionary is
833
+ // masked so /Filter or marker text inside strings/comments cannot
834
+ // impersonate the real entries.
835
+ const filterChain = parseFilterChain(dictSlice);
836
+ const flateOnly = filterChain.length === 1 && filterChain[0] === "FlateDecode";
837
+ // Non-content streams (embedded images, object/metadata containers,
838
+ // font programs) must not contribute text even when their bytes
839
+ // look like content syntax. Full page-resource-graph resolution is
840
+ // out of scope for the pure path; dictionary markers cover the
841
+ // common non-content families.
842
+ if (isNonContentDictionary(dictSlice)) {
843
+ continue;
844
+ }
845
+ let streamBytes = null;
846
+ if (flateOnly) {
847
+ if (totalDecompressedBytes >= MAX_TOTAL_DECOMPRESSED_BYTES) {
848
+ break; // Document-wide decompression budget reached
849
+ }
850
+ const maxStreamOutput = Math.min(20 * 1024 * 1024, MAX_TOTAL_DECOMPRESSED_BYTES - totalDecompressedBytes);
851
+ try {
852
+ streamBytes = zlib.inflateSync(rawStreamSlice, { maxOutputLength: maxStreamOutput });
853
+ totalDecompressedBytes += streamBytes.length;
854
+ }
855
+ catch {
856
+ try {
857
+ streamBytes = zlib.inflateRawSync(rawStreamSlice, { maxOutputLength: maxStreamOutput });
858
+ totalDecompressedBytes += streamBytes.length;
859
+ }
860
+ catch {
861
+ // Stream may be raw or uncompressed
862
+ }
863
+ }
864
+ }
865
+ else if (filterChain.length === 0) {
866
+ streamBytes = rawStreamSlice;
867
+ }
868
+ else {
869
+ // Unsupported filter chain: do not attempt to inflate partially
870
+ // encoded bytes; defer to the external text layer.
871
+ ctx.sawUnmappable = true;
872
+ }
873
+ if (streamBytes) {
874
+ const decodedStr = streamBytes.toString("latin1");
875
+ const text = parseContentStreamText(decodedStr, ctx);
876
+ if (text.trim().length > 0) {
877
+ extractedSections.push(text.trim());
878
+ }
879
+ }
880
+ }
881
+ return { text: extractedSections.join("\n\n").trim(), lowConfidence: ctx.sawUnmappable };
882
+ }
883
+ // External tools process untrusted input; cap accumulated stdout so a
884
+ // hostile PDF cannot exhaust the heap through pdftotext/qpdf output.
885
+ const MAX_EXTERNAL_TOOL_OUTPUT_BYTES = 50 * 1024 * 1024;
886
+ function runExternalTextTool(cmd, args, input, timeoutMs = 5000) {
887
+ return new Promise((resolve) => {
888
+ let finished = false;
889
+ let timer = null;
890
+ const finish = (val) => {
891
+ if (!finished) {
892
+ finished = true;
893
+ if (timer)
894
+ clearTimeout(timer);
895
+ resolve(val);
896
+ }
897
+ };
898
+ try {
899
+ const child = spawn(cmd, args, { stdio: ["pipe", "pipe", "ignore"] });
900
+ timer = setTimeout(() => {
901
+ child.kill("SIGKILL");
902
+ finish(null);
903
+ }, timeoutMs);
904
+ let stdout = "";
905
+ let stdoutBytes = 0;
906
+ child.stdout.setEncoding("utf8");
907
+ child.stdout.on("data", (chunk) => {
908
+ stdoutBytes += Buffer.byteLength(chunk);
909
+ if (stdoutBytes > MAX_EXTERNAL_TOOL_OUTPUT_BYTES) {
910
+ child.kill("SIGKILL");
911
+ finish(null);
912
+ return;
913
+ }
914
+ stdout += chunk;
915
+ });
916
+ child.on("error", () => finish(null));
917
+ child.on("close", (code) => {
918
+ if (code === 0 && stdout.trim().length > 0) {
919
+ finish(stdout.trim());
920
+ }
921
+ else {
922
+ finish(null);
923
+ }
924
+ });
925
+ child.stdin.on("error", () => finish(null));
926
+ child.stdin.write(input);
927
+ child.stdin.end();
928
+ }
929
+ catch {
930
+ finish(null);
931
+ }
932
+ });
933
+ }
934
+ function runExternalBinaryTool(cmd, args, input, timeoutMs = 5000) {
935
+ return new Promise((resolve) => {
936
+ let finished = false;
937
+ let timer = null;
938
+ const finish = (val) => {
939
+ if (!finished) {
940
+ finished = true;
941
+ if (timer)
942
+ clearTimeout(timer);
943
+ resolve(val);
944
+ }
945
+ };
946
+ try {
947
+ const child = spawn(cmd, args, { stdio: ["pipe", "pipe", "ignore"] });
948
+ timer = setTimeout(() => {
949
+ child.kill("SIGKILL");
950
+ finish(null);
951
+ }, timeoutMs);
952
+ const chunks = [];
953
+ let stdoutBytes = 0;
954
+ child.stdout.on("data", (chunk) => {
955
+ stdoutBytes += chunk.length;
956
+ if (stdoutBytes > MAX_EXTERNAL_TOOL_OUTPUT_BYTES) {
957
+ child.kill("SIGKILL");
958
+ finish(null);
959
+ return;
960
+ }
961
+ chunks.push(chunk);
962
+ });
963
+ child.on("error", () => finish(null));
964
+ child.on("close", (code) => {
965
+ if (code === 0 && chunks.length > 0) {
966
+ finish(Buffer.concat(chunks));
967
+ }
968
+ else {
969
+ finish(null);
970
+ }
971
+ });
972
+ child.stdin.on("error", () => finish(null));
973
+ child.stdin.write(input);
974
+ child.stdin.end();
975
+ }
976
+ catch {
977
+ finish(null);
978
+ }
979
+ });
980
+ }
981
+ /**
982
+ * Extract text from a PDF buffer using opportunistic delegation or pure-Node fallback.
983
+ */
984
+ export async function extractPdfText(buffer, timeoutMs = 5000) {
985
+ // 1. Pure Node fast path. Low-confidence results (hex strings the
986
+ // pure path cannot decode with font context) defer to the external
987
+ // text layer when one is installed.
988
+ const pure = extractPureNodePdfText(buffer);
989
+ if (pure.text.length > 0 && !pure.lowConfidence) {
990
+ return pure.text;
991
+ }
992
+ // 2. Opportunistic pdftotext if installed
993
+ const externalText = await runExternalTextTool("pdftotext", ["-", "-"], buffer, timeoutMs);
994
+ if (externalText) {
995
+ return externalText;
996
+ }
997
+ // 3. No external tool available: serve the best-effort pure text.
998
+ if (pure.text.length > 0) {
999
+ return pure.text;
1000
+ }
1001
+ return "[PDF document with no extractable text layer or scanned raster pages]";
1002
+ }
1003
+ /**
1004
+ * Blank out literal strings, hex strings, and comments so textual
1005
+ * lookalikes inside string data cannot satisfy dictionary probes.
1006
+ */
1007
+ function maskStringsAndComments(src) {
1008
+ let out = "";
1009
+ let i = 0;
1010
+ while (i < src.length) {
1011
+ const ch = src[i];
1012
+ if (ch === "%") {
1013
+ let j = i;
1014
+ while (j < src.length && src[j] !== "\r" && src[j] !== "\n")
1015
+ j++;
1016
+ out += " ".repeat(j - i);
1017
+ i = j;
1018
+ continue;
1019
+ }
1020
+ if (ch === "(") {
1021
+ let depth = 1;
1022
+ let j = i + 1;
1023
+ while (j < src.length && depth > 0) {
1024
+ const c = src[j];
1025
+ if (c === "\\") {
1026
+ j += 2;
1027
+ continue;
1028
+ }
1029
+ if (c === "(")
1030
+ depth++;
1031
+ else if (c === ")")
1032
+ depth--;
1033
+ j++;
1034
+ }
1035
+ out += " ".repeat(Math.min(j, src.length) - i);
1036
+ i = j;
1037
+ continue;
1038
+ }
1039
+ if (ch === "<") {
1040
+ // Dictionary opener: contents must stay visible to probes —
1041
+ // do NOT fall through to the hex-string mask on the second "<".
1042
+ if (src[i + 1] === "<") {
1043
+ out += "<<";
1044
+ i += 2;
1045
+ continue;
1046
+ }
1047
+ const gt = src.indexOf(">", i);
1048
+ const j = gt === -1 ? src.length : gt + 1;
1049
+ out += " ".repeat(j - i);
1050
+ i = j;
1051
+ continue;
1052
+ }
1053
+ out += ch;
1054
+ i++;
1055
+ }
1056
+ return out;
1057
+ }
1058
+ /**
1059
+ * Attempt structural repair on damaged PDFs (e.g. broken xref table).
1060
+ */
1061
+ export async function repairPdf(buffer, timeoutMs = 5000) {
1062
+ // 1. Pure Node xref reconstruction
1063
+ const MAX_REPAIR_OBJECTS = 500_000;
1064
+ const fileStr = buffer.toString("latin1");
1065
+ // Lexically scan indirect-object headers OUTSIDE comments, literal and
1066
+ // hex strings, and stream bodies: object-like text inside string data
1067
+ // must never become an xref entry pointing into that string, and
1068
+ // duplicate ids must resolve to the real headers (later occurrences
1069
+ // win below, matching incremental-update semantics). /Type /ObjStm is
1070
+ // detected in an actual object dictionary, not anywhere in the file.
1071
+ const objects = [];
1072
+ let overLimit = false;
1073
+ let objStmDetected = false;
1074
+ {
1075
+ let i = 0;
1076
+ while (i < fileStr.length) {
1077
+ const ch = fileStr[i];
1078
+ if (ch === "%") {
1079
+ while (i < fileStr.length && fileStr[i] !== "\r" && fileStr[i] !== "\n")
1080
+ i++;
1081
+ continue;
1082
+ }
1083
+ if (ch === "(") {
1084
+ let depth = 1;
1085
+ i++;
1086
+ while (i < fileStr.length && depth > 0) {
1087
+ const c = fileStr[i];
1088
+ if (c === "\\") {
1089
+ i += 2;
1090
+ continue;
1091
+ }
1092
+ if (c === "(")
1093
+ depth++;
1094
+ else if (c === ")")
1095
+ depth--;
1096
+ i++;
1097
+ }
1098
+ continue;
1099
+ }
1100
+ if (ch === "<" && fileStr[i + 1] === "<") {
1101
+ // Dictionary: skip with string awareness — `>>` inside literal
1102
+ // or hex strings must not close the walk.
1103
+ i = readObjectDictionaryAt(fileStr, i).next;
1104
+ continue;
1105
+ }
1106
+ if (ch === "<") {
1107
+ const gt = fileStr.indexOf(">", i);
1108
+ i = gt === -1 ? fileStr.length : gt + 1;
1109
+ continue;
1110
+ }
1111
+ if (fileStr.startsWith("stream", i) &&
1112
+ (i === 0 || fileStr.slice(Math.max(0, i - 3), i) !== "end") &&
1113
+ /[\r\n]/.test(fileStr[i + 6] ?? "\n")) {
1114
+ // Skip the ENTIRE stream body using the dictionary /Length
1115
+ // (validated) so an embedded "endstream" sequence inside the
1116
+ // data cannot make the scanner resume mid-stream and index
1117
+ // object-like text living inside the stream.
1118
+ const interval = resolveStreamInterval(fileStr, i, maskStringsAndComments(scanBackwardDictionary(fileStr, i)));
1119
+ i = interval.resumeAfter;
1120
+ continue;
1121
+ }
1122
+ if (/[0-9]/.test(ch)) {
1123
+ const header = /^(\d+)\s+(\d+)\s+obj\b/.exec(fileStr.slice(i, i + 32));
1124
+ if (header) {
1125
+ const id = Number(header[1]);
1126
+ const generation = Number(header[2]);
1127
+ if (id > 0 && id <= MAX_REPAIR_OBJECTS) {
1128
+ objects.push({ id, generation, offset: i });
1129
+ if (objects.length > MAX_REPAIR_OBJECTS) {
1130
+ overLimit = true;
1131
+ break;
1132
+ }
1133
+ }
1134
+ i += header[0].length;
1135
+ if (!objStmDetected) {
1136
+ // Inspect the object's actual dictionary (no fixed byte
1137
+ // window) with strings masked: /Type /ObjStm separated from
1138
+ // the header by padding/comments must still be found, while
1139
+ // lookalikes inside string data must not match.
1140
+ const dict = readObjectDictionaryAt(fileStr, i);
1141
+ if (/\/Type\s*\/ObjStm\b/.test(maskStringsAndComments(dict.text))) {
1142
+ objStmDetected = true;
1143
+ }
1144
+ }
1145
+ continue;
1146
+ }
1147
+ }
1148
+ i++;
1149
+ }
1150
+ }
1151
+ if (objects.length > 0 && !overLimit && !objStmDetected) {
1152
+ objects.sort((a, b) => a.id - b.id);
1153
+ const maxId = objects[objects.length - 1].id;
1154
+ if (maxId <= MAX_REPAIR_OBJECTS) {
1155
+ const prefix = buffer.length > 0 && buffer[buffer.length - 1] === 0x0a ? "" : "\n";
1156
+ const xrefOffset = buffer.length + prefix.length;
1157
+ let xrefStr = `${prefix}xref\n0 ${maxId + 1}\n0000000000 65535 f \n`;
1158
+ const objMap = new Map(objects.map((o) => [o.id, o]));
1159
+ for (let i = 1; i <= maxId; i++) {
1160
+ const obj = objMap.get(i);
1161
+ if (obj !== undefined) {
1162
+ const genStr = String(obj.generation).padStart(5, "0");
1163
+ xrefStr += `${String(obj.offset).padStart(10, "0")} ${genStr} n \n`;
1164
+ }
1165
+ else {
1166
+ xrefStr += `0000000000 65535 f \n`;
1167
+ }
1168
+ }
1169
+ // Search for /Root, /Encrypt, /ID in trailer dictionaries first (last trailer wins)
1170
+ let rootMatch = null;
1171
+ let encryptMatch = null;
1172
+ let idMatch = null;
1173
+ const trailerRegex = /trailer\s*<<([\s\S]*?)>>/g;
1174
+ let trailerMatch;
1175
+ while ((trailerMatch = trailerRegex.exec(fileStr)) !== null) {
1176
+ const candidateRoot = trailerMatch[1]?.match(/\/Root\s+(\d+\s+\d+\s+R)/);
1177
+ if (candidateRoot)
1178
+ rootMatch = candidateRoot;
1179
+ const candidateEncrypt = trailerMatch[1]?.match(/\/Encrypt\s+(\d+\s+\d+\s+R)/);
1180
+ if (candidateEncrypt)
1181
+ encryptMatch = candidateEncrypt;
1182
+ const candidateId = trailerMatch[1]?.match(/\/ID\s*(\[[^\]]*\])/);
1183
+ if (candidateId)
1184
+ idMatch = candidateId;
1185
+ }
1186
+ if (!rootMatch) {
1187
+ rootMatch = fileStr.match(/\/Root\s+(\d+\s+\d+\s+R)/);
1188
+ }
1189
+ if (!encryptMatch) {
1190
+ encryptMatch = fileStr.match(/\/Encrypt\s+(\d+\s+\d+\s+R)/);
1191
+ }
1192
+ if (!idMatch) {
1193
+ idMatch = fileStr.match(/\/ID\s*(\[[^\]]*\])/);
1194
+ }
1195
+ let extraTrailerEntries = "";
1196
+ if (rootMatch)
1197
+ extraTrailerEntries += ` /Root ${rootMatch[1]}`;
1198
+ if (encryptMatch)
1199
+ extraTrailerEntries += ` /Encrypt ${encryptMatch[1]}`;
1200
+ if (idMatch)
1201
+ extraTrailerEntries += ` /ID ${idMatch[1]}`;
1202
+ const trailerStr = `trailer\n<< /Size ${maxId + 1}${extraTrailerEntries} >>\nstartxref\n${xrefOffset}\n%%EOF\n`;
1203
+ return Buffer.concat([buffer, Buffer.from(xrefStr + trailerStr, "latin1")]);
1204
+ }
1205
+ }
1206
+ // 2. Opportunistic qpdf if installed
1207
+ const externalRepaired = await runExternalBinaryTool("qpdf", ["--qdf", "-", "-"], buffer, timeoutMs);
1208
+ if (externalRepaired) {
1209
+ return externalRepaired;
1210
+ }
1211
+ return buffer;
1212
+ }
1213
+ //# sourceMappingURL=pdf.js.map