officeparser 5.2.2 → 6.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,850 @@
1
+ "use strict";
2
+ /**
3
+ * PDF Parser
4
+ *
5
+ * Extracts text, metadata, images, links, and attachments from PDF files using PDF.js (pdfjs-dist).
6
+ *
7
+ * **Features:**
8
+ * - Text extraction with formatting (bold, italic, font, size)
9
+ * - Comprehensive metadata extraction (title, author, subject, creator, producer, creation/modification dates)
10
+ * - Hyperlink extraction from PDF annotations
11
+ * - Heading detection via font size heuristics
12
+ * - Image extraction as attachments with optional OCR (using Tesseract.js)
13
+ * - Embedded file attachment extraction
14
+ * - Layout preservation (respects order of text and images)
15
+ *
16
+ * **PDF Format Limitations (compared to DOCX/ODT):**
17
+ *
18
+ * PDF was designed as a "page description language" for visual fidelity, not semantic structure.
19
+ * The following features **cannot be reliably extracted** from PDFs:
20
+ *
21
+ * - **Tables**: PDF has no table structure. Tables are just text positioned to look tabular.
22
+ * Extracting tables would require complex spatial analysis with many false positives.
23
+ * See: https://stackoverflow.com/questions/36978446/why-is-it-difficult-to-extract-data-from-pdfs
24
+ *
25
+ * - **Lists**: PDF has no list structure. Bullets/numbers are just text characters.
26
+ * No hierarchy or list type information is stored. Would require heuristic detection
27
+ * that would have many edge cases and errors.
28
+ *
29
+ * - **Styles**: PDF has no style definitions like "Heading1" or "Normal". Only visual
30
+ * properties (font, size) exist. We use font size heuristics to detect headings.
31
+ *
32
+ * - **Notes (Footnotes/Endnotes)**: PDF has no concept of footnotes/endnotes as structured
33
+ * elements. They're just smaller text at the bottom of pages.
34
+ *
35
+ * - **Text Color**: While PDF stores color, pdfjs-dist doesn't expose text color in the
36
+ * textContent API. Would require parsing the operator stream which is complex.
37
+ *
38
+ * - **Background Color**: Same limitation as text color.
39
+ *
40
+ * - **Underline/Strikethrough**: These are drawn as separate line elements in PDF,
41
+ * not properties of text. Association would require spatial analysis.
42
+ *
43
+ * **Parsing Approach:**
44
+ * 1. Load PDF document using pdfjs-dist.
45
+ * 2. Extract global metadata from document info dictionary.
46
+ * 3. Extract embedded file attachments.
47
+ * 4. Iterate through each page:
48
+ * a. Collect text items with position and formatting.
49
+ * b. Collect link annotations with associated text.
50
+ * c. Collect images from the operator list.
51
+ * d. Sort all items by vertical position (top-to-bottom reading order).
52
+ * e. Group text into paragraphs/headings based on line breaks and font sizes.
53
+ * f. Process images as attachments with optional OCR.
54
+ * 5. Apply heading detection based on font size heuristics.
55
+ *
56
+ * @module PdfParser
57
+ * @see https://mozilla.github.io/pdf.js/ PDF.js documentation
58
+ * @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
59
+ */
60
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
61
+ if (k2 === undefined) k2 = k;
62
+ var desc = Object.getOwnPropertyDescriptor(m, k);
63
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
64
+ desc = { enumerable: true, get: function() { return m[k]; } };
65
+ }
66
+ Object.defineProperty(o, k2, desc);
67
+ }) : (function(o, m, k, k2) {
68
+ if (k2 === undefined) k2 = k;
69
+ o[k2] = m[k];
70
+ }));
71
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
72
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
73
+ }) : function(o, v) {
74
+ o["default"] = v;
75
+ });
76
+ var __importStar = (this && this.__importStar) || function (mod) {
77
+ if (mod && mod.__esModule) return mod;
78
+ var result = {};
79
+ if (mod != null) for (var k in mod) if (k !== "default" && Object.prototype.hasOwnProperty.call(mod, k)) __createBinding(result, mod, k);
80
+ __setModuleDefault(result, mod);
81
+ return result;
82
+ };
83
+ Object.defineProperty(exports, "__esModule", { value: true });
84
+ exports.parsePdf = void 0;
85
+ const errorUtils_1 = require("../utils/errorUtils");
86
+ const imageUtils_1 = require("../utils/imageUtils");
87
+ const ocrUtils_1 = require("../utils/ocrUtils");
88
+ /**
89
+ * Parses PDF creation/modification date strings.
90
+ * PDF dates are in format: D:YYYYMMDDHHmmSSOHH'mm'
91
+ * Where O is timezone offset direction (+/-), HH is hours, mm is minutes.
92
+ *
93
+ * @param dateString - The PDF date string
94
+ * @returns Parsed Date object or undefined if parsing fails
95
+ */
96
+ function parsePdfDate(dateString) {
97
+ if (!dateString)
98
+ return undefined;
99
+ try {
100
+ // Remove "D:" prefix if present
101
+ let str = dateString.startsWith('D:') ? dateString.slice(2) : dateString;
102
+ // Extract components: YYYYMMDDHHmmSS
103
+ const year = parseInt(str.slice(0, 4), 10);
104
+ const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
105
+ const day = parseInt(str.slice(6, 8), 10) || 1;
106
+ const hour = parseInt(str.slice(8, 10), 10) || 0;
107
+ const minute = parseInt(str.slice(10, 12), 10) || 0;
108
+ const second = parseInt(str.slice(12, 14), 10) || 0;
109
+ // Handle timezone if present
110
+ const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
111
+ if (tzMatch) {
112
+ const tzSign = tzMatch[1] === '-' ? -1 : 1;
113
+ const tzHours = parseInt(tzMatch[2], 10) || 0;
114
+ const tzMinutes = parseInt(tzMatch[3], 10) || 0;
115
+ const offset = tzSign * (tzHours * 60 + tzMinutes);
116
+ // Create date in UTC and adjust for timezone
117
+ const utc = Date.UTC(year, month, day, hour, minute, second);
118
+ return new Date(utc - offset * 60000);
119
+ }
120
+ return new Date(year, month, day, hour, minute, second);
121
+ }
122
+ catch {
123
+ // Fallback: try native Date parsing
124
+ try {
125
+ return new Date(dateString);
126
+ }
127
+ catch {
128
+ return undefined;
129
+ }
130
+ }
131
+ }
132
+ /**
133
+ * Calculates statistics about font sizes in the document.
134
+ * Used for heading detection heuristics.
135
+ */
136
+ function calculateFontStats(pageItems) {
137
+ const sizes = [];
138
+ for (const page of pageItems) {
139
+ for (const item of page) {
140
+ if (item.type === 'text' && item.height > 0) {
141
+ sizes.push(item.height);
142
+ }
143
+ }
144
+ }
145
+ if (sizes.length === 0)
146
+ return { median: 12, max: 12 };
147
+ sizes.sort((a, b) => a - b);
148
+ const median = sizes[Math.floor(sizes.length / 2)];
149
+ const max = sizes[sizes.length - 1];
150
+ return { median, max };
151
+ }
152
+ /**
153
+ * Determines if a text item should be considered a heading based on its font size.
154
+ *
155
+ * Heuristic: Text that is at least 20% larger than the median body text size
156
+ * is considered a heading. The level (1-6) is determined by relative size.
157
+ *
158
+ * @param fontSize - The font size of the text
159
+ * @param fontStats - Statistics about fonts in the document
160
+ * @returns Heading level (1-6) or 0 if not a heading
161
+ */
162
+ function detectHeadingLevel(fontSize, fontStats) {
163
+ // If font is less than 20% larger than median, it's not a heading
164
+ if (fontSize <= fontStats.median * 1.2)
165
+ return 0;
166
+ // Calculate heading level based on how much larger than median
167
+ const ratio = fontSize / fontStats.median;
168
+ if (ratio >= 2.0)
169
+ return 1; // 2x or more = H1
170
+ if (ratio >= 1.7)
171
+ return 2; // 1.7x-2x = H2
172
+ if (ratio >= 1.5)
173
+ return 3; // 1.5x-1.7x = H3
174
+ if (ratio >= 1.35)
175
+ return 4; // 1.35x-1.5x = H4
176
+ if (ratio >= 1.2)
177
+ return 5; // 1.2x-1.35x = H5
178
+ return 0; // Below threshold
179
+ }
180
+ /**
181
+ * Checks if a text position falls within a link annotation rectangle.
182
+ */
183
+ /**
184
+ * Checks if a text item falls within a link annotation rectangle.
185
+ * Uses the center point of the text item to avoid false positives for text
186
+ * that starts immediately after a link (which would share the same x coordinate boundary).
187
+ */
188
+ function findLinkForText(item, annotations) {
189
+ const centerX = item.x + (item.width / 2);
190
+ const centerY = item.y + (item.height / 2);
191
+ for (const annot of annotations) {
192
+ // PDF annotation rects are [x1, y1, x2, y2] in page coordinates
193
+ const [x1, y1, x2, y2] = annot.rect;
194
+ const minX = Math.min(x1, x2);
195
+ const maxX = Math.max(x1, x2);
196
+ const minY = Math.min(y1, y2);
197
+ const maxY = Math.max(y1, y2);
198
+ // Check if center point is within the rect (strict, no tolerance)
199
+ if (centerX >= minX && centerX <= maxX &&
200
+ centerY >= minY && centerY <= maxY) {
201
+ return annot;
202
+ }
203
+ }
204
+ return undefined;
205
+ }
206
+ /**
207
+ * Encodes raw RGBA data into a 24-bit BMP buffer with a white background.
208
+ * Transparency (alpha channel) is flattened against white.
209
+ *
210
+ * @param width - Image width
211
+ * @param height - Image height
212
+ * @param data - RGBA pixel data
213
+ * @returns BMP Buffer
214
+ */
215
+ function encodeBmp(width, height, data) {
216
+ // BMP row size must be a multiple of 4 bytes
217
+ const rowSize = Math.floor((24 * width + 31) / 32) * 4;
218
+ const padding = rowSize - (width * 3);
219
+ const headerSize = 54; // 14 (File Header) + 40 (DIB Header)
220
+ const imageSize = rowSize * height;
221
+ const fileSize = headerSize + imageSize;
222
+ const buffer = Buffer.alloc(fileSize);
223
+ // --- File Header (14 bytes) ---
224
+ buffer.write('BM', 0); // Signature
225
+ buffer.writeUInt32LE(fileSize, 2); // File Size
226
+ buffer.writeUInt32LE(0, 6); // Reserved
227
+ buffer.writeUInt32LE(headerSize, 10); // Offset to pixel data
228
+ // --- DIB Header (BITMAPINFOHEADER - 40 bytes) ---
229
+ buffer.writeUInt32LE(40, 14); // Header Size
230
+ buffer.writeInt32LE(width, 18); // Width
231
+ buffer.writeInt32LE(-height, 22); // Height (negative for top-down)
232
+ buffer.writeUInt16LE(1, 26); // Planes
233
+ buffer.writeUInt16LE(24, 28); // Bit Count (24-bit RGB)
234
+ buffer.writeUInt32LE(0, 30); // Compression (BI_RGB)
235
+ buffer.writeUInt32LE(imageSize, 34); // Image Size
236
+ buffer.writeInt32LE(2835, 38); // X PixelsPerMeter (72 DPI)
237
+ buffer.writeInt32LE(2835, 42); // Y PixelsPerMeter (72 DPI)
238
+ buffer.writeUInt32LE(0, 46); // Colors Used
239
+ buffer.writeUInt32LE(0, 50); // Colors Important
240
+ // --- Pixel Data ---
241
+ let offset = headerSize;
242
+ for (let y = 0; y < height; y++) {
243
+ for (let x = 0; x < width; x++) {
244
+ const i = (y * width + x) * 4;
245
+ // RGBA input
246
+ const r = data[i + 0];
247
+ const g = data[i + 1];
248
+ const b = data[i + 2];
249
+ const a = data[i + 3];
250
+ // Flatten alpha against white background
251
+ // out = alpha * pixel + (1 - alpha) * white
252
+ // white = 255
253
+ const alpha = a / 255;
254
+ const outR = Math.round(r * alpha + 255 * (1 - alpha));
255
+ const outG = Math.round(g * alpha + 255 * (1 - alpha));
256
+ const outB = Math.round(b * alpha + 255 * (1 - alpha));
257
+ // Write as BGR (BMP standard)
258
+ buffer[offset + 0] = outB;
259
+ buffer[offset + 1] = outG;
260
+ buffer[offset + 2] = outR;
261
+ offset += 3;
262
+ }
263
+ // Write padding
264
+ for (let p = 0; p < padding; p++) {
265
+ buffer[offset] = 0;
266
+ offset++;
267
+ }
268
+ }
269
+ return buffer;
270
+ }
271
+ /**
272
+ * Converts raw PDF image data to a buffer for attachment extraction.
273
+ *
274
+ * **Important Limitation:**
275
+ * PDF images are stored as raw pixel data (RGB, RGBA, or grayscale), not as encoded
276
+ * image files like PNG or JPEG. This function converts the raw data to a normalized
277
+ * RGBA buffer, but this is NOT a valid image file format.
278
+ *
279
+ * For display, the raw RGBA data would need to be encoded to PNG/JPEG, which requires
280
+ * an additional library like `sharp` or `pngjs`. Currently, this is stored as raw bytes.
281
+ *
282
+ * OCR is NOT supported for PDF images because Tesseract.js requires encoded image files
283
+ * (PNG, JPEG, etc.), not raw pixel data. To enable OCR, a PNG encoder would need to be added.
284
+ *
285
+ * @param data - Raw pixel data from PDF.js
286
+ * @param width - Image width in pixels
287
+ * @param height - Image height in pixels
288
+ * @param kind - PDF.js image kind (1=Grayscale, 2=RGB, 3=RGBA)
289
+ * @returns Buffer containing RGBA pixel data (NOT an encoded image file)
290
+ */
291
+ function convertToRgbaBuffer(data, width, height, kind) {
292
+ // PDF.js image kind values:
293
+ // 1 = GRAYSCALE
294
+ // 2 = RGB
295
+ // 3 = RGBA
296
+ let rgbaData;
297
+ if (kind === 1) {
298
+ // Grayscale - expand to RGBA
299
+ rgbaData = new Uint8ClampedArray(width * height * 4);
300
+ for (let i = 0; i < width * height; i++) {
301
+ const gray = data[i];
302
+ rgbaData[i * 4] = gray;
303
+ rgbaData[i * 4 + 1] = gray;
304
+ rgbaData[i * 4 + 2] = gray;
305
+ rgbaData[i * 4 + 3] = 255;
306
+ }
307
+ }
308
+ else if (kind === 2 || data.length === width * height * 3) {
309
+ // RGB - add alpha channel
310
+ rgbaData = new Uint8ClampedArray(width * height * 4);
311
+ for (let i = 0; i < width * height; i++) {
312
+ rgbaData[i * 4] = data[i * 3];
313
+ rgbaData[i * 4 + 1] = data[i * 3 + 1];
314
+ rgbaData[i * 4 + 2] = data[i * 3 + 2];
315
+ rgbaData[i * 4 + 3] = 255;
316
+ }
317
+ }
318
+ else {
319
+ // Assume RGBA
320
+ rgbaData = data instanceof Uint8ClampedArray ? data : new Uint8ClampedArray(data);
321
+ }
322
+ return Buffer.from(rgbaData.buffer, rgbaData.byteOffset, rgbaData.byteLength);
323
+ }
324
+ /**
325
+ * Parses a PDF file and extracts content.
326
+ *
327
+ * @param buffer - The PDF file buffer
328
+ * @param config - Parser configuration
329
+ * @returns Promise resolving to the parsed AST
330
+ */
331
+ const parsePdf = async (buffer, config) => {
332
+ let pdfjs;
333
+ // Check if we are in a Node.js environment
334
+ // @ts-ignore
335
+ if (typeof window === 'undefined') {
336
+ // Helper to bypass TS/Webpack/Other transpilers converting import() to require() when compiling to CJS
337
+ // Defined here to avoid CSP 'unsafe-eval' issues in browser environments
338
+ const dynamicImport = new Function('specifier', 'return import(specifier)');
339
+ try {
340
+ // Use legacy build for Node.js (required for pdfjs-dist v5+)
341
+ pdfjs = await dynamicImport('pdfjs-dist/legacy/build/pdf.mjs');
342
+ }
343
+ catch (e) {
344
+ // Fallback to standard if legacy not found (e.g. older versions)
345
+ pdfjs = await dynamicImport('pdfjs-dist');
346
+ }
347
+ }
348
+ else {
349
+ pdfjs = await Promise.resolve().then(() => __importStar(require('pdfjs-dist')));
350
+ }
351
+ // Configure worker
352
+ if (config.pdfWorkerSrc) {
353
+ pdfjs.GlobalWorkerOptions.workerSrc = config.pdfWorkerSrc;
354
+ }
355
+ else {
356
+ // Fallbacks when no workerSrc is provided
357
+ // @ts-ignore
358
+ if (typeof window !== 'undefined') {
359
+ // Browser: Default to CDN
360
+ pdfjs.GlobalWorkerOptions.workerSrc = 'https://unpkg.com/pdfjs-dist@5.4.530/build/pdf.worker.min.mjs';
361
+ }
362
+ else {
363
+ // Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
364
+ try {
365
+ // We use require.resolve to find the exact path of the installed package.
366
+ // @ts-ignore - 'require' is available in Node.js/CommonJS environment
367
+ const workerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
368
+ pdfjs.GlobalWorkerOptions.workerSrc = workerPath;
369
+ }
370
+ catch (e) {
371
+ if (config.outputErrorToConsole)
372
+ console.warn("[PdfParser] Could not auto-resolve local worker path:", e);
373
+ }
374
+ }
375
+ }
376
+ const uint8Array = new Uint8Array(buffer);
377
+ const loadingTask = pdfjs.getDocument({
378
+ data: uint8Array,
379
+ verbosity: 0 // ERRORS only, suppresses warnings
380
+ });
381
+ // Handle loading errors, specifically missing worker in browser
382
+ let pdfDocument;
383
+ try {
384
+ pdfDocument = await loadingTask.promise;
385
+ }
386
+ catch (e) {
387
+ if (e.message?.includes('workerSrc') || e.message?.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
388
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.PDF_WORKER_MISSING, config);
389
+ }
390
+ throw e;
391
+ }
392
+ const content = [];
393
+ const attachments = [];
394
+ const numPages = pdfDocument.numPages;
395
+ // Collect all page items for font statistics before processing
396
+ const allPageItems = [];
397
+ // --- Metadata Extraction ---
398
+ // Extract all available metadata from the PDF info dictionary.
399
+ // Note: Some metadata fields depend on how the PDF was created.
400
+ // - Producer: Software that created the PDF
401
+ // - Creator: Application that made the original document
402
+ // - Keywords, Description are rarely present
403
+ const meta = await pdfDocument.getMetadata().catch(() => ({ info: {} }));
404
+ const info = meta.info;
405
+ const metadata = {
406
+ pages: numPages,
407
+ title: info?.Title || undefined,
408
+ author: info?.Author || undefined,
409
+ subject: info?.Subject || undefined,
410
+ description: info?.Keywords || undefined,
411
+ created: parsePdfDate(info?.CreationDate),
412
+ modified: parsePdfDate(info?.ModDate),
413
+ // Note: lastModifiedBy is not available in PDF format - there's no concept of "last modifier"
414
+ // The Author field only tracks original author.
415
+ };
416
+ // --- Embedded File Attachment Extraction ---
417
+ /**
418
+ * PDF can contain embedded files (not images in content, but attached files).
419
+ * These are separate from images in the page content stream.
420
+ */
421
+ try {
422
+ const embeddedFiles = await pdfDocument.getAttachments();
423
+ if (embeddedFiles && config.extractAttachments) {
424
+ for (const name in embeddedFiles) {
425
+ const file = embeddedFiles[name];
426
+ const fileBuffer = Buffer.from(file.content);
427
+ const attachment = (0, imageUtils_1.createAttachment)(file.filename, fileBuffer);
428
+ attachments.push(attachment);
429
+ }
430
+ }
431
+ }
432
+ catch (e) {
433
+ if (config.outputErrorToConsole)
434
+ console.error("Error extracting embedded attachments:", e);
435
+ }
436
+ // --- First Pass: Collect all items for font statistics ---
437
+ for (let i = 1; i <= numPages; i++) {
438
+ let page;
439
+ let textContent;
440
+ const pageItems = [];
441
+ try {
442
+ page = await pdfDocument.getPage(i);
443
+ // Extract text content
444
+ textContent = await page.getTextContent();
445
+ }
446
+ catch (e) {
447
+ if (config.outputErrorToConsole)
448
+ console.warn(`[PdfParser] Error loading page ${i}:`, e);
449
+ // Push empty items to maintain index alignment for second pass
450
+ allPageItems.push(pageItems);
451
+ continue;
452
+ }
453
+ const commonObjs = page.commonObjs;
454
+ for (const item of textContent.items) {
455
+ const transform = item.transform;
456
+ const x = transform[4];
457
+ const y = transform[5];
458
+ const width = item.width || 0;
459
+ const height = item.height || Math.abs(transform[3]) || 12;
460
+ // Extract formatting from font
461
+ const formatting = {};
462
+ let fontName;
463
+ if (item.fontName && commonObjs) {
464
+ try {
465
+ if (commonObjs.has(item.fontName)) {
466
+ // Use callback-based get to ensure safe resolution
467
+ const fontData = await new Promise((resolve) => {
468
+ // @ts-ignore
469
+ commonObjs.get(item.fontName, (data) => resolve(data));
470
+ });
471
+ if (fontData?.name) {
472
+ // Remove PDF subset prefix (6 uppercase letters + '+')
473
+ fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
474
+ formatting.font = fontName;
475
+ // Detect bold/italic from font name
476
+ const lowerName = fontData.name.toLowerCase();
477
+ if (lowerName.includes('bold'))
478
+ formatting.bold = true;
479
+ if (lowerName.includes('italic') || lowerName.includes('oblique'))
480
+ formatting.italic = true;
481
+ }
482
+ }
483
+ }
484
+ catch {
485
+ // Font lookup failed, continue without font info
486
+ }
487
+ }
488
+ if (height > 0) {
489
+ formatting.size = Math.round(height).toString();
490
+ }
491
+ pageItems.push({
492
+ type: 'text',
493
+ x,
494
+ y,
495
+ width,
496
+ height,
497
+ text: item.str,
498
+ fontName,
499
+ formatting
500
+ });
501
+ }
502
+ // Extract images if enabled
503
+ if (config.extractAttachments || config.ocr) {
504
+ try {
505
+ const ops = await page.getOperatorList();
506
+ const fnArray = ops.fnArray;
507
+ const argsArray = ops.argsArray;
508
+ for (let j = 0; j < fnArray.length; j++) {
509
+ const fn = fnArray[j];
510
+ if (fn === pdfjs.OPS.dependency) {
511
+ const deps = argsArray[j];
512
+ for (const dep of deps) {
513
+ // In pdfjs-dist v3+, get() throws if not resolved unless a callback is provided.
514
+ // We must use the callback pattern to wait for resolution.
515
+ try {
516
+ if (page.objs.has(dep))
517
+ continue;
518
+ await new Promise((resolve) => {
519
+ const timeout = setTimeout(() => {
520
+ resolve();
521
+ }, 500);
522
+ page.objs.get(dep, (data) => {
523
+ clearTimeout(timeout);
524
+ resolve();
525
+ });
526
+ });
527
+ }
528
+ catch (e) {
529
+ if (config.outputErrorToConsole) {
530
+ console.error(`[PdfParser] Failed to load dependency ${dep}:`, e);
531
+ }
532
+ }
533
+ }
534
+ }
535
+ if (fn === pdfjs.OPS.paintImageXObject || fn === pdfjs.OPS.paintXObject) {
536
+ const imgName = argsArray[j][0];
537
+ try {
538
+ let hasObj = page.objs.has(imgName);
539
+ let targetObjs = page.objs;
540
+ if (!hasObj && page.commonObjs.has(imgName)) {
541
+ hasObj = true;
542
+ targetObjs = page.commonObjs;
543
+ }
544
+ if (hasObj) {
545
+ // Use callback-based get to ensure safe resolution
546
+ const rawObj = await new Promise((resolve) => {
547
+ // @ts-ignore
548
+ targetObjs.get(imgName, (data) => resolve(data));
549
+ });
550
+ const imgObj = rawObj;
551
+ // Browser-specific: Handle ImageBitmap if data is missing
552
+ if (typeof window !== 'undefined' && !imgObj.data && imgObj.bitmap) {
553
+ try {
554
+ const canvas = document.createElement('canvas');
555
+ canvas.width = imgObj.width;
556
+ canvas.height = imgObj.height;
557
+ const ctx = canvas.getContext('2d');
558
+ if (ctx) {
559
+ ctx.drawImage(imgObj.bitmap, 0, 0);
560
+ imgObj.data = ctx.getImageData(0, 0, imgObj.width, imgObj.height).data;
561
+ imgObj.kind = 3; // RGBA
562
+ }
563
+ }
564
+ catch (e) {
565
+ if (config.outputErrorToConsole)
566
+ console.error(`[PdfParser] Failed to extract from ImageBitmap:`, e);
567
+ }
568
+ }
569
+ if (imgObj?.data && imgObj.width > 0 && imgObj.height > 0) {
570
+ // Find position from transform matrix
571
+ let imgX = 0, imgY = 0;
572
+ for (let k = j - 1; k >= 0; k--) {
573
+ if (fnArray[k] === pdfjs.OPS.transform) {
574
+ imgX = argsArray[k][4];
575
+ imgY = argsArray[k][5];
576
+ break;
577
+ }
578
+ }
579
+ pageItems.push({
580
+ type: 'image',
581
+ x: imgX,
582
+ y: imgY,
583
+ name: imgName,
584
+ data: imgObj.data,
585
+ width: imgObj.width,
586
+ height: imgObj.height,
587
+ kind: imgObj.kind
588
+ });
589
+ }
590
+ }
591
+ }
592
+ catch {
593
+ // Image access failed, continue
594
+ }
595
+ }
596
+ }
597
+ }
598
+ catch (e) {
599
+ if (config.outputErrorToConsole)
600
+ console.error(`Error extracting images from page ${i}:`, e);
601
+ }
602
+ }
603
+ allPageItems.push(pageItems);
604
+ }
605
+ // Calculate font statistics for heading detection
606
+ const fontStats = calculateFontStats(allPageItems);
607
+ // --- Second Pass: Process pages with font statistics ---
608
+ for (let i = 0; i < allPageItems.length; i++) {
609
+ const pageNum = i + 1;
610
+ let page;
611
+ try {
612
+ page = await pdfDocument.getPage(pageNum);
613
+ }
614
+ catch (e) {
615
+ if (config.outputErrorToConsole)
616
+ console.warn(`[PdfParser] Error loading page ${pageNum} in second pass:`, e);
617
+ continue;
618
+ }
619
+ const pageItems = allPageItems[i];
620
+ const pageContent = [];
621
+ // Extract link annotations for this page
622
+ const annotations = [];
623
+ try {
624
+ const annots = await page.getAnnotations();
625
+ for (const annot of annots) {
626
+ if (annot.subtype === 'Link' && annot.rect) {
627
+ annotations.push({
628
+ rect: annot.rect,
629
+ url: annot.url,
630
+ dest: annot.dest
631
+ });
632
+ }
633
+ }
634
+ }
635
+ catch (e) {
636
+ if (config.outputErrorToConsole)
637
+ console.error(`Error extracting annotations from page ${pageNum}:`, e);
638
+ }
639
+ // Sort items: Y descending (top to bottom), then X ascending (left to right)
640
+ pageItems.sort((a, b) => {
641
+ if (Math.abs(b.y - a.y) > 5)
642
+ return b.y - a.y;
643
+ return a.x - b.x;
644
+ });
645
+ // Process sorted items into content nodes
646
+ let currentNode = null;
647
+ let currentNodeFontSize = 0;
648
+ let lastY = -1;
649
+ let imageCounter = 0;
650
+ for (const item of pageItems) {
651
+ if (item.type === 'text') {
652
+ const text = item.text;
653
+ if (!text)
654
+ continue;
655
+ // Check for new line
656
+ const isNewLine = lastY !== -1 && Math.abs(item.y - lastY) > 5;
657
+ if (isNewLine && currentNode) {
658
+ // Finalize and push current node
659
+ if ((currentNode.text || '').trim().length > 0) {
660
+ pageContent.push(currentNode);
661
+ }
662
+ currentNode = null;
663
+ }
664
+ // Skip pure whitespace at start of lines
665
+ if (!currentNode && text.trim().length === 0) {
666
+ lastY = item.y;
667
+ continue;
668
+ }
669
+ // Determine if this should be a heading
670
+ const headingLevel = detectHeadingLevel(item.height, fontStats);
671
+ if (!currentNode) {
672
+ // Start new node
673
+ if (headingLevel > 0) {
674
+ currentNode = {
675
+ type: 'heading',
676
+ text: '',
677
+ children: [],
678
+ metadata: { level: headingLevel }
679
+ };
680
+ }
681
+ else {
682
+ currentNode = {
683
+ type: 'paragraph',
684
+ text: '',
685
+ children: []
686
+ };
687
+ }
688
+ currentNodeFontSize = item.height;
689
+ }
690
+ // Handle whitespace
691
+ if (text.trim().length === 0) {
692
+ if (currentNode.children && currentNode.children.length > 0) {
693
+ const lastChild = currentNode.children[currentNode.children.length - 1];
694
+ if (lastChild.type === 'text' && lastChild.text) {
695
+ lastChild.text += text;
696
+ currentNode.text += text;
697
+ }
698
+ }
699
+ lastY = item.y;
700
+ continue;
701
+ }
702
+ // Add space between words if needed
703
+ if (currentNode.text && currentNode.text.length > 0 && !currentNode.text.endsWith(' ')) {
704
+ currentNode.text += ' ';
705
+ if (currentNode.children && currentNode.children.length > 0) {
706
+ const lastChild = currentNode.children[currentNode.children.length - 1];
707
+ if (lastChild.type === 'text' && lastChild.text) {
708
+ lastChild.text += ' ';
709
+ }
710
+ }
711
+ }
712
+ currentNode.text += text;
713
+ // Check for link
714
+ const link = findLinkForText(item, annotations);
715
+ let textMetadata;
716
+ if (link) {
717
+ if (link.url) {
718
+ textMetadata = {
719
+ link: link.url,
720
+ linkType: link.url.startsWith('#') ? 'internal' : 'external'
721
+ };
722
+ }
723
+ else if (link.dest) {
724
+ // Internal destination
725
+ textMetadata = {
726
+ link: typeof link.dest === 'string' ? `#${link.dest}` : '#internal',
727
+ linkType: 'internal'
728
+ };
729
+ }
730
+ }
731
+ // Try to merge with last child if same formatting and no link change
732
+ let merged = false;
733
+ if (currentNode.children && currentNode.children.length > 0 && !textMetadata) {
734
+ const lastChild = currentNode.children[currentNode.children.length - 1];
735
+ if (lastChild.type === 'text' &&
736
+ isSameFormatting(lastChild.formatting, item.formatting) &&
737
+ !lastChild.metadata) {
738
+ lastChild.text = (lastChild.text || '') + text;
739
+ merged = true;
740
+ }
741
+ }
742
+ if (!merged) {
743
+ const textNode = {
744
+ type: 'text',
745
+ text: text,
746
+ formatting: Object.keys(item.formatting).length > 0 ? item.formatting : undefined
747
+ };
748
+ if (textMetadata) {
749
+ textNode.metadata = textMetadata;
750
+ }
751
+ currentNode.children?.push(textNode);
752
+ }
753
+ lastY = item.y;
754
+ }
755
+ else if (item.type === 'image') {
756
+ // Flush current node
757
+ if (currentNode) {
758
+ if ((currentNode.text || '').trim().length > 0) {
759
+ pageContent.push(currentNode);
760
+ }
761
+ currentNode = null;
762
+ }
763
+ imageCounter++;
764
+ // Note: Using .bmp extension since we encode to BMP for broad compatibility
765
+ const attachmentName = `pdf_image_p${pageNum}_${imageCounter}.bmp`;
766
+ /**
767
+ * Image extraction for PDF files.
768
+ *
769
+ * PDF stores images as raw pixel data. We convert to BMP for compatibility.
770
+ */
771
+ if (config.extractAttachments) {
772
+ try {
773
+ const imageBuffer = convertToRgbaBuffer(item.data, item.width, item.height, item.kind);
774
+ // Encode as BMP
775
+ const bmpBuffer = encodeBmp(item.width, item.height, new Uint8Array(imageBuffer));
776
+ const attachment = (0, imageUtils_1.createAttachment)(attachmentName, bmpBuffer);
777
+ attachment.mimeType = 'image/bmp';
778
+ // Perform OCR if enabled
779
+ if (config.ocr) {
780
+ try {
781
+ // Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
782
+ if (item.width >= 10 && item.height >= 10) {
783
+ attachment.ocrText = (await (0, ocrUtils_1.performOcr)(bmpBuffer, config.ocrLanguage)).trim();
784
+ }
785
+ }
786
+ catch (e) {
787
+ if (config.outputErrorToConsole)
788
+ console.error(`OCR failed for ${attachmentName}:`, e);
789
+ }
790
+ }
791
+ attachments.push(attachment);
792
+ // Create image content node
793
+ const imageMetadata = {
794
+ attachmentName,
795
+ };
796
+ pageContent.push({
797
+ type: 'image',
798
+ text: attachment.ocrText || '',
799
+ metadata: { ...imageMetadata }
800
+ });
801
+ }
802
+ catch (e) {
803
+ (0, errorUtils_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
804
+ }
805
+ }
806
+ }
807
+ }
808
+ // Flush last node
809
+ if (currentNode && (currentNode.text || '').trim().length > 0) {
810
+ pageContent.push(currentNode);
811
+ }
812
+ // Add page node to content
813
+ content.push({
814
+ type: 'page',
815
+ children: pageContent,
816
+ text: pageContent.map(node => node.text).join(config.newlineDelimiter ?? '\n\n'),
817
+ metadata: { pageNumber: pageNum }
818
+ });
819
+ }
820
+ return {
821
+ type: 'pdf',
822
+ metadata: metadata,
823
+ content: content,
824
+ attachments: attachments,
825
+ toText: () => content.map(c => c.text).join(config.newlineDelimiter ?? '\n\n')
826
+ };
827
+ };
828
+ exports.parsePdf = parsePdf;
829
+ /**
830
+ * Helper to compare two text formatting objects.
831
+ * Returns true if both have the same properties with the same values.
832
+ */
833
+ function isSameFormatting(a, b) {
834
+ if (!a && !b)
835
+ return true;
836
+ if (!a || !b)
837
+ return false;
838
+ const keysA = Object.keys(a).sort();
839
+ const keysB = Object.keys(b).sort();
840
+ if (keysA.length !== keysB.length)
841
+ return false;
842
+ for (let i = 0; i < keysA.length; i++) {
843
+ const key = keysA[i];
844
+ if (keysA[i] !== keysB[i])
845
+ return false;
846
+ if (a[key] !== b[key])
847
+ return false;
848
+ }
849
+ return true;
850
+ }