officeparser 5.2.1 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,847 @@
1
+ "use strict";
2
+ /**
3
+ * PDF Parser
4
+ *
5
+ * Extracts text, metadata, images, links, and attachments from PDF files using PDF.js (pdfjs-dist).
6
+ *
7
+ * **Features:**
8
+ * - Text extraction with formatting (bold, italic, font, size)
9
+ * - Comprehensive metadata extraction (title, author, subject, creator, producer, creation/modification dates)
10
+ * - Hyperlink extraction from PDF annotations
11
+ * - Heading detection via font size heuristics
12
+ * - Image extraction as attachments with optional OCR (using Tesseract.js)
13
+ * - Embedded file attachment extraction
14
+ * - Layout preservation (respects order of text and images)
15
+ *
16
+ * **PDF Format Limitations (compared to DOCX/ODT):**
17
+ *
18
+ * PDF was designed as a "page description language" for visual fidelity, not semantic structure.
19
+ * The following features **cannot be reliably extracted** from PDFs:
20
+ *
21
+ * - **Tables**: PDF has no table structure. Tables are just text positioned to look tabular.
22
+ * Extracting tables would require complex spatial analysis with many false positives.
23
+ * See: https://stackoverflow.com/questions/36978446/why-is-it-difficult-to-extract-data-from-pdfs
24
+ *
25
+ * - **Lists**: PDF has no list structure. Bullets/numbers are just text characters.
26
+ * No hierarchy or list type information is stored. Would require heuristic detection
27
+ * that would have many edge cases and errors.
28
+ *
29
+ * - **Styles**: PDF has no style definitions like "Heading1" or "Normal". Only visual
30
+ * properties (font, size) exist. We use font size heuristics to detect headings.
31
+ *
32
+ * - **Notes (Footnotes/Endnotes)**: PDF has no concept of footnotes/endnotes as structured
33
+ * elements. They're just smaller text at the bottom of pages.
34
+ *
35
+ * - **Text Color**: While PDF stores color, pdfjs-dist doesn't expose text color in the
36
+ * textContent API. Would require parsing the operator stream which is complex.
37
+ *
38
+ * - **Background Color**: Same limitation as text color.
39
+ *
40
+ * - **Underline/Strikethrough**: These are drawn as separate line elements in PDF,
41
+ * not properties of text. Association would require spatial analysis.
42
+ *
43
+ * **Parsing Approach:**
44
+ * 1. Load PDF document using pdfjs-dist.
45
+ * 2. Extract global metadata from document info dictionary.
46
+ * 3. Extract embedded file attachments.
47
+ * 4. Iterate through each page:
48
+ * a. Collect text items with position and formatting.
49
+ * b. Collect link annotations with associated text.
50
+ * c. Collect images from the operator list.
51
+ * d. Sort all items by vertical position (top-to-bottom reading order).
52
+ * e. Group text into paragraphs/headings based on line breaks and font sizes.
53
+ * f. Process images as attachments with optional OCR.
54
+ * 5. Apply heading detection based on font size heuristics.
55
+ *
56
+ * @module PdfParser
57
+ * @see https://mozilla.github.io/pdf.js/ PDF.js documentation
58
+ * @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
59
+ */
60
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
61
+ if (k2 === undefined) k2 = k;
62
+ var desc = Object.getOwnPropertyDescriptor(m, k);
63
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
64
+ desc = { enumerable: true, get: function() { return m[k]; } };
65
+ }
66
+ Object.defineProperty(o, k2, desc);
67
+ }) : (function(o, m, k, k2) {
68
+ if (k2 === undefined) k2 = k;
69
+ o[k2] = m[k];
70
+ }));
71
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
72
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
73
+ }) : function(o, v) {
74
+ o["default"] = v;
75
+ });
76
+ var __importStar = (this && this.__importStar) || function (mod) {
77
+ if (mod && mod.__esModule) return mod;
78
+ var result = {};
79
+ if (mod != null) for (var k in mod) if (k !== "default" && Object.prototype.hasOwnProperty.call(mod, k)) __createBinding(result, mod, k);
80
+ __setModuleDefault(result, mod);
81
+ return result;
82
+ };
83
+ Object.defineProperty(exports, "__esModule", { value: true });
84
+ exports.parsePdf = void 0;
85
+ const errorUtils_1 = require("../utils/errorUtils");
86
+ const imageUtils_1 = require("../utils/imageUtils");
87
+ const ocrUtils_1 = require("../utils/ocrUtils");
88
+ /**
89
+ * Parses PDF creation/modification date strings.
90
+ * PDF dates are in format: D:YYYYMMDDHHmmSSOHH'mm'
91
+ * Where O is timezone offset direction (+/-), HH is hours, mm is minutes.
92
+ *
93
+ * @param dateString - The PDF date string
94
+ * @returns Parsed Date object or undefined if parsing fails
95
+ */
96
+ function parsePdfDate(dateString) {
97
+ if (!dateString)
98
+ return undefined;
99
+ try {
100
+ // Remove "D:" prefix if present
101
+ let str = dateString.startsWith('D:') ? dateString.slice(2) : dateString;
102
+ // Extract components: YYYYMMDDHHmmSS
103
+ const year = parseInt(str.slice(0, 4), 10);
104
+ const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
105
+ const day = parseInt(str.slice(6, 8), 10) || 1;
106
+ const hour = parseInt(str.slice(8, 10), 10) || 0;
107
+ const minute = parseInt(str.slice(10, 12), 10) || 0;
108
+ const second = parseInt(str.slice(12, 14), 10) || 0;
109
+ // Handle timezone if present
110
+ const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
111
+ if (tzMatch) {
112
+ const tzSign = tzMatch[1] === '-' ? -1 : 1;
113
+ const tzHours = parseInt(tzMatch[2], 10) || 0;
114
+ const tzMinutes = parseInt(tzMatch[3], 10) || 0;
115
+ const offset = tzSign * (tzHours * 60 + tzMinutes);
116
+ // Create date in UTC and adjust for timezone
117
+ const utc = Date.UTC(year, month, day, hour, minute, second);
118
+ return new Date(utc - offset * 60000);
119
+ }
120
+ return new Date(year, month, day, hour, minute, second);
121
+ }
122
+ catch {
123
+ // Fallback: try native Date parsing
124
+ try {
125
+ return new Date(dateString);
126
+ }
127
+ catch {
128
+ return undefined;
129
+ }
130
+ }
131
+ }
132
+ /**
133
+ * Calculates statistics about font sizes in the document.
134
+ * Used for heading detection heuristics.
135
+ */
136
+ function calculateFontStats(pageItems) {
137
+ const sizes = [];
138
+ for (const page of pageItems) {
139
+ for (const item of page) {
140
+ if (item.type === 'text' && item.height > 0) {
141
+ sizes.push(item.height);
142
+ }
143
+ }
144
+ }
145
+ if (sizes.length === 0)
146
+ return { median: 12, max: 12 };
147
+ sizes.sort((a, b) => a - b);
148
+ const median = sizes[Math.floor(sizes.length / 2)];
149
+ const max = sizes[sizes.length - 1];
150
+ return { median, max };
151
+ }
152
+ /**
153
+ * Determines if a text item should be considered a heading based on its font size.
154
+ *
155
+ * Heuristic: Text that is at least 20% larger than the median body text size
156
+ * is considered a heading. The level (1-6) is determined by relative size.
157
+ *
158
+ * @param fontSize - The font size of the text
159
+ * @param fontStats - Statistics about fonts in the document
160
+ * @returns Heading level (1-6) or 0 if not a heading
161
+ */
162
+ function detectHeadingLevel(fontSize, fontStats) {
163
+ // If font is less than 20% larger than median, it's not a heading
164
+ if (fontSize <= fontStats.median * 1.2)
165
+ return 0;
166
+ // Calculate heading level based on how much larger than median
167
+ const ratio = fontSize / fontStats.median;
168
+ if (ratio >= 2.0)
169
+ return 1; // 2x or more = H1
170
+ if (ratio >= 1.7)
171
+ return 2; // 1.7x-2x = H2
172
+ if (ratio >= 1.5)
173
+ return 3; // 1.5x-1.7x = H3
174
+ if (ratio >= 1.35)
175
+ return 4; // 1.35x-1.5x = H4
176
+ if (ratio >= 1.2)
177
+ return 5; // 1.2x-1.35x = H5
178
+ return 0; // Below threshold
179
+ }
180
+ /**
181
+ * Checks if a text position falls within a link annotation rectangle.
182
+ */
183
+ /**
184
+ * Checks if a text item falls within a link annotation rectangle.
185
+ * Uses the center point of the text item to avoid false positives for text
186
+ * that starts immediately after a link (which would share the same x coordinate boundary).
187
+ */
188
+ function findLinkForText(item, annotations) {
189
+ const centerX = item.x + (item.width / 2);
190
+ const centerY = item.y + (item.height / 2);
191
+ for (const annot of annotations) {
192
+ // PDF annotation rects are [x1, y1, x2, y2] in page coordinates
193
+ const [x1, y1, x2, y2] = annot.rect;
194
+ const minX = Math.min(x1, x2);
195
+ const maxX = Math.max(x1, x2);
196
+ const minY = Math.min(y1, y2);
197
+ const maxY = Math.max(y1, y2);
198
+ // Check if center point is within the rect (strict, no tolerance)
199
+ if (centerX >= minX && centerX <= maxX &&
200
+ centerY >= minY && centerY <= maxY) {
201
+ return annot;
202
+ }
203
+ }
204
+ return undefined;
205
+ }
206
+ /**
207
+ * Encodes raw RGBA data into a 24-bit BMP buffer with a white background.
208
+ * Transparency (alpha channel) is flattened against white.
209
+ *
210
+ * @param width - Image width
211
+ * @param height - Image height
212
+ * @param data - RGBA pixel data
213
+ * @returns BMP Buffer
214
+ */
215
+ function encodeBmp(width, height, data) {
216
+ // BMP row size must be a multiple of 4 bytes
217
+ const rowSize = Math.floor((24 * width + 31) / 32) * 4;
218
+ const padding = rowSize - (width * 3);
219
+ const headerSize = 54; // 14 (File Header) + 40 (DIB Header)
220
+ const imageSize = rowSize * height;
221
+ const fileSize = headerSize + imageSize;
222
+ const buffer = Buffer.alloc(fileSize);
223
+ // --- File Header (14 bytes) ---
224
+ buffer.write('BM', 0); // Signature
225
+ buffer.writeUInt32LE(fileSize, 2); // File Size
226
+ buffer.writeUInt32LE(0, 6); // Reserved
227
+ buffer.writeUInt32LE(headerSize, 10); // Offset to pixel data
228
+ // --- DIB Header (BITMAPINFOHEADER - 40 bytes) ---
229
+ buffer.writeUInt32LE(40, 14); // Header Size
230
+ buffer.writeInt32LE(width, 18); // Width
231
+ buffer.writeInt32LE(-height, 22); // Height (negative for top-down)
232
+ buffer.writeUInt16LE(1, 26); // Planes
233
+ buffer.writeUInt16LE(24, 28); // Bit Count (24-bit RGB)
234
+ buffer.writeUInt32LE(0, 30); // Compression (BI_RGB)
235
+ buffer.writeUInt32LE(imageSize, 34); // Image Size
236
+ buffer.writeInt32LE(2835, 38); // X PixelsPerMeter (72 DPI)
237
+ buffer.writeInt32LE(2835, 42); // Y PixelsPerMeter (72 DPI)
238
+ buffer.writeUInt32LE(0, 46); // Colors Used
239
+ buffer.writeUInt32LE(0, 50); // Colors Important
240
+ // --- Pixel Data ---
241
+ let offset = headerSize;
242
+ for (let y = 0; y < height; y++) {
243
+ for (let x = 0; x < width; x++) {
244
+ const i = (y * width + x) * 4;
245
+ // RGBA input
246
+ const r = data[i + 0];
247
+ const g = data[i + 1];
248
+ const b = data[i + 2];
249
+ const a = data[i + 3];
250
+ // Flatten alpha against white background
251
+ // out = alpha * pixel + (1 - alpha) * white
252
+ // white = 255
253
+ const alpha = a / 255;
254
+ const outR = Math.round(r * alpha + 255 * (1 - alpha));
255
+ const outG = Math.round(g * alpha + 255 * (1 - alpha));
256
+ const outB = Math.round(b * alpha + 255 * (1 - alpha));
257
+ // Write as BGR (BMP standard)
258
+ buffer[offset + 0] = outB;
259
+ buffer[offset + 1] = outG;
260
+ buffer[offset + 2] = outR;
261
+ offset += 3;
262
+ }
263
+ // Write padding
264
+ for (let p = 0; p < padding; p++) {
265
+ buffer[offset] = 0;
266
+ offset++;
267
+ }
268
+ }
269
+ return buffer;
270
+ }
271
+ /**
272
+ * Converts raw PDF image data to a buffer for attachment extraction.
273
+ *
274
+ * **Important Limitation:**
275
+ * PDF images are stored as raw pixel data (RGB, RGBA, or grayscale), not as encoded
276
+ * image files like PNG or JPEG. This function converts the raw data to a normalized
277
+ * RGBA buffer, but this is NOT a valid image file format.
278
+ *
279
+ * For display, the raw RGBA data would need to be encoded to PNG/JPEG, which requires
280
+ * an additional library like `sharp` or `pngjs`. Currently, this is stored as raw bytes.
281
+ *
282
+ * OCR is NOT supported for PDF images because Tesseract.js requires encoded image files
283
+ * (PNG, JPEG, etc.), not raw pixel data. To enable OCR, a PNG encoder would need to be added.
284
+ *
285
+ * @param data - Raw pixel data from PDF.js
286
+ * @param width - Image width in pixels
287
+ * @param height - Image height in pixels
288
+ * @param kind - PDF.js image kind (1=Grayscale, 2=RGB, 3=RGBA)
289
+ * @returns Buffer containing RGBA pixel data (NOT an encoded image file)
290
+ */
291
+ function convertToRgbaBuffer(data, width, height, kind) {
292
+ // PDF.js image kind values:
293
+ // 1 = GRAYSCALE
294
+ // 2 = RGB
295
+ // 3 = RGBA
296
+ let rgbaData;
297
+ if (kind === 1) {
298
+ // Grayscale - expand to RGBA
299
+ rgbaData = new Uint8ClampedArray(width * height * 4);
300
+ for (let i = 0; i < width * height; i++) {
301
+ const gray = data[i];
302
+ rgbaData[i * 4] = gray;
303
+ rgbaData[i * 4 + 1] = gray;
304
+ rgbaData[i * 4 + 2] = gray;
305
+ rgbaData[i * 4 + 3] = 255;
306
+ }
307
+ }
308
+ else if (kind === 2 || data.length === width * height * 3) {
309
+ // RGB - add alpha channel
310
+ rgbaData = new Uint8ClampedArray(width * height * 4);
311
+ for (let i = 0; i < width * height; i++) {
312
+ rgbaData[i * 4] = data[i * 3];
313
+ rgbaData[i * 4 + 1] = data[i * 3 + 1];
314
+ rgbaData[i * 4 + 2] = data[i * 3 + 2];
315
+ rgbaData[i * 4 + 3] = 255;
316
+ }
317
+ }
318
+ else {
319
+ // Assume RGBA
320
+ rgbaData = data instanceof Uint8ClampedArray ? data : new Uint8ClampedArray(data);
321
+ }
322
+ return Buffer.from(rgbaData.buffer, rgbaData.byteOffset, rgbaData.byteLength);
323
+ }
324
+ /**
325
+ * Parses a PDF file and extracts content.
326
+ *
327
+ * @param buffer - The PDF file buffer
328
+ * @param config - Parser configuration
329
+ * @returns Promise resolving to the parsed AST
330
+ */
331
+ const parsePdf = async (buffer, config) => {
332
+ let pdfjs;
333
+ // Check if we are in a Node.js environment
334
+ // @ts-ignore
335
+ if (typeof window === 'undefined') {
336
+ try {
337
+ // Use legacy build for Node.js (required for pdfjs-dist v5+)
338
+ pdfjs = await Promise.resolve().then(() => __importStar(require('pdfjs-dist/legacy/build/pdf.mjs')));
339
+ }
340
+ catch (e) {
341
+ // Fallback to standard if legacy not found (e.g. older versions)
342
+ pdfjs = await Promise.resolve().then(() => __importStar(require('pdfjs-dist')));
343
+ }
344
+ }
345
+ else {
346
+ pdfjs = await Promise.resolve().then(() => __importStar(require('pdfjs-dist')));
347
+ }
348
+ // Configure worker
349
+ if (config.pdfWorkerSrc) {
350
+ pdfjs.GlobalWorkerOptions.workerSrc = config.pdfWorkerSrc;
351
+ }
352
+ else {
353
+ // Fallbacks when no workerSrc is provided
354
+ // @ts-ignore
355
+ if (typeof window !== 'undefined') {
356
+ // Browser: Default to CDN
357
+ pdfjs.GlobalWorkerOptions.workerSrc = 'https://unpkg.com/pdfjs-dist@5.4.530/build/pdf.worker.min.mjs';
358
+ }
359
+ else {
360
+ // Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
361
+ try {
362
+ // We use require.resolve to find the exact path of the installed package.
363
+ // @ts-ignore - 'require' is available in Node.js/CommonJS environment
364
+ const workerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
365
+ pdfjs.GlobalWorkerOptions.workerSrc = workerPath;
366
+ }
367
+ catch (e) {
368
+ if (config.outputErrorToConsole)
369
+ console.warn("[PdfParser] Could not auto-resolve local worker path:", e);
370
+ }
371
+ }
372
+ }
373
+ const uint8Array = new Uint8Array(buffer);
374
+ const loadingTask = pdfjs.getDocument({
375
+ data: uint8Array,
376
+ verbosity: 0 // ERRORS only, suppresses warnings
377
+ });
378
+ // Handle loading errors, specifically missing worker in browser
379
+ let pdfDocument;
380
+ try {
381
+ pdfDocument = await loadingTask.promise;
382
+ }
383
+ catch (e) {
384
+ if (e.message?.includes('workerSrc') || e.message?.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
385
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.PDF_WORKER_MISSING, config);
386
+ }
387
+ throw e;
388
+ }
389
+ const content = [];
390
+ const attachments = [];
391
+ const numPages = pdfDocument.numPages;
392
+ // Collect all page items for font statistics before processing
393
+ const allPageItems = [];
394
+ // --- Metadata Extraction ---
395
+ // Extract all available metadata from the PDF info dictionary.
396
+ // Note: Some metadata fields depend on how the PDF was created.
397
+ // - Producer: Software that created the PDF
398
+ // - Creator: Application that made the original document
399
+ // - Keywords, Description are rarely present
400
+ const meta = await pdfDocument.getMetadata().catch(() => ({ info: {} }));
401
+ const info = meta.info;
402
+ const metadata = {
403
+ pages: numPages,
404
+ title: info?.Title || undefined,
405
+ author: info?.Author || undefined,
406
+ subject: info?.Subject || undefined,
407
+ description: info?.Keywords || undefined,
408
+ created: parsePdfDate(info?.CreationDate),
409
+ modified: parsePdfDate(info?.ModDate),
410
+ // Note: lastModifiedBy is not available in PDF format - there's no concept of "last modifier"
411
+ // The Author field only tracks original author.
412
+ };
413
+ // --- Embedded File Attachment Extraction ---
414
+ /**
415
+ * PDF can contain embedded files (not images in content, but attached files).
416
+ * These are separate from images in the page content stream.
417
+ */
418
+ try {
419
+ const embeddedFiles = await pdfDocument.getAttachments();
420
+ if (embeddedFiles && config.extractAttachments) {
421
+ for (const name in embeddedFiles) {
422
+ const file = embeddedFiles[name];
423
+ const fileBuffer = Buffer.from(file.content);
424
+ const attachment = (0, imageUtils_1.createAttachment)(file.filename, fileBuffer);
425
+ attachments.push(attachment);
426
+ }
427
+ }
428
+ }
429
+ catch (e) {
430
+ if (config.outputErrorToConsole)
431
+ console.error("Error extracting embedded attachments:", e);
432
+ }
433
+ // --- First Pass: Collect all items for font statistics ---
434
+ for (let i = 1; i <= numPages; i++) {
435
+ let page;
436
+ let textContent;
437
+ const pageItems = [];
438
+ try {
439
+ page = await pdfDocument.getPage(i);
440
+ // Extract text content
441
+ textContent = await page.getTextContent();
442
+ }
443
+ catch (e) {
444
+ if (config.outputErrorToConsole)
445
+ console.warn(`[PdfParser] Error loading page ${i}:`, e);
446
+ // Push empty items to maintain index alignment for second pass
447
+ allPageItems.push(pageItems);
448
+ continue;
449
+ }
450
+ const commonObjs = page.commonObjs;
451
+ for (const item of textContent.items) {
452
+ const transform = item.transform;
453
+ const x = transform[4];
454
+ const y = transform[5];
455
+ const width = item.width || 0;
456
+ const height = item.height || Math.abs(transform[3]) || 12;
457
+ // Extract formatting from font
458
+ const formatting = {};
459
+ let fontName;
460
+ if (item.fontName && commonObjs) {
461
+ try {
462
+ if (commonObjs.has(item.fontName)) {
463
+ // Use callback-based get to ensure safe resolution
464
+ const fontData = await new Promise((resolve) => {
465
+ // @ts-ignore
466
+ commonObjs.get(item.fontName, (data) => resolve(data));
467
+ });
468
+ if (fontData?.name) {
469
+ // Remove PDF subset prefix (6 uppercase letters + '+')
470
+ fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
471
+ formatting.font = fontName;
472
+ // Detect bold/italic from font name
473
+ const lowerName = fontData.name.toLowerCase();
474
+ if (lowerName.includes('bold'))
475
+ formatting.bold = true;
476
+ if (lowerName.includes('italic') || lowerName.includes('oblique'))
477
+ formatting.italic = true;
478
+ }
479
+ }
480
+ }
481
+ catch {
482
+ // Font lookup failed, continue without font info
483
+ }
484
+ }
485
+ if (height > 0) {
486
+ formatting.size = Math.round(height).toString();
487
+ }
488
+ pageItems.push({
489
+ type: 'text',
490
+ x,
491
+ y,
492
+ width,
493
+ height,
494
+ text: item.str,
495
+ fontName,
496
+ formatting
497
+ });
498
+ }
499
+ // Extract images if enabled
500
+ if (config.extractAttachments || config.ocr) {
501
+ try {
502
+ const ops = await page.getOperatorList();
503
+ const fnArray = ops.fnArray;
504
+ const argsArray = ops.argsArray;
505
+ for (let j = 0; j < fnArray.length; j++) {
506
+ const fn = fnArray[j];
507
+ if (fn === pdfjs.OPS.dependency) {
508
+ const deps = argsArray[j];
509
+ for (const dep of deps) {
510
+ // In pdfjs-dist v3+, get() throws if not resolved unless a callback is provided.
511
+ // We must use the callback pattern to wait for resolution.
512
+ try {
513
+ if (page.objs.has(dep))
514
+ continue;
515
+ await new Promise((resolve) => {
516
+ const timeout = setTimeout(() => {
517
+ resolve();
518
+ }, 500);
519
+ page.objs.get(dep, (data) => {
520
+ clearTimeout(timeout);
521
+ resolve();
522
+ });
523
+ });
524
+ }
525
+ catch (e) {
526
+ if (config.outputErrorToConsole) {
527
+ console.error(`[PdfParser] Failed to load dependency ${dep}:`, e);
528
+ }
529
+ }
530
+ }
531
+ }
532
+ if (fn === pdfjs.OPS.paintImageXObject || fn === pdfjs.OPS.paintXObject) {
533
+ const imgName = argsArray[j][0];
534
+ try {
535
+ let hasObj = page.objs.has(imgName);
536
+ let targetObjs = page.objs;
537
+ if (!hasObj && page.commonObjs.has(imgName)) {
538
+ hasObj = true;
539
+ targetObjs = page.commonObjs;
540
+ }
541
+ if (hasObj) {
542
+ // Use callback-based get to ensure safe resolution
543
+ const rawObj = await new Promise((resolve) => {
544
+ // @ts-ignore
545
+ targetObjs.get(imgName, (data) => resolve(data));
546
+ });
547
+ const imgObj = rawObj;
548
+ // Browser-specific: Handle ImageBitmap if data is missing
549
+ if (typeof window !== 'undefined' && !imgObj.data && imgObj.bitmap) {
550
+ try {
551
+ const canvas = document.createElement('canvas');
552
+ canvas.width = imgObj.width;
553
+ canvas.height = imgObj.height;
554
+ const ctx = canvas.getContext('2d');
555
+ if (ctx) {
556
+ ctx.drawImage(imgObj.bitmap, 0, 0);
557
+ imgObj.data = ctx.getImageData(0, 0, imgObj.width, imgObj.height).data;
558
+ imgObj.kind = 3; // RGBA
559
+ }
560
+ }
561
+ catch (e) {
562
+ if (config.outputErrorToConsole)
563
+ console.error(`[PdfParser] Failed to extract from ImageBitmap:`, e);
564
+ }
565
+ }
566
+ if (imgObj?.data && imgObj.width > 0 && imgObj.height > 0) {
567
+ // Find position from transform matrix
568
+ let imgX = 0, imgY = 0;
569
+ for (let k = j - 1; k >= 0; k--) {
570
+ if (fnArray[k] === pdfjs.OPS.transform) {
571
+ imgX = argsArray[k][4];
572
+ imgY = argsArray[k][5];
573
+ break;
574
+ }
575
+ }
576
+ pageItems.push({
577
+ type: 'image',
578
+ x: imgX,
579
+ y: imgY,
580
+ name: imgName,
581
+ data: imgObj.data,
582
+ width: imgObj.width,
583
+ height: imgObj.height,
584
+ kind: imgObj.kind
585
+ });
586
+ }
587
+ }
588
+ }
589
+ catch {
590
+ // Image access failed, continue
591
+ }
592
+ }
593
+ }
594
+ }
595
+ catch (e) {
596
+ if (config.outputErrorToConsole)
597
+ console.error(`Error extracting images from page ${i}:`, e);
598
+ }
599
+ }
600
+ allPageItems.push(pageItems);
601
+ }
602
+ // Calculate font statistics for heading detection
603
+ const fontStats = calculateFontStats(allPageItems);
604
+ // --- Second Pass: Process pages with font statistics ---
605
+ for (let i = 0; i < allPageItems.length; i++) {
606
+ const pageNum = i + 1;
607
+ let page;
608
+ try {
609
+ page = await pdfDocument.getPage(pageNum);
610
+ }
611
+ catch (e) {
612
+ if (config.outputErrorToConsole)
613
+ console.warn(`[PdfParser] Error loading page ${pageNum} in second pass:`, e);
614
+ continue;
615
+ }
616
+ const pageItems = allPageItems[i];
617
+ const pageContent = [];
618
+ // Extract link annotations for this page
619
+ const annotations = [];
620
+ try {
621
+ const annots = await page.getAnnotations();
622
+ for (const annot of annots) {
623
+ if (annot.subtype === 'Link' && annot.rect) {
624
+ annotations.push({
625
+ rect: annot.rect,
626
+ url: annot.url,
627
+ dest: annot.dest
628
+ });
629
+ }
630
+ }
631
+ }
632
+ catch (e) {
633
+ if (config.outputErrorToConsole)
634
+ console.error(`Error extracting annotations from page ${pageNum}:`, e);
635
+ }
636
+ // Sort items: Y descending (top to bottom), then X ascending (left to right)
637
+ pageItems.sort((a, b) => {
638
+ if (Math.abs(b.y - a.y) > 5)
639
+ return b.y - a.y;
640
+ return a.x - b.x;
641
+ });
642
+ // Process sorted items into content nodes
643
+ let currentNode = null;
644
+ let currentNodeFontSize = 0;
645
+ let lastY = -1;
646
+ let imageCounter = 0;
647
+ for (const item of pageItems) {
648
+ if (item.type === 'text') {
649
+ const text = item.text;
650
+ if (!text)
651
+ continue;
652
+ // Check for new line
653
+ const isNewLine = lastY !== -1 && Math.abs(item.y - lastY) > 5;
654
+ if (isNewLine && currentNode) {
655
+ // Finalize and push current node
656
+ if ((currentNode.text || '').trim().length > 0) {
657
+ pageContent.push(currentNode);
658
+ }
659
+ currentNode = null;
660
+ }
661
+ // Skip pure whitespace at start of lines
662
+ if (!currentNode && text.trim().length === 0) {
663
+ lastY = item.y;
664
+ continue;
665
+ }
666
+ // Determine if this should be a heading
667
+ const headingLevel = detectHeadingLevel(item.height, fontStats);
668
+ if (!currentNode) {
669
+ // Start new node
670
+ if (headingLevel > 0) {
671
+ currentNode = {
672
+ type: 'heading',
673
+ text: '',
674
+ children: [],
675
+ metadata: { level: headingLevel }
676
+ };
677
+ }
678
+ else {
679
+ currentNode = {
680
+ type: 'paragraph',
681
+ text: '',
682
+ children: []
683
+ };
684
+ }
685
+ currentNodeFontSize = item.height;
686
+ }
687
+ // Handle whitespace
688
+ if (text.trim().length === 0) {
689
+ if (currentNode.children && currentNode.children.length > 0) {
690
+ const lastChild = currentNode.children[currentNode.children.length - 1];
691
+ if (lastChild.type === 'text' && lastChild.text) {
692
+ lastChild.text += text;
693
+ currentNode.text += text;
694
+ }
695
+ }
696
+ lastY = item.y;
697
+ continue;
698
+ }
699
+ // Add space between words if needed
700
+ if (currentNode.text && currentNode.text.length > 0 && !currentNode.text.endsWith(' ')) {
701
+ currentNode.text += ' ';
702
+ if (currentNode.children && currentNode.children.length > 0) {
703
+ const lastChild = currentNode.children[currentNode.children.length - 1];
704
+ if (lastChild.type === 'text' && lastChild.text) {
705
+ lastChild.text += ' ';
706
+ }
707
+ }
708
+ }
709
+ currentNode.text += text;
710
+ // Check for link
711
+ const link = findLinkForText(item, annotations);
712
+ let textMetadata;
713
+ if (link) {
714
+ if (link.url) {
715
+ textMetadata = {
716
+ link: link.url,
717
+ linkType: link.url.startsWith('#') ? 'internal' : 'external'
718
+ };
719
+ }
720
+ else if (link.dest) {
721
+ // Internal destination
722
+ textMetadata = {
723
+ link: typeof link.dest === 'string' ? `#${link.dest}` : '#internal',
724
+ linkType: 'internal'
725
+ };
726
+ }
727
+ }
728
+ // Try to merge with last child if same formatting and no link change
729
+ let merged = false;
730
+ if (currentNode.children && currentNode.children.length > 0 && !textMetadata) {
731
+ const lastChild = currentNode.children[currentNode.children.length - 1];
732
+ if (lastChild.type === 'text' &&
733
+ isSameFormatting(lastChild.formatting, item.formatting) &&
734
+ !lastChild.metadata) {
735
+ lastChild.text = (lastChild.text || '') + text;
736
+ merged = true;
737
+ }
738
+ }
739
+ if (!merged) {
740
+ const textNode = {
741
+ type: 'text',
742
+ text: text,
743
+ formatting: Object.keys(item.formatting).length > 0 ? item.formatting : undefined
744
+ };
745
+ if (textMetadata) {
746
+ textNode.metadata = textMetadata;
747
+ }
748
+ currentNode.children?.push(textNode);
749
+ }
750
+ lastY = item.y;
751
+ }
752
+ else if (item.type === 'image') {
753
+ // Flush current node
754
+ if (currentNode) {
755
+ if ((currentNode.text || '').trim().length > 0) {
756
+ pageContent.push(currentNode);
757
+ }
758
+ currentNode = null;
759
+ }
760
+ imageCounter++;
761
+ // Note: Using .bmp extension since we encode to BMP for broad compatibility
762
+ const attachmentName = `pdf_image_p${pageNum}_${imageCounter}.bmp`;
763
+ /**
764
+ * Image extraction for PDF files.
765
+ *
766
+ * PDF stores images as raw pixel data. We convert to BMP for compatibility.
767
+ */
768
+ if (config.extractAttachments) {
769
+ try {
770
+ const imageBuffer = convertToRgbaBuffer(item.data, item.width, item.height, item.kind);
771
+ // Encode as BMP
772
+ const bmpBuffer = encodeBmp(item.width, item.height, new Uint8Array(imageBuffer));
773
+ const attachment = (0, imageUtils_1.createAttachment)(attachmentName, bmpBuffer);
774
+ attachment.mimeType = 'image/bmp';
775
+ // Perform OCR if enabled
776
+ if (config.ocr) {
777
+ try {
778
+ // Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
779
+ if (item.width >= 10 && item.height >= 10) {
780
+ attachment.ocrText = (await (0, ocrUtils_1.performOcr)(bmpBuffer, config.ocrLanguage)).trim();
781
+ }
782
+ }
783
+ catch (e) {
784
+ if (config.outputErrorToConsole)
785
+ console.error(`OCR failed for ${attachmentName}:`, e);
786
+ }
787
+ }
788
+ attachments.push(attachment);
789
+ // Create image content node
790
+ const imageMetadata = {
791
+ attachmentName,
792
+ };
793
+ pageContent.push({
794
+ type: 'image',
795
+ text: attachment.ocrText || '',
796
+ metadata: { ...imageMetadata }
797
+ });
798
+ }
799
+ catch (e) {
800
+ (0, errorUtils_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
801
+ }
802
+ }
803
+ }
804
+ }
805
+ // Flush last node
806
+ if (currentNode && (currentNode.text || '').trim().length > 0) {
807
+ pageContent.push(currentNode);
808
+ }
809
+ // Add page node to content
810
+ content.push({
811
+ type: 'page',
812
+ children: pageContent,
813
+ text: pageContent.map(node => node.text).join(config.newlineDelimiter ?? '\n\n'),
814
+ metadata: { pageNumber: pageNum }
815
+ });
816
+ }
817
+ return {
818
+ type: 'pdf',
819
+ metadata: metadata,
820
+ content: content,
821
+ attachments: attachments,
822
+ toText: () => content.map(c => c.text).join(config.newlineDelimiter ?? '\n\n')
823
+ };
824
+ };
825
+ exports.parsePdf = parsePdf;
826
+ /**
827
+ * Helper to compare two text formatting objects.
828
+ * Returns true if both have the same properties with the same values.
829
+ */
830
+ function isSameFormatting(a, b) {
831
+ if (!a && !b)
832
+ return true;
833
+ if (!a || !b)
834
+ return false;
835
+ const keysA = Object.keys(a).sort();
836
+ const keysB = Object.keys(b).sort();
837
+ if (keysA.length !== keysB.length)
838
+ return false;
839
+ for (let i = 0; i < keysA.length; i++) {
840
+ const key = keysA[i];
841
+ if (keysA[i] !== keysB[i])
842
+ return false;
843
+ if (a[key] !== b[key])
844
+ return false;
845
+ }
846
+ return true;
847
+ }