officeparser 5.2.2 → 6.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +152 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +850 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -790
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,850 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* PDF Parser
|
|
4
|
+
*
|
|
5
|
+
* Extracts text, metadata, images, links, and attachments from PDF files using PDF.js (pdfjs-dist).
|
|
6
|
+
*
|
|
7
|
+
* **Features:**
|
|
8
|
+
* - Text extraction with formatting (bold, italic, font, size)
|
|
9
|
+
* - Comprehensive metadata extraction (title, author, subject, creator, producer, creation/modification dates)
|
|
10
|
+
* - Hyperlink extraction from PDF annotations
|
|
11
|
+
* - Heading detection via font size heuristics
|
|
12
|
+
* - Image extraction as attachments with optional OCR (using Tesseract.js)
|
|
13
|
+
* - Embedded file attachment extraction
|
|
14
|
+
* - Layout preservation (respects order of text and images)
|
|
15
|
+
*
|
|
16
|
+
* **PDF Format Limitations (compared to DOCX/ODT):**
|
|
17
|
+
*
|
|
18
|
+
* PDF was designed as a "page description language" for visual fidelity, not semantic structure.
|
|
19
|
+
* The following features **cannot be reliably extracted** from PDFs:
|
|
20
|
+
*
|
|
21
|
+
* - **Tables**: PDF has no table structure. Tables are just text positioned to look tabular.
|
|
22
|
+
* Extracting tables would require complex spatial analysis with many false positives.
|
|
23
|
+
* See: https://stackoverflow.com/questions/36978446/why-is-it-difficult-to-extract-data-from-pdfs
|
|
24
|
+
*
|
|
25
|
+
* - **Lists**: PDF has no list structure. Bullets/numbers are just text characters.
|
|
26
|
+
* No hierarchy or list type information is stored. Would require heuristic detection
|
|
27
|
+
* that would have many edge cases and errors.
|
|
28
|
+
*
|
|
29
|
+
* - **Styles**: PDF has no style definitions like "Heading1" or "Normal". Only visual
|
|
30
|
+
* properties (font, size) exist. We use font size heuristics to detect headings.
|
|
31
|
+
*
|
|
32
|
+
* - **Notes (Footnotes/Endnotes)**: PDF has no concept of footnotes/endnotes as structured
|
|
33
|
+
* elements. They're just smaller text at the bottom of pages.
|
|
34
|
+
*
|
|
35
|
+
* - **Text Color**: While PDF stores color, pdfjs-dist doesn't expose text color in the
|
|
36
|
+
* textContent API. Would require parsing the operator stream which is complex.
|
|
37
|
+
*
|
|
38
|
+
* - **Background Color**: Same limitation as text color.
|
|
39
|
+
*
|
|
40
|
+
* - **Underline/Strikethrough**: These are drawn as separate line elements in PDF,
|
|
41
|
+
* not properties of text. Association would require spatial analysis.
|
|
42
|
+
*
|
|
43
|
+
* **Parsing Approach:**
|
|
44
|
+
* 1. Load PDF document using pdfjs-dist.
|
|
45
|
+
* 2. Extract global metadata from document info dictionary.
|
|
46
|
+
* 3. Extract embedded file attachments.
|
|
47
|
+
* 4. Iterate through each page:
|
|
48
|
+
* a. Collect text items with position and formatting.
|
|
49
|
+
* b. Collect link annotations with associated text.
|
|
50
|
+
* c. Collect images from the operator list.
|
|
51
|
+
* d. Sort all items by vertical position (top-to-bottom reading order).
|
|
52
|
+
* e. Group text into paragraphs/headings based on line breaks and font sizes.
|
|
53
|
+
* f. Process images as attachments with optional OCR.
|
|
54
|
+
* 5. Apply heading detection based on font size heuristics.
|
|
55
|
+
*
|
|
56
|
+
* @module PdfParser
|
|
57
|
+
* @see https://mozilla.github.io/pdf.js/ PDF.js documentation
|
|
58
|
+
* @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
|
|
59
|
+
*/
|
|
60
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
61
|
+
if (k2 === undefined) k2 = k;
|
|
62
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
63
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
64
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
65
|
+
}
|
|
66
|
+
Object.defineProperty(o, k2, desc);
|
|
67
|
+
}) : (function(o, m, k, k2) {
|
|
68
|
+
if (k2 === undefined) k2 = k;
|
|
69
|
+
o[k2] = m[k];
|
|
70
|
+
}));
|
|
71
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
72
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
73
|
+
}) : function(o, v) {
|
|
74
|
+
o["default"] = v;
|
|
75
|
+
});
|
|
76
|
+
var __importStar = (this && this.__importStar) || function (mod) {
|
|
77
|
+
if (mod && mod.__esModule) return mod;
|
|
78
|
+
var result = {};
|
|
79
|
+
if (mod != null) for (var k in mod) if (k !== "default" && Object.prototype.hasOwnProperty.call(mod, k)) __createBinding(result, mod, k);
|
|
80
|
+
__setModuleDefault(result, mod);
|
|
81
|
+
return result;
|
|
82
|
+
};
|
|
83
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
84
|
+
exports.parsePdf = void 0;
|
|
85
|
+
const errorUtils_1 = require("../utils/errorUtils");
|
|
86
|
+
const imageUtils_1 = require("../utils/imageUtils");
|
|
87
|
+
const ocrUtils_1 = require("../utils/ocrUtils");
|
|
88
|
+
/**
|
|
89
|
+
* Parses PDF creation/modification date strings.
|
|
90
|
+
* PDF dates are in format: D:YYYYMMDDHHmmSSOHH'mm'
|
|
91
|
+
* Where O is timezone offset direction (+/-), HH is hours, mm is minutes.
|
|
92
|
+
*
|
|
93
|
+
* @param dateString - The PDF date string
|
|
94
|
+
* @returns Parsed Date object or undefined if parsing fails
|
|
95
|
+
*/
|
|
96
|
+
function parsePdfDate(dateString) {
|
|
97
|
+
if (!dateString)
|
|
98
|
+
return undefined;
|
|
99
|
+
try {
|
|
100
|
+
// Remove "D:" prefix if present
|
|
101
|
+
let str = dateString.startsWith('D:') ? dateString.slice(2) : dateString;
|
|
102
|
+
// Extract components: YYYYMMDDHHmmSS
|
|
103
|
+
const year = parseInt(str.slice(0, 4), 10);
|
|
104
|
+
const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
|
|
105
|
+
const day = parseInt(str.slice(6, 8), 10) || 1;
|
|
106
|
+
const hour = parseInt(str.slice(8, 10), 10) || 0;
|
|
107
|
+
const minute = parseInt(str.slice(10, 12), 10) || 0;
|
|
108
|
+
const second = parseInt(str.slice(12, 14), 10) || 0;
|
|
109
|
+
// Handle timezone if present
|
|
110
|
+
const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
|
|
111
|
+
if (tzMatch) {
|
|
112
|
+
const tzSign = tzMatch[1] === '-' ? -1 : 1;
|
|
113
|
+
const tzHours = parseInt(tzMatch[2], 10) || 0;
|
|
114
|
+
const tzMinutes = parseInt(tzMatch[3], 10) || 0;
|
|
115
|
+
const offset = tzSign * (tzHours * 60 + tzMinutes);
|
|
116
|
+
// Create date in UTC and adjust for timezone
|
|
117
|
+
const utc = Date.UTC(year, month, day, hour, minute, second);
|
|
118
|
+
return new Date(utc - offset * 60000);
|
|
119
|
+
}
|
|
120
|
+
return new Date(year, month, day, hour, minute, second);
|
|
121
|
+
}
|
|
122
|
+
catch {
|
|
123
|
+
// Fallback: try native Date parsing
|
|
124
|
+
try {
|
|
125
|
+
return new Date(dateString);
|
|
126
|
+
}
|
|
127
|
+
catch {
|
|
128
|
+
return undefined;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
/**
|
|
133
|
+
* Calculates statistics about font sizes in the document.
|
|
134
|
+
* Used for heading detection heuristics.
|
|
135
|
+
*/
|
|
136
|
+
function calculateFontStats(pageItems) {
|
|
137
|
+
const sizes = [];
|
|
138
|
+
for (const page of pageItems) {
|
|
139
|
+
for (const item of page) {
|
|
140
|
+
if (item.type === 'text' && item.height > 0) {
|
|
141
|
+
sizes.push(item.height);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
if (sizes.length === 0)
|
|
146
|
+
return { median: 12, max: 12 };
|
|
147
|
+
sizes.sort((a, b) => a - b);
|
|
148
|
+
const median = sizes[Math.floor(sizes.length / 2)];
|
|
149
|
+
const max = sizes[sizes.length - 1];
|
|
150
|
+
return { median, max };
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* Determines if a text item should be considered a heading based on its font size.
|
|
154
|
+
*
|
|
155
|
+
* Heuristic: Text that is at least 20% larger than the median body text size
|
|
156
|
+
* is considered a heading. The level (1-6) is determined by relative size.
|
|
157
|
+
*
|
|
158
|
+
* @param fontSize - The font size of the text
|
|
159
|
+
* @param fontStats - Statistics about fonts in the document
|
|
160
|
+
* @returns Heading level (1-6) or 0 if not a heading
|
|
161
|
+
*/
|
|
162
|
+
function detectHeadingLevel(fontSize, fontStats) {
|
|
163
|
+
// If font is less than 20% larger than median, it's not a heading
|
|
164
|
+
if (fontSize <= fontStats.median * 1.2)
|
|
165
|
+
return 0;
|
|
166
|
+
// Calculate heading level based on how much larger than median
|
|
167
|
+
const ratio = fontSize / fontStats.median;
|
|
168
|
+
if (ratio >= 2.0)
|
|
169
|
+
return 1; // 2x or more = H1
|
|
170
|
+
if (ratio >= 1.7)
|
|
171
|
+
return 2; // 1.7x-2x = H2
|
|
172
|
+
if (ratio >= 1.5)
|
|
173
|
+
return 3; // 1.5x-1.7x = H3
|
|
174
|
+
if (ratio >= 1.35)
|
|
175
|
+
return 4; // 1.35x-1.5x = H4
|
|
176
|
+
if (ratio >= 1.2)
|
|
177
|
+
return 5; // 1.2x-1.35x = H5
|
|
178
|
+
return 0; // Below threshold
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* Checks if a text position falls within a link annotation rectangle.
|
|
182
|
+
*/
|
|
183
|
+
/**
|
|
184
|
+
* Checks if a text item falls within a link annotation rectangle.
|
|
185
|
+
* Uses the center point of the text item to avoid false positives for text
|
|
186
|
+
* that starts immediately after a link (which would share the same x coordinate boundary).
|
|
187
|
+
*/
|
|
188
|
+
function findLinkForText(item, annotations) {
|
|
189
|
+
const centerX = item.x + (item.width / 2);
|
|
190
|
+
const centerY = item.y + (item.height / 2);
|
|
191
|
+
for (const annot of annotations) {
|
|
192
|
+
// PDF annotation rects are [x1, y1, x2, y2] in page coordinates
|
|
193
|
+
const [x1, y1, x2, y2] = annot.rect;
|
|
194
|
+
const minX = Math.min(x1, x2);
|
|
195
|
+
const maxX = Math.max(x1, x2);
|
|
196
|
+
const minY = Math.min(y1, y2);
|
|
197
|
+
const maxY = Math.max(y1, y2);
|
|
198
|
+
// Check if center point is within the rect (strict, no tolerance)
|
|
199
|
+
if (centerX >= minX && centerX <= maxX &&
|
|
200
|
+
centerY >= minY && centerY <= maxY) {
|
|
201
|
+
return annot;
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
return undefined;
|
|
205
|
+
}
|
|
206
|
+
/**
|
|
207
|
+
* Encodes raw RGBA data into a 24-bit BMP buffer with a white background.
|
|
208
|
+
* Transparency (alpha channel) is flattened against white.
|
|
209
|
+
*
|
|
210
|
+
* @param width - Image width
|
|
211
|
+
* @param height - Image height
|
|
212
|
+
* @param data - RGBA pixel data
|
|
213
|
+
* @returns BMP Buffer
|
|
214
|
+
*/
|
|
215
|
+
function encodeBmp(width, height, data) {
|
|
216
|
+
// BMP row size must be a multiple of 4 bytes
|
|
217
|
+
const rowSize = Math.floor((24 * width + 31) / 32) * 4;
|
|
218
|
+
const padding = rowSize - (width * 3);
|
|
219
|
+
const headerSize = 54; // 14 (File Header) + 40 (DIB Header)
|
|
220
|
+
const imageSize = rowSize * height;
|
|
221
|
+
const fileSize = headerSize + imageSize;
|
|
222
|
+
const buffer = Buffer.alloc(fileSize);
|
|
223
|
+
// --- File Header (14 bytes) ---
|
|
224
|
+
buffer.write('BM', 0); // Signature
|
|
225
|
+
buffer.writeUInt32LE(fileSize, 2); // File Size
|
|
226
|
+
buffer.writeUInt32LE(0, 6); // Reserved
|
|
227
|
+
buffer.writeUInt32LE(headerSize, 10); // Offset to pixel data
|
|
228
|
+
// --- DIB Header (BITMAPINFOHEADER - 40 bytes) ---
|
|
229
|
+
buffer.writeUInt32LE(40, 14); // Header Size
|
|
230
|
+
buffer.writeInt32LE(width, 18); // Width
|
|
231
|
+
buffer.writeInt32LE(-height, 22); // Height (negative for top-down)
|
|
232
|
+
buffer.writeUInt16LE(1, 26); // Planes
|
|
233
|
+
buffer.writeUInt16LE(24, 28); // Bit Count (24-bit RGB)
|
|
234
|
+
buffer.writeUInt32LE(0, 30); // Compression (BI_RGB)
|
|
235
|
+
buffer.writeUInt32LE(imageSize, 34); // Image Size
|
|
236
|
+
buffer.writeInt32LE(2835, 38); // X PixelsPerMeter (72 DPI)
|
|
237
|
+
buffer.writeInt32LE(2835, 42); // Y PixelsPerMeter (72 DPI)
|
|
238
|
+
buffer.writeUInt32LE(0, 46); // Colors Used
|
|
239
|
+
buffer.writeUInt32LE(0, 50); // Colors Important
|
|
240
|
+
// --- Pixel Data ---
|
|
241
|
+
let offset = headerSize;
|
|
242
|
+
for (let y = 0; y < height; y++) {
|
|
243
|
+
for (let x = 0; x < width; x++) {
|
|
244
|
+
const i = (y * width + x) * 4;
|
|
245
|
+
// RGBA input
|
|
246
|
+
const r = data[i + 0];
|
|
247
|
+
const g = data[i + 1];
|
|
248
|
+
const b = data[i + 2];
|
|
249
|
+
const a = data[i + 3];
|
|
250
|
+
// Flatten alpha against white background
|
|
251
|
+
// out = alpha * pixel + (1 - alpha) * white
|
|
252
|
+
// white = 255
|
|
253
|
+
const alpha = a / 255;
|
|
254
|
+
const outR = Math.round(r * alpha + 255 * (1 - alpha));
|
|
255
|
+
const outG = Math.round(g * alpha + 255 * (1 - alpha));
|
|
256
|
+
const outB = Math.round(b * alpha + 255 * (1 - alpha));
|
|
257
|
+
// Write as BGR (BMP standard)
|
|
258
|
+
buffer[offset + 0] = outB;
|
|
259
|
+
buffer[offset + 1] = outG;
|
|
260
|
+
buffer[offset + 2] = outR;
|
|
261
|
+
offset += 3;
|
|
262
|
+
}
|
|
263
|
+
// Write padding
|
|
264
|
+
for (let p = 0; p < padding; p++) {
|
|
265
|
+
buffer[offset] = 0;
|
|
266
|
+
offset++;
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
return buffer;
|
|
270
|
+
}
|
|
271
|
+
/**
|
|
272
|
+
* Converts raw PDF image data to a buffer for attachment extraction.
|
|
273
|
+
*
|
|
274
|
+
* **Important Limitation:**
|
|
275
|
+
* PDF images are stored as raw pixel data (RGB, RGBA, or grayscale), not as encoded
|
|
276
|
+
* image files like PNG or JPEG. This function converts the raw data to a normalized
|
|
277
|
+
* RGBA buffer, but this is NOT a valid image file format.
|
|
278
|
+
*
|
|
279
|
+
* For display, the raw RGBA data would need to be encoded to PNG/JPEG, which requires
|
|
280
|
+
* an additional library like `sharp` or `pngjs`. Currently, this is stored as raw bytes.
|
|
281
|
+
*
|
|
282
|
+
* OCR is NOT supported for PDF images because Tesseract.js requires encoded image files
|
|
283
|
+
* (PNG, JPEG, etc.), not raw pixel data. To enable OCR, a PNG encoder would need to be added.
|
|
284
|
+
*
|
|
285
|
+
* @param data - Raw pixel data from PDF.js
|
|
286
|
+
* @param width - Image width in pixels
|
|
287
|
+
* @param height - Image height in pixels
|
|
288
|
+
* @param kind - PDF.js image kind (1=Grayscale, 2=RGB, 3=RGBA)
|
|
289
|
+
* @returns Buffer containing RGBA pixel data (NOT an encoded image file)
|
|
290
|
+
*/
|
|
291
|
+
function convertToRgbaBuffer(data, width, height, kind) {
|
|
292
|
+
// PDF.js image kind values:
|
|
293
|
+
// 1 = GRAYSCALE
|
|
294
|
+
// 2 = RGB
|
|
295
|
+
// 3 = RGBA
|
|
296
|
+
let rgbaData;
|
|
297
|
+
if (kind === 1) {
|
|
298
|
+
// Grayscale - expand to RGBA
|
|
299
|
+
rgbaData = new Uint8ClampedArray(width * height * 4);
|
|
300
|
+
for (let i = 0; i < width * height; i++) {
|
|
301
|
+
const gray = data[i];
|
|
302
|
+
rgbaData[i * 4] = gray;
|
|
303
|
+
rgbaData[i * 4 + 1] = gray;
|
|
304
|
+
rgbaData[i * 4 + 2] = gray;
|
|
305
|
+
rgbaData[i * 4 + 3] = 255;
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
else if (kind === 2 || data.length === width * height * 3) {
|
|
309
|
+
// RGB - add alpha channel
|
|
310
|
+
rgbaData = new Uint8ClampedArray(width * height * 4);
|
|
311
|
+
for (let i = 0; i < width * height; i++) {
|
|
312
|
+
rgbaData[i * 4] = data[i * 3];
|
|
313
|
+
rgbaData[i * 4 + 1] = data[i * 3 + 1];
|
|
314
|
+
rgbaData[i * 4 + 2] = data[i * 3 + 2];
|
|
315
|
+
rgbaData[i * 4 + 3] = 255;
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
else {
|
|
319
|
+
// Assume RGBA
|
|
320
|
+
rgbaData = data instanceof Uint8ClampedArray ? data : new Uint8ClampedArray(data);
|
|
321
|
+
}
|
|
322
|
+
return Buffer.from(rgbaData.buffer, rgbaData.byteOffset, rgbaData.byteLength);
|
|
323
|
+
}
|
|
324
|
+
/**
|
|
325
|
+
* Parses a PDF file and extracts content.
|
|
326
|
+
*
|
|
327
|
+
* @param buffer - The PDF file buffer
|
|
328
|
+
* @param config - Parser configuration
|
|
329
|
+
* @returns Promise resolving to the parsed AST
|
|
330
|
+
*/
|
|
331
|
+
const parsePdf = async (buffer, config) => {
|
|
332
|
+
let pdfjs;
|
|
333
|
+
// Check if we are in a Node.js environment
|
|
334
|
+
// @ts-ignore
|
|
335
|
+
if (typeof window === 'undefined') {
|
|
336
|
+
// Helper to bypass TS/Webpack/Other transpilers converting import() to require() when compiling to CJS
|
|
337
|
+
// Defined here to avoid CSP 'unsafe-eval' issues in browser environments
|
|
338
|
+
const dynamicImport = new Function('specifier', 'return import(specifier)');
|
|
339
|
+
try {
|
|
340
|
+
// Use legacy build for Node.js (required for pdfjs-dist v5+)
|
|
341
|
+
pdfjs = await dynamicImport('pdfjs-dist/legacy/build/pdf.mjs');
|
|
342
|
+
}
|
|
343
|
+
catch (e) {
|
|
344
|
+
// Fallback to standard if legacy not found (e.g. older versions)
|
|
345
|
+
pdfjs = await dynamicImport('pdfjs-dist');
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
else {
|
|
349
|
+
pdfjs = await Promise.resolve().then(() => __importStar(require('pdfjs-dist')));
|
|
350
|
+
}
|
|
351
|
+
// Configure worker
|
|
352
|
+
if (config.pdfWorkerSrc) {
|
|
353
|
+
pdfjs.GlobalWorkerOptions.workerSrc = config.pdfWorkerSrc;
|
|
354
|
+
}
|
|
355
|
+
else {
|
|
356
|
+
// Fallbacks when no workerSrc is provided
|
|
357
|
+
// @ts-ignore
|
|
358
|
+
if (typeof window !== 'undefined') {
|
|
359
|
+
// Browser: Default to CDN
|
|
360
|
+
pdfjs.GlobalWorkerOptions.workerSrc = 'https://unpkg.com/pdfjs-dist@5.4.530/build/pdf.worker.min.mjs';
|
|
361
|
+
}
|
|
362
|
+
else {
|
|
363
|
+
// Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
|
|
364
|
+
try {
|
|
365
|
+
// We use require.resolve to find the exact path of the installed package.
|
|
366
|
+
// @ts-ignore - 'require' is available in Node.js/CommonJS environment
|
|
367
|
+
const workerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
|
|
368
|
+
pdfjs.GlobalWorkerOptions.workerSrc = workerPath;
|
|
369
|
+
}
|
|
370
|
+
catch (e) {
|
|
371
|
+
if (config.outputErrorToConsole)
|
|
372
|
+
console.warn("[PdfParser] Could not auto-resolve local worker path:", e);
|
|
373
|
+
}
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
const uint8Array = new Uint8Array(buffer);
|
|
377
|
+
const loadingTask = pdfjs.getDocument({
|
|
378
|
+
data: uint8Array,
|
|
379
|
+
verbosity: 0 // ERRORS only, suppresses warnings
|
|
380
|
+
});
|
|
381
|
+
// Handle loading errors, specifically missing worker in browser
|
|
382
|
+
let pdfDocument;
|
|
383
|
+
try {
|
|
384
|
+
pdfDocument = await loadingTask.promise;
|
|
385
|
+
}
|
|
386
|
+
catch (e) {
|
|
387
|
+
if (e.message?.includes('workerSrc') || e.message?.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
|
|
388
|
+
throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.PDF_WORKER_MISSING, config);
|
|
389
|
+
}
|
|
390
|
+
throw e;
|
|
391
|
+
}
|
|
392
|
+
const content = [];
|
|
393
|
+
const attachments = [];
|
|
394
|
+
const numPages = pdfDocument.numPages;
|
|
395
|
+
// Collect all page items for font statistics before processing
|
|
396
|
+
const allPageItems = [];
|
|
397
|
+
// --- Metadata Extraction ---
|
|
398
|
+
// Extract all available metadata from the PDF info dictionary.
|
|
399
|
+
// Note: Some metadata fields depend on how the PDF was created.
|
|
400
|
+
// - Producer: Software that created the PDF
|
|
401
|
+
// - Creator: Application that made the original document
|
|
402
|
+
// - Keywords, Description are rarely present
|
|
403
|
+
const meta = await pdfDocument.getMetadata().catch(() => ({ info: {} }));
|
|
404
|
+
const info = meta.info;
|
|
405
|
+
const metadata = {
|
|
406
|
+
pages: numPages,
|
|
407
|
+
title: info?.Title || undefined,
|
|
408
|
+
author: info?.Author || undefined,
|
|
409
|
+
subject: info?.Subject || undefined,
|
|
410
|
+
description: info?.Keywords || undefined,
|
|
411
|
+
created: parsePdfDate(info?.CreationDate),
|
|
412
|
+
modified: parsePdfDate(info?.ModDate),
|
|
413
|
+
// Note: lastModifiedBy is not available in PDF format - there's no concept of "last modifier"
|
|
414
|
+
// The Author field only tracks original author.
|
|
415
|
+
};
|
|
416
|
+
// --- Embedded File Attachment Extraction ---
|
|
417
|
+
/**
|
|
418
|
+
* PDF can contain embedded files (not images in content, but attached files).
|
|
419
|
+
* These are separate from images in the page content stream.
|
|
420
|
+
*/
|
|
421
|
+
try {
|
|
422
|
+
const embeddedFiles = await pdfDocument.getAttachments();
|
|
423
|
+
if (embeddedFiles && config.extractAttachments) {
|
|
424
|
+
for (const name in embeddedFiles) {
|
|
425
|
+
const file = embeddedFiles[name];
|
|
426
|
+
const fileBuffer = Buffer.from(file.content);
|
|
427
|
+
const attachment = (0, imageUtils_1.createAttachment)(file.filename, fileBuffer);
|
|
428
|
+
attachments.push(attachment);
|
|
429
|
+
}
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
catch (e) {
|
|
433
|
+
if (config.outputErrorToConsole)
|
|
434
|
+
console.error("Error extracting embedded attachments:", e);
|
|
435
|
+
}
|
|
436
|
+
// --- First Pass: Collect all items for font statistics ---
|
|
437
|
+
for (let i = 1; i <= numPages; i++) {
|
|
438
|
+
let page;
|
|
439
|
+
let textContent;
|
|
440
|
+
const pageItems = [];
|
|
441
|
+
try {
|
|
442
|
+
page = await pdfDocument.getPage(i);
|
|
443
|
+
// Extract text content
|
|
444
|
+
textContent = await page.getTextContent();
|
|
445
|
+
}
|
|
446
|
+
catch (e) {
|
|
447
|
+
if (config.outputErrorToConsole)
|
|
448
|
+
console.warn(`[PdfParser] Error loading page ${i}:`, e);
|
|
449
|
+
// Push empty items to maintain index alignment for second pass
|
|
450
|
+
allPageItems.push(pageItems);
|
|
451
|
+
continue;
|
|
452
|
+
}
|
|
453
|
+
const commonObjs = page.commonObjs;
|
|
454
|
+
for (const item of textContent.items) {
|
|
455
|
+
const transform = item.transform;
|
|
456
|
+
const x = transform[4];
|
|
457
|
+
const y = transform[5];
|
|
458
|
+
const width = item.width || 0;
|
|
459
|
+
const height = item.height || Math.abs(transform[3]) || 12;
|
|
460
|
+
// Extract formatting from font
|
|
461
|
+
const formatting = {};
|
|
462
|
+
let fontName;
|
|
463
|
+
if (item.fontName && commonObjs) {
|
|
464
|
+
try {
|
|
465
|
+
if (commonObjs.has(item.fontName)) {
|
|
466
|
+
// Use callback-based get to ensure safe resolution
|
|
467
|
+
const fontData = await new Promise((resolve) => {
|
|
468
|
+
// @ts-ignore
|
|
469
|
+
commonObjs.get(item.fontName, (data) => resolve(data));
|
|
470
|
+
});
|
|
471
|
+
if (fontData?.name) {
|
|
472
|
+
// Remove PDF subset prefix (6 uppercase letters + '+')
|
|
473
|
+
fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
|
|
474
|
+
formatting.font = fontName;
|
|
475
|
+
// Detect bold/italic from font name
|
|
476
|
+
const lowerName = fontData.name.toLowerCase();
|
|
477
|
+
if (lowerName.includes('bold'))
|
|
478
|
+
formatting.bold = true;
|
|
479
|
+
if (lowerName.includes('italic') || lowerName.includes('oblique'))
|
|
480
|
+
formatting.italic = true;
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
catch {
|
|
485
|
+
// Font lookup failed, continue without font info
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
if (height > 0) {
|
|
489
|
+
formatting.size = Math.round(height).toString();
|
|
490
|
+
}
|
|
491
|
+
pageItems.push({
|
|
492
|
+
type: 'text',
|
|
493
|
+
x,
|
|
494
|
+
y,
|
|
495
|
+
width,
|
|
496
|
+
height,
|
|
497
|
+
text: item.str,
|
|
498
|
+
fontName,
|
|
499
|
+
formatting
|
|
500
|
+
});
|
|
501
|
+
}
|
|
502
|
+
// Extract images if enabled
|
|
503
|
+
if (config.extractAttachments || config.ocr) {
|
|
504
|
+
try {
|
|
505
|
+
const ops = await page.getOperatorList();
|
|
506
|
+
const fnArray = ops.fnArray;
|
|
507
|
+
const argsArray = ops.argsArray;
|
|
508
|
+
for (let j = 0; j < fnArray.length; j++) {
|
|
509
|
+
const fn = fnArray[j];
|
|
510
|
+
if (fn === pdfjs.OPS.dependency) {
|
|
511
|
+
const deps = argsArray[j];
|
|
512
|
+
for (const dep of deps) {
|
|
513
|
+
// In pdfjs-dist v3+, get() throws if not resolved unless a callback is provided.
|
|
514
|
+
// We must use the callback pattern to wait for resolution.
|
|
515
|
+
try {
|
|
516
|
+
if (page.objs.has(dep))
|
|
517
|
+
continue;
|
|
518
|
+
await new Promise((resolve) => {
|
|
519
|
+
const timeout = setTimeout(() => {
|
|
520
|
+
resolve();
|
|
521
|
+
}, 500);
|
|
522
|
+
page.objs.get(dep, (data) => {
|
|
523
|
+
clearTimeout(timeout);
|
|
524
|
+
resolve();
|
|
525
|
+
});
|
|
526
|
+
});
|
|
527
|
+
}
|
|
528
|
+
catch (e) {
|
|
529
|
+
if (config.outputErrorToConsole) {
|
|
530
|
+
console.error(`[PdfParser] Failed to load dependency ${dep}:`, e);
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
if (fn === pdfjs.OPS.paintImageXObject || fn === pdfjs.OPS.paintXObject) {
|
|
536
|
+
const imgName = argsArray[j][0];
|
|
537
|
+
try {
|
|
538
|
+
let hasObj = page.objs.has(imgName);
|
|
539
|
+
let targetObjs = page.objs;
|
|
540
|
+
if (!hasObj && page.commonObjs.has(imgName)) {
|
|
541
|
+
hasObj = true;
|
|
542
|
+
targetObjs = page.commonObjs;
|
|
543
|
+
}
|
|
544
|
+
if (hasObj) {
|
|
545
|
+
// Use callback-based get to ensure safe resolution
|
|
546
|
+
const rawObj = await new Promise((resolve) => {
|
|
547
|
+
// @ts-ignore
|
|
548
|
+
targetObjs.get(imgName, (data) => resolve(data));
|
|
549
|
+
});
|
|
550
|
+
const imgObj = rawObj;
|
|
551
|
+
// Browser-specific: Handle ImageBitmap if data is missing
|
|
552
|
+
if (typeof window !== 'undefined' && !imgObj.data && imgObj.bitmap) {
|
|
553
|
+
try {
|
|
554
|
+
const canvas = document.createElement('canvas');
|
|
555
|
+
canvas.width = imgObj.width;
|
|
556
|
+
canvas.height = imgObj.height;
|
|
557
|
+
const ctx = canvas.getContext('2d');
|
|
558
|
+
if (ctx) {
|
|
559
|
+
ctx.drawImage(imgObj.bitmap, 0, 0);
|
|
560
|
+
imgObj.data = ctx.getImageData(0, 0, imgObj.width, imgObj.height).data;
|
|
561
|
+
imgObj.kind = 3; // RGBA
|
|
562
|
+
}
|
|
563
|
+
}
|
|
564
|
+
catch (e) {
|
|
565
|
+
if (config.outputErrorToConsole)
|
|
566
|
+
console.error(`[PdfParser] Failed to extract from ImageBitmap:`, e);
|
|
567
|
+
}
|
|
568
|
+
}
|
|
569
|
+
if (imgObj?.data && imgObj.width > 0 && imgObj.height > 0) {
|
|
570
|
+
// Find position from transform matrix
|
|
571
|
+
let imgX = 0, imgY = 0;
|
|
572
|
+
for (let k = j - 1; k >= 0; k--) {
|
|
573
|
+
if (fnArray[k] === pdfjs.OPS.transform) {
|
|
574
|
+
imgX = argsArray[k][4];
|
|
575
|
+
imgY = argsArray[k][5];
|
|
576
|
+
break;
|
|
577
|
+
}
|
|
578
|
+
}
|
|
579
|
+
pageItems.push({
|
|
580
|
+
type: 'image',
|
|
581
|
+
x: imgX,
|
|
582
|
+
y: imgY,
|
|
583
|
+
name: imgName,
|
|
584
|
+
data: imgObj.data,
|
|
585
|
+
width: imgObj.width,
|
|
586
|
+
height: imgObj.height,
|
|
587
|
+
kind: imgObj.kind
|
|
588
|
+
});
|
|
589
|
+
}
|
|
590
|
+
}
|
|
591
|
+
}
|
|
592
|
+
catch {
|
|
593
|
+
// Image access failed, continue
|
|
594
|
+
}
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
}
|
|
598
|
+
catch (e) {
|
|
599
|
+
if (config.outputErrorToConsole)
|
|
600
|
+
console.error(`Error extracting images from page ${i}:`, e);
|
|
601
|
+
}
|
|
602
|
+
}
|
|
603
|
+
allPageItems.push(pageItems);
|
|
604
|
+
}
|
|
605
|
+
// Calculate font statistics for heading detection
|
|
606
|
+
const fontStats = calculateFontStats(allPageItems);
|
|
607
|
+
// --- Second Pass: Process pages with font statistics ---
|
|
608
|
+
for (let i = 0; i < allPageItems.length; i++) {
|
|
609
|
+
const pageNum = i + 1;
|
|
610
|
+
let page;
|
|
611
|
+
try {
|
|
612
|
+
page = await pdfDocument.getPage(pageNum);
|
|
613
|
+
}
|
|
614
|
+
catch (e) {
|
|
615
|
+
if (config.outputErrorToConsole)
|
|
616
|
+
console.warn(`[PdfParser] Error loading page ${pageNum} in second pass:`, e);
|
|
617
|
+
continue;
|
|
618
|
+
}
|
|
619
|
+
const pageItems = allPageItems[i];
|
|
620
|
+
const pageContent = [];
|
|
621
|
+
// Extract link annotations for this page
|
|
622
|
+
const annotations = [];
|
|
623
|
+
try {
|
|
624
|
+
const annots = await page.getAnnotations();
|
|
625
|
+
for (const annot of annots) {
|
|
626
|
+
if (annot.subtype === 'Link' && annot.rect) {
|
|
627
|
+
annotations.push({
|
|
628
|
+
rect: annot.rect,
|
|
629
|
+
url: annot.url,
|
|
630
|
+
dest: annot.dest
|
|
631
|
+
});
|
|
632
|
+
}
|
|
633
|
+
}
|
|
634
|
+
}
|
|
635
|
+
catch (e) {
|
|
636
|
+
if (config.outputErrorToConsole)
|
|
637
|
+
console.error(`Error extracting annotations from page ${pageNum}:`, e);
|
|
638
|
+
}
|
|
639
|
+
// Sort items: Y descending (top to bottom), then X ascending (left to right)
|
|
640
|
+
pageItems.sort((a, b) => {
|
|
641
|
+
if (Math.abs(b.y - a.y) > 5)
|
|
642
|
+
return b.y - a.y;
|
|
643
|
+
return a.x - b.x;
|
|
644
|
+
});
|
|
645
|
+
// Process sorted items into content nodes
|
|
646
|
+
let currentNode = null;
|
|
647
|
+
let currentNodeFontSize = 0;
|
|
648
|
+
let lastY = -1;
|
|
649
|
+
let imageCounter = 0;
|
|
650
|
+
for (const item of pageItems) {
|
|
651
|
+
if (item.type === 'text') {
|
|
652
|
+
const text = item.text;
|
|
653
|
+
if (!text)
|
|
654
|
+
continue;
|
|
655
|
+
// Check for new line
|
|
656
|
+
const isNewLine = lastY !== -1 && Math.abs(item.y - lastY) > 5;
|
|
657
|
+
if (isNewLine && currentNode) {
|
|
658
|
+
// Finalize and push current node
|
|
659
|
+
if ((currentNode.text || '').trim().length > 0) {
|
|
660
|
+
pageContent.push(currentNode);
|
|
661
|
+
}
|
|
662
|
+
currentNode = null;
|
|
663
|
+
}
|
|
664
|
+
// Skip pure whitespace at start of lines
|
|
665
|
+
if (!currentNode && text.trim().length === 0) {
|
|
666
|
+
lastY = item.y;
|
|
667
|
+
continue;
|
|
668
|
+
}
|
|
669
|
+
// Determine if this should be a heading
|
|
670
|
+
const headingLevel = detectHeadingLevel(item.height, fontStats);
|
|
671
|
+
if (!currentNode) {
|
|
672
|
+
// Start new node
|
|
673
|
+
if (headingLevel > 0) {
|
|
674
|
+
currentNode = {
|
|
675
|
+
type: 'heading',
|
|
676
|
+
text: '',
|
|
677
|
+
children: [],
|
|
678
|
+
metadata: { level: headingLevel }
|
|
679
|
+
};
|
|
680
|
+
}
|
|
681
|
+
else {
|
|
682
|
+
currentNode = {
|
|
683
|
+
type: 'paragraph',
|
|
684
|
+
text: '',
|
|
685
|
+
children: []
|
|
686
|
+
};
|
|
687
|
+
}
|
|
688
|
+
currentNodeFontSize = item.height;
|
|
689
|
+
}
|
|
690
|
+
// Handle whitespace
|
|
691
|
+
if (text.trim().length === 0) {
|
|
692
|
+
if (currentNode.children && currentNode.children.length > 0) {
|
|
693
|
+
const lastChild = currentNode.children[currentNode.children.length - 1];
|
|
694
|
+
if (lastChild.type === 'text' && lastChild.text) {
|
|
695
|
+
lastChild.text += text;
|
|
696
|
+
currentNode.text += text;
|
|
697
|
+
}
|
|
698
|
+
}
|
|
699
|
+
lastY = item.y;
|
|
700
|
+
continue;
|
|
701
|
+
}
|
|
702
|
+
// Add space between words if needed
|
|
703
|
+
if (currentNode.text && currentNode.text.length > 0 && !currentNode.text.endsWith(' ')) {
|
|
704
|
+
currentNode.text += ' ';
|
|
705
|
+
if (currentNode.children && currentNode.children.length > 0) {
|
|
706
|
+
const lastChild = currentNode.children[currentNode.children.length - 1];
|
|
707
|
+
if (lastChild.type === 'text' && lastChild.text) {
|
|
708
|
+
lastChild.text += ' ';
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
}
|
|
712
|
+
currentNode.text += text;
|
|
713
|
+
// Check for link
|
|
714
|
+
const link = findLinkForText(item, annotations);
|
|
715
|
+
let textMetadata;
|
|
716
|
+
if (link) {
|
|
717
|
+
if (link.url) {
|
|
718
|
+
textMetadata = {
|
|
719
|
+
link: link.url,
|
|
720
|
+
linkType: link.url.startsWith('#') ? 'internal' : 'external'
|
|
721
|
+
};
|
|
722
|
+
}
|
|
723
|
+
else if (link.dest) {
|
|
724
|
+
// Internal destination
|
|
725
|
+
textMetadata = {
|
|
726
|
+
link: typeof link.dest === 'string' ? `#${link.dest}` : '#internal',
|
|
727
|
+
linkType: 'internal'
|
|
728
|
+
};
|
|
729
|
+
}
|
|
730
|
+
}
|
|
731
|
+
// Try to merge with last child if same formatting and no link change
|
|
732
|
+
let merged = false;
|
|
733
|
+
if (currentNode.children && currentNode.children.length > 0 && !textMetadata) {
|
|
734
|
+
const lastChild = currentNode.children[currentNode.children.length - 1];
|
|
735
|
+
if (lastChild.type === 'text' &&
|
|
736
|
+
isSameFormatting(lastChild.formatting, item.formatting) &&
|
|
737
|
+
!lastChild.metadata) {
|
|
738
|
+
lastChild.text = (lastChild.text || '') + text;
|
|
739
|
+
merged = true;
|
|
740
|
+
}
|
|
741
|
+
}
|
|
742
|
+
if (!merged) {
|
|
743
|
+
const textNode = {
|
|
744
|
+
type: 'text',
|
|
745
|
+
text: text,
|
|
746
|
+
formatting: Object.keys(item.formatting).length > 0 ? item.formatting : undefined
|
|
747
|
+
};
|
|
748
|
+
if (textMetadata) {
|
|
749
|
+
textNode.metadata = textMetadata;
|
|
750
|
+
}
|
|
751
|
+
currentNode.children?.push(textNode);
|
|
752
|
+
}
|
|
753
|
+
lastY = item.y;
|
|
754
|
+
}
|
|
755
|
+
else if (item.type === 'image') {
|
|
756
|
+
// Flush current node
|
|
757
|
+
if (currentNode) {
|
|
758
|
+
if ((currentNode.text || '').trim().length > 0) {
|
|
759
|
+
pageContent.push(currentNode);
|
|
760
|
+
}
|
|
761
|
+
currentNode = null;
|
|
762
|
+
}
|
|
763
|
+
imageCounter++;
|
|
764
|
+
// Note: Using .bmp extension since we encode to BMP for broad compatibility
|
|
765
|
+
const attachmentName = `pdf_image_p${pageNum}_${imageCounter}.bmp`;
|
|
766
|
+
/**
|
|
767
|
+
* Image extraction for PDF files.
|
|
768
|
+
*
|
|
769
|
+
* PDF stores images as raw pixel data. We convert to BMP for compatibility.
|
|
770
|
+
*/
|
|
771
|
+
if (config.extractAttachments) {
|
|
772
|
+
try {
|
|
773
|
+
const imageBuffer = convertToRgbaBuffer(item.data, item.width, item.height, item.kind);
|
|
774
|
+
// Encode as BMP
|
|
775
|
+
const bmpBuffer = encodeBmp(item.width, item.height, new Uint8Array(imageBuffer));
|
|
776
|
+
const attachment = (0, imageUtils_1.createAttachment)(attachmentName, bmpBuffer);
|
|
777
|
+
attachment.mimeType = 'image/bmp';
|
|
778
|
+
// Perform OCR if enabled
|
|
779
|
+
if (config.ocr) {
|
|
780
|
+
try {
|
|
781
|
+
// Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
|
|
782
|
+
if (item.width >= 10 && item.height >= 10) {
|
|
783
|
+
attachment.ocrText = (await (0, ocrUtils_1.performOcr)(bmpBuffer, config.ocrLanguage)).trim();
|
|
784
|
+
}
|
|
785
|
+
}
|
|
786
|
+
catch (e) {
|
|
787
|
+
if (config.outputErrorToConsole)
|
|
788
|
+
console.error(`OCR failed for ${attachmentName}:`, e);
|
|
789
|
+
}
|
|
790
|
+
}
|
|
791
|
+
attachments.push(attachment);
|
|
792
|
+
// Create image content node
|
|
793
|
+
const imageMetadata = {
|
|
794
|
+
attachmentName,
|
|
795
|
+
};
|
|
796
|
+
pageContent.push({
|
|
797
|
+
type: 'image',
|
|
798
|
+
text: attachment.ocrText || '',
|
|
799
|
+
metadata: { ...imageMetadata }
|
|
800
|
+
});
|
|
801
|
+
}
|
|
802
|
+
catch (e) {
|
|
803
|
+
(0, errorUtils_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
|
|
804
|
+
}
|
|
805
|
+
}
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
// Flush last node
|
|
809
|
+
if (currentNode && (currentNode.text || '').trim().length > 0) {
|
|
810
|
+
pageContent.push(currentNode);
|
|
811
|
+
}
|
|
812
|
+
// Add page node to content
|
|
813
|
+
content.push({
|
|
814
|
+
type: 'page',
|
|
815
|
+
children: pageContent,
|
|
816
|
+
text: pageContent.map(node => node.text).join(config.newlineDelimiter ?? '\n\n'),
|
|
817
|
+
metadata: { pageNumber: pageNum }
|
|
818
|
+
});
|
|
819
|
+
}
|
|
820
|
+
return {
|
|
821
|
+
type: 'pdf',
|
|
822
|
+
metadata: metadata,
|
|
823
|
+
content: content,
|
|
824
|
+
attachments: attachments,
|
|
825
|
+
toText: () => content.map(c => c.text).join(config.newlineDelimiter ?? '\n\n')
|
|
826
|
+
};
|
|
827
|
+
};
|
|
828
|
+
exports.parsePdf = parsePdf;
|
|
829
|
+
/**
|
|
830
|
+
* Helper to compare two text formatting objects.
|
|
831
|
+
* Returns true if both have the same properties with the same values.
|
|
832
|
+
*/
|
|
833
|
+
function isSameFormatting(a, b) {
|
|
834
|
+
if (!a && !b)
|
|
835
|
+
return true;
|
|
836
|
+
if (!a || !b)
|
|
837
|
+
return false;
|
|
838
|
+
const keysA = Object.keys(a).sort();
|
|
839
|
+
const keysB = Object.keys(b).sort();
|
|
840
|
+
if (keysA.length !== keysB.length)
|
|
841
|
+
return false;
|
|
842
|
+
for (let i = 0; i < keysA.length; i++) {
|
|
843
|
+
const key = keysA[i];
|
|
844
|
+
if (keysA[i] !== keysB[i])
|
|
845
|
+
return false;
|
|
846
|
+
if (a[key] !== b[key])
|
|
847
|
+
return false;
|
|
848
|
+
}
|
|
849
|
+
return true;
|
|
850
|
+
}
|