officeparser 6.0.6 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -13
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +43 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +116 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +756 -0
- package/dist/officeparser.browser.iife.js +112 -0
- package/dist/officeparser.browser.mjs +111 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +71 -63
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +131 -114
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +85 -88
- package/dist/parsers/RtfParser.d.ts +1 -1
- package/dist/parsers/RtfParser.js +10 -6
- package/dist/parsers/WordParser.d.ts +1 -1
- package/dist/parsers/WordParser.js +109 -101
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +69 -1
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +39 -18
- package/dist/officeparser.browser.js +0 -153
- package/dist/officeparser.browser.js.map +0 -7
|
@@ -56,7 +56,7 @@
|
|
|
56
56
|
* @see https://mozilla.github.io/pdf.js/ PDF.js documentation
|
|
57
57
|
* @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
|
|
58
58
|
*/
|
|
59
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
59
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
60
60
|
/**
|
|
61
61
|
* Parses a PDF file and extracts content.
|
|
62
62
|
*
|
|
@@ -59,53 +59,15 @@
|
|
|
59
59
|
*/
|
|
60
60
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
61
61
|
exports.parsePdf = void 0;
|
|
62
|
-
const
|
|
63
|
-
const
|
|
64
|
-
const
|
|
65
|
-
const
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
* @param dateString - The PDF date string
|
|
72
|
-
* @returns Parsed Date object or undefined if parsing fails
|
|
73
|
-
*/
|
|
74
|
-
function parsePdfDate(dateString) {
|
|
75
|
-
if (!dateString)
|
|
76
|
-
return undefined;
|
|
77
|
-
try {
|
|
78
|
-
// Remove "D:" prefix if present
|
|
79
|
-
let str = dateString.startsWith('D:') ? dateString.slice(2) : dateString;
|
|
80
|
-
// Extract components: YYYYMMDDHHmmSS
|
|
81
|
-
const year = parseInt(str.slice(0, 4), 10);
|
|
82
|
-
const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
|
|
83
|
-
const day = parseInt(str.slice(6, 8), 10) || 1;
|
|
84
|
-
const hour = parseInt(str.slice(8, 10), 10) || 0;
|
|
85
|
-
const minute = parseInt(str.slice(10, 12), 10) || 0;
|
|
86
|
-
const second = parseInt(str.slice(12, 14), 10) || 0;
|
|
87
|
-
// Handle timezone if present
|
|
88
|
-
const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
|
|
89
|
-
if (tzMatch) {
|
|
90
|
-
const tzSign = tzMatch[1] === '-' ? -1 : 1;
|
|
91
|
-
const tzHours = parseInt(tzMatch[2], 10) || 0;
|
|
92
|
-
const tzMinutes = parseInt(tzMatch[3], 10) || 0;
|
|
93
|
-
const offset = tzSign * (tzHours * 60 + tzMinutes);
|
|
94
|
-
// Create date in UTC and adjust for timezone
|
|
95
|
-
const utc = Date.UTC(year, month, day, hour, minute, second);
|
|
96
|
-
return new Date(utc - offset * 60000);
|
|
97
|
-
}
|
|
98
|
-
return new Date(year, month, day, hour, minute, second);
|
|
99
|
-
}
|
|
100
|
-
catch {
|
|
101
|
-
// Fallback: try native Date parsing
|
|
102
|
-
try {
|
|
103
|
-
return new Date(dateString);
|
|
104
|
-
}
|
|
105
|
-
catch {
|
|
106
|
-
return undefined;
|
|
107
|
-
}
|
|
108
|
-
}
|
|
62
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
63
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
64
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
65
|
+
const moduleLoader_js_1 = require("../utils/moduleLoader.js");
|
|
66
|
+
const dateUtils_js_1 = require("../utils/dateUtils.js");
|
|
67
|
+
const envUtils_js_1 = require("../utils/envUtils.js");
|
|
68
|
+
/** Type guard for TextItem in PDF.js 5.x */
|
|
69
|
+
function isTextItem(item) {
|
|
70
|
+
return item && typeof item.str === 'string' && Array.isArray(item.transform) && item.transform.length >= 6;
|
|
109
71
|
}
|
|
110
72
|
/**
|
|
111
73
|
* Calculates statistics about font sizes in the document.
|
|
@@ -155,27 +117,21 @@ function detectHeadingLevel(fontSize, fontStats) {
|
|
|
155
117
|
return 5; // 1.2x-1.35x = H5
|
|
156
118
|
return 0; // Below threshold
|
|
157
119
|
}
|
|
158
|
-
/**
|
|
159
|
-
* Checks if a text position falls within a link annotation rectangle.
|
|
160
|
-
*/
|
|
161
|
-
/**
|
|
162
|
-
* Checks if a text item falls within a link annotation rectangle.
|
|
163
|
-
* Uses the center point of the text item to avoid false positives for text
|
|
164
|
-
* that starts immediately after a link (which would share the same x coordinate boundary).
|
|
165
|
-
*/
|
|
166
120
|
function findLinkForText(item, annotations) {
|
|
167
|
-
const
|
|
168
|
-
const
|
|
121
|
+
const itemMinX = item.x;
|
|
122
|
+
const itemMaxX = item.x + item.width;
|
|
123
|
+
const itemMinY = item.y;
|
|
124
|
+
const itemMaxY = item.y + item.height;
|
|
169
125
|
for (const annot of annotations) {
|
|
170
|
-
// PDF annotation rects are [x1, y1, x2, y2] in page coordinates
|
|
171
126
|
const [x1, y1, x2, y2] = annot.rect;
|
|
172
|
-
const
|
|
173
|
-
const
|
|
174
|
-
const
|
|
175
|
-
const
|
|
176
|
-
// Check
|
|
177
|
-
|
|
178
|
-
|
|
127
|
+
const annotMinX = Math.min(x1, x2);
|
|
128
|
+
const annotMaxX = Math.max(x1, x2);
|
|
129
|
+
const annotMinY = Math.min(y1, y2);
|
|
130
|
+
const annotMaxY = Math.max(y1, y2);
|
|
131
|
+
// Check for any intersection between boxes
|
|
132
|
+
const intersects = (itemMinX < annotMaxX && itemMaxX > annotMinX) &&
|
|
133
|
+
(itemMinY < annotMaxY && itemMaxY > annotMinY);
|
|
134
|
+
if (intersects) {
|
|
179
135
|
return annot;
|
|
180
136
|
}
|
|
181
137
|
}
|
|
@@ -307,20 +263,20 @@ function convertToRgbaBuffer(data, width, height, kind) {
|
|
|
307
263
|
* @returns Promise resolving to the parsed AST
|
|
308
264
|
*/
|
|
309
265
|
const parsePdf = async (buffer, config) => {
|
|
310
|
-
const pdfjs = await (0,
|
|
266
|
+
const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
|
|
311
267
|
// Configure worker
|
|
312
268
|
if (config.pdfWorkerSrc) {
|
|
313
269
|
pdfjs.GlobalWorkerOptions.workerSrc = config.pdfWorkerSrc;
|
|
314
270
|
}
|
|
315
271
|
else {
|
|
316
272
|
// Fallbacks when no workerSrc is provided
|
|
317
|
-
|
|
318
|
-
if (typeof window !== 'undefined') {
|
|
273
|
+
if (envUtils_js_1.isBrowser) {
|
|
319
274
|
// Browser: Default to CDN
|
|
320
275
|
pdfjs.GlobalWorkerOptions.workerSrc = `https://unpkg.com/pdfjs-dist@${pdfjs.version}/build/pdf.worker.min.mjs`;
|
|
321
276
|
}
|
|
322
277
|
else {
|
|
323
278
|
// Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
|
|
279
|
+
(0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
|
|
324
280
|
try {
|
|
325
281
|
// We use require.resolve to find the exact path of the installed package.
|
|
326
282
|
// @ts-ignore - 'require' is available in Node.js/CommonJS environment
|
|
@@ -344,8 +300,9 @@ const parsePdf = async (buffer, config) => {
|
|
|
344
300
|
pdfDocument = await loadingTask.promise;
|
|
345
301
|
}
|
|
346
302
|
catch (e) {
|
|
347
|
-
|
|
348
|
-
|
|
303
|
+
const message = e instanceof Error ? e.message : String(e);
|
|
304
|
+
if (message.includes('workerSrc') || message.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
|
|
305
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
|
|
349
306
|
}
|
|
350
307
|
throw e;
|
|
351
308
|
}
|
|
@@ -364,15 +321,50 @@ const parsePdf = async (buffer, config) => {
|
|
|
364
321
|
const info = meta.info;
|
|
365
322
|
const metadata = {
|
|
366
323
|
pages: numPages,
|
|
367
|
-
title: info?.Title
|
|
368
|
-
author: info?.Author
|
|
369
|
-
subject: info?.Subject
|
|
370
|
-
description: info?.Keywords
|
|
371
|
-
created:
|
|
372
|
-
modified:
|
|
324
|
+
title: info?.Title,
|
|
325
|
+
author: info?.Author,
|
|
326
|
+
subject: info?.Subject,
|
|
327
|
+
description: info?.Keywords, // Map Keywords to description as closest match
|
|
328
|
+
created: (0, dateUtils_js_1.parseOfficeDate)(info?.CreationDate),
|
|
329
|
+
modified: (0, dateUtils_js_1.parseOfficeDate)(info?.ModDate),
|
|
373
330
|
// Note: lastModifiedBy is not available in PDF format - there's no concept of "last modifier"
|
|
374
331
|
// The Author field only tracks original author.
|
|
375
332
|
};
|
|
333
|
+
// Extract non-standard entries from the PDF Info dictionary as custom properties.
|
|
334
|
+
// The standard keys are defined by the PDF spec; anything else is user/tool-defined.
|
|
335
|
+
const standardPdfInfoKeys = new Set([
|
|
336
|
+
'Title', 'Author', 'Subject', 'Keywords', 'Creator', 'Producer',
|
|
337
|
+
'CreationDate', 'ModDate', 'Trapped', 'IsAcroFormPresent', 'IsXFAPresent',
|
|
338
|
+
'IsCollectionPresent', 'IsSignaturesPresent', 'PDFFormatVersion'
|
|
339
|
+
]);
|
|
340
|
+
if (info) {
|
|
341
|
+
const customProperties = {};
|
|
342
|
+
for (const key of Object.keys(info)) {
|
|
343
|
+
if (standardPdfInfoKeys.has(key))
|
|
344
|
+
continue;
|
|
345
|
+
const val = info[key];
|
|
346
|
+
if (val === null || val === undefined)
|
|
347
|
+
continue;
|
|
348
|
+
// pdf.js groups document-level custom metadata under a 'Custom' object.
|
|
349
|
+
// Flatten its entries directly into customProperties.
|
|
350
|
+
if (key === 'Custom' && typeof val === 'object' && !Array.isArray(val) && !(val instanceof Date)) {
|
|
351
|
+
for (const [customKey, customVal] of Object.entries(val)) {
|
|
352
|
+
if (customVal === null || customVal === undefined)
|
|
353
|
+
continue;
|
|
354
|
+
if (typeof customVal === 'string' || typeof customVal === 'number' || typeof customVal === 'boolean' || customVal instanceof Date) {
|
|
355
|
+
customProperties[customKey] = customVal;
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
continue;
|
|
359
|
+
}
|
|
360
|
+
if (typeof val === 'string' || typeof val === 'number' || typeof val === 'boolean' || val instanceof Date) {
|
|
361
|
+
customProperties[key] = val;
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
if (Object.keys(customProperties).length > 0) {
|
|
365
|
+
metadata.customProperties = customProperties;
|
|
366
|
+
}
|
|
367
|
+
}
|
|
376
368
|
// --- Embedded File Attachment Extraction ---
|
|
377
369
|
/**
|
|
378
370
|
* PDF can contain embedded files (not images in content, but attached files).
|
|
@@ -384,7 +376,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
384
376
|
for (const name in embeddedFiles) {
|
|
385
377
|
const file = embeddedFiles[name];
|
|
386
378
|
const fileBuffer = Buffer.from(file.content);
|
|
387
|
-
const attachment = (0,
|
|
379
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(file.filename, fileBuffer);
|
|
388
380
|
attachments.push(attachment);
|
|
389
381
|
}
|
|
390
382
|
}
|
|
@@ -412,23 +404,30 @@ const parsePdf = async (buffer, config) => {
|
|
|
412
404
|
}
|
|
413
405
|
const commonObjs = page.commonObjs;
|
|
414
406
|
for (const item of textContent.items) {
|
|
415
|
-
|
|
407
|
+
// PDF.js 5.x: textContent.items can contain TextMarkedContent which lack
|
|
408
|
+
// 'str' and 'transform'. Skip these to avoid crashes and page skipping.
|
|
409
|
+
if (!isTextItem(item)) {
|
|
410
|
+
continue;
|
|
411
|
+
}
|
|
412
|
+
// At this point we know the item is a TextItem
|
|
413
|
+
const textItem = item;
|
|
414
|
+
const transform = textItem.transform;
|
|
416
415
|
const x = transform[4];
|
|
417
416
|
const y = transform[5];
|
|
418
|
-
const width =
|
|
419
|
-
const height =
|
|
417
|
+
const width = textItem.width || 0;
|
|
418
|
+
const height = textItem.height || Math.abs(transform[3]) || 12;
|
|
420
419
|
// Extract formatting from font
|
|
421
420
|
const formatting = {};
|
|
422
421
|
let fontName;
|
|
423
|
-
if (
|
|
422
|
+
if (textItem.fontName && commonObjs) {
|
|
424
423
|
try {
|
|
425
|
-
if (commonObjs.has(
|
|
424
|
+
if (commonObjs.has(textItem.fontName)) {
|
|
426
425
|
// Use callback-based get to ensure safe resolution
|
|
427
426
|
const fontData = await new Promise((resolve) => {
|
|
428
|
-
// @ts-ignore
|
|
429
|
-
commonObjs.get(
|
|
427
|
+
// @ts-ignore - commonObjs.get is callback-based in legacy builds
|
|
428
|
+
commonObjs.get(textItem.fontName, (data) => resolve(data));
|
|
430
429
|
});
|
|
431
|
-
if (fontData?.name) {
|
|
430
|
+
if (fontData?.name && typeof fontData.name === 'string') {
|
|
432
431
|
// Remove PDF subset prefix (6 uppercase letters + '+')
|
|
433
432
|
fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
|
|
434
433
|
formatting.font = fontName;
|
|
@@ -454,7 +453,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
454
453
|
y,
|
|
455
454
|
width,
|
|
456
455
|
height,
|
|
457
|
-
text:
|
|
456
|
+
text: textItem.str,
|
|
458
457
|
fontName,
|
|
459
458
|
formatting
|
|
460
459
|
});
|
|
@@ -503,11 +502,10 @@ const parsePdf = async (buffer, config) => {
|
|
|
503
502
|
}
|
|
504
503
|
if (hasObj) {
|
|
505
504
|
// Use callback-based get to ensure safe resolution
|
|
506
|
-
const
|
|
507
|
-
// @ts-ignore
|
|
505
|
+
const imgObj = await new Promise((resolve) => {
|
|
506
|
+
// @ts-ignore - targetObjs.get is callback-based
|
|
508
507
|
targetObjs.get(imgName, (data) => resolve(data));
|
|
509
508
|
});
|
|
510
|
-
const imgObj = rawObj;
|
|
511
509
|
// Browser-specific: Handle ImageBitmap if data is missing
|
|
512
510
|
if (typeof window !== 'undefined' && !imgObj.data && imgObj.bitmap) {
|
|
513
511
|
try {
|
|
@@ -580,13 +578,16 @@ const parsePdf = async (buffer, config) => {
|
|
|
580
578
|
const pageContent = [];
|
|
581
579
|
// Extract link annotations for this page
|
|
582
580
|
const annotations = [];
|
|
581
|
+
const matchedAnnotations = new Set();
|
|
583
582
|
try {
|
|
584
583
|
const annots = await page.getAnnotations();
|
|
585
584
|
for (const annot of annots) {
|
|
586
585
|
if (annot.subtype === 'Link' && annot.rect) {
|
|
586
|
+
// PDF.js 5.x compatibility: url might be in 'url', 'unsafeUrl', or 'data.url'
|
|
587
|
+
const url = annot.url || annot.unsafeUrl || annot.data?.url;
|
|
587
588
|
annotations.push({
|
|
588
589
|
rect: annot.rect,
|
|
589
|
-
url:
|
|
590
|
+
url: url,
|
|
590
591
|
dest: annot.dest
|
|
591
592
|
});
|
|
592
593
|
}
|
|
@@ -598,7 +599,8 @@ const parsePdf = async (buffer, config) => {
|
|
|
598
599
|
}
|
|
599
600
|
// Sort items: Y descending (top to bottom), then X ascending (left to right)
|
|
600
601
|
pageItems.sort((a, b) => {
|
|
601
|
-
|
|
602
|
+
// Relax tolerance slightly (5 -> 7) for PDF.js 5.x coordinate precision
|
|
603
|
+
if (Math.abs(b.y - a.y) > 7)
|
|
602
604
|
return b.y - a.y;
|
|
603
605
|
return a.x - b.x;
|
|
604
606
|
});
|
|
@@ -672,6 +674,8 @@ const parsePdf = async (buffer, config) => {
|
|
|
672
674
|
currentNode.text += text;
|
|
673
675
|
// Check for link
|
|
674
676
|
const link = findLinkForText(item, annotations);
|
|
677
|
+
if (link)
|
|
678
|
+
matchedAnnotations.add(link);
|
|
675
679
|
let textMetadata;
|
|
676
680
|
if (link) {
|
|
677
681
|
if (link.url) {
|
|
@@ -733,14 +737,14 @@ const parsePdf = async (buffer, config) => {
|
|
|
733
737
|
const imageBuffer = convertToRgbaBuffer(item.data, item.width, item.height, item.kind);
|
|
734
738
|
// Encode as BMP
|
|
735
739
|
const bmpBuffer = encodeBmp(item.width, item.height, new Uint8Array(imageBuffer));
|
|
736
|
-
const attachment = (0,
|
|
740
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(attachmentName, bmpBuffer);
|
|
737
741
|
attachment.mimeType = 'image/bmp';
|
|
738
742
|
// Perform OCR if enabled
|
|
739
743
|
if (config.ocr) {
|
|
740
744
|
try {
|
|
741
745
|
// Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
|
|
742
746
|
if (item.width >= 10 && item.height >= 10) {
|
|
743
|
-
attachment.ocrText = (await (0,
|
|
747
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
744
748
|
}
|
|
745
749
|
}
|
|
746
750
|
catch (e) {
|
|
@@ -760,7 +764,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
760
764
|
});
|
|
761
765
|
}
|
|
762
766
|
catch (e) {
|
|
763
|
-
(0,
|
|
767
|
+
(0, errorUtils_js_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
|
|
764
768
|
}
|
|
765
769
|
}
|
|
766
770
|
}
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* @module PowerPointParser
|
|
22
22
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
23
23
|
*/
|
|
24
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
24
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
25
25
|
/**
|
|
26
26
|
* Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
|
|
27
27
|
*
|