officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
@@ -56,7 +56,7 @@
56
56
  * @see https://mozilla.github.io/pdf.js/ PDF.js documentation
57
57
  * @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
58
58
  */
59
- import { OfficeParserAST, OfficeParserConfig } from '../types';
59
+ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
60
60
  /**
61
61
  * Parses a PDF file and extracts content.
62
62
  *
@@ -59,53 +59,15 @@
59
59
  */
60
60
  Object.defineProperty(exports, "__esModule", { value: true });
61
61
  exports.parsePdf = void 0;
62
- const errorUtils_1 = require("../utils/errorUtils");
63
- const imageUtils_1 = require("../utils/imageUtils");
64
- const ocrUtils_1 = require("../utils/ocrUtils");
65
- const moduleLoader_1 = require("../utils/moduleLoader");
66
- /**
67
- * Parses PDF creation/modification date strings.
68
- * PDF dates are in format: D:YYYYMMDDHHmmSSOHH'mm'
69
- * Where O is timezone offset direction (+/-), HH is hours, mm is minutes.
70
- *
71
- * @param dateString - The PDF date string
72
- * @returns Parsed Date object or undefined if parsing fails
73
- */
74
- function parsePdfDate(dateString) {
75
- if (!dateString)
76
- return undefined;
77
- try {
78
- // Remove "D:" prefix if present
79
- let str = dateString.startsWith('D:') ? dateString.slice(2) : dateString;
80
- // Extract components: YYYYMMDDHHmmSS
81
- const year = parseInt(str.slice(0, 4), 10);
82
- const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
83
- const day = parseInt(str.slice(6, 8), 10) || 1;
84
- const hour = parseInt(str.slice(8, 10), 10) || 0;
85
- const minute = parseInt(str.slice(10, 12), 10) || 0;
86
- const second = parseInt(str.slice(12, 14), 10) || 0;
87
- // Handle timezone if present
88
- const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
89
- if (tzMatch) {
90
- const tzSign = tzMatch[1] === '-' ? -1 : 1;
91
- const tzHours = parseInt(tzMatch[2], 10) || 0;
92
- const tzMinutes = parseInt(tzMatch[3], 10) || 0;
93
- const offset = tzSign * (tzHours * 60 + tzMinutes);
94
- // Create date in UTC and adjust for timezone
95
- const utc = Date.UTC(year, month, day, hour, minute, second);
96
- return new Date(utc - offset * 60000);
97
- }
98
- return new Date(year, month, day, hour, minute, second);
99
- }
100
- catch {
101
- // Fallback: try native Date parsing
102
- try {
103
- return new Date(dateString);
104
- }
105
- catch {
106
- return undefined;
107
- }
108
- }
62
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
63
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
64
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
65
+ const moduleLoader_js_1 = require("../utils/moduleLoader.js");
66
+ const dateUtils_js_1 = require("../utils/dateUtils.js");
67
+ const envUtils_js_1 = require("../utils/envUtils.js");
68
+ /** Type guard for TextItem in PDF.js 5.x */
69
+ function isTextItem(item) {
70
+ return item && typeof item.str === 'string' && Array.isArray(item.transform) && item.transform.length >= 6;
109
71
  }
110
72
  /**
111
73
  * Calculates statistics about font sizes in the document.
@@ -155,27 +117,21 @@ function detectHeadingLevel(fontSize, fontStats) {
155
117
  return 5; // 1.2x-1.35x = H5
156
118
  return 0; // Below threshold
157
119
  }
158
- /**
159
- * Checks if a text position falls within a link annotation rectangle.
160
- */
161
- /**
162
- * Checks if a text item falls within a link annotation rectangle.
163
- * Uses the center point of the text item to avoid false positives for text
164
- * that starts immediately after a link (which would share the same x coordinate boundary).
165
- */
166
120
  function findLinkForText(item, annotations) {
167
- const centerX = item.x + (item.width / 2);
168
- const centerY = item.y + (item.height / 2);
121
+ const itemMinX = item.x;
122
+ const itemMaxX = item.x + item.width;
123
+ const itemMinY = item.y;
124
+ const itemMaxY = item.y + item.height;
169
125
  for (const annot of annotations) {
170
- // PDF annotation rects are [x1, y1, x2, y2] in page coordinates
171
126
  const [x1, y1, x2, y2] = annot.rect;
172
- const minX = Math.min(x1, x2);
173
- const maxX = Math.max(x1, x2);
174
- const minY = Math.min(y1, y2);
175
- const maxY = Math.max(y1, y2);
176
- // Check if center point is within the rect (strict, no tolerance)
177
- if (centerX >= minX && centerX <= maxX &&
178
- centerY >= minY && centerY <= maxY) {
127
+ const annotMinX = Math.min(x1, x2);
128
+ const annotMaxX = Math.max(x1, x2);
129
+ const annotMinY = Math.min(y1, y2);
130
+ const annotMaxY = Math.max(y1, y2);
131
+ // Check for any intersection between boxes
132
+ const intersects = (itemMinX < annotMaxX && itemMaxX > annotMinX) &&
133
+ (itemMinY < annotMaxY && itemMaxY > annotMinY);
134
+ if (intersects) {
179
135
  return annot;
180
136
  }
181
137
  }
@@ -307,20 +263,20 @@ function convertToRgbaBuffer(data, width, height, kind) {
307
263
  * @returns Promise resolving to the parsed AST
308
264
  */
309
265
  const parsePdf = async (buffer, config) => {
310
- const pdfjs = await (0, moduleLoader_1.loadPdfJs)();
266
+ const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
311
267
  // Configure worker
312
268
  if (config.pdfWorkerSrc) {
313
269
  pdfjs.GlobalWorkerOptions.workerSrc = config.pdfWorkerSrc;
314
270
  }
315
271
  else {
316
272
  // Fallbacks when no workerSrc is provided
317
- // @ts-ignore
318
- if (typeof window !== 'undefined') {
273
+ if (envUtils_js_1.isBrowser) {
319
274
  // Browser: Default to CDN
320
275
  pdfjs.GlobalWorkerOptions.workerSrc = `https://unpkg.com/pdfjs-dist@${pdfjs.version}/build/pdf.worker.min.mjs`;
321
276
  }
322
277
  else {
323
278
  // Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
279
+ (0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
324
280
  try {
325
281
  // We use require.resolve to find the exact path of the installed package.
326
282
  // @ts-ignore - 'require' is available in Node.js/CommonJS environment
@@ -344,8 +300,9 @@ const parsePdf = async (buffer, config) => {
344
300
  pdfDocument = await loadingTask.promise;
345
301
  }
346
302
  catch (e) {
347
- if (e.message?.includes('workerSrc') || e.message?.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
348
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.PDF_WORKER_MISSING, config);
303
+ const message = e instanceof Error ? e.message : String(e);
304
+ if (message.includes('workerSrc') || message.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
305
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
349
306
  }
350
307
  throw e;
351
308
  }
@@ -364,15 +321,50 @@ const parsePdf = async (buffer, config) => {
364
321
  const info = meta.info;
365
322
  const metadata = {
366
323
  pages: numPages,
367
- title: info?.Title || undefined,
368
- author: info?.Author || undefined,
369
- subject: info?.Subject || undefined,
370
- description: info?.Keywords || undefined, // Map Keywords to description as closest match
371
- created: parsePdfDate(info?.CreationDate),
372
- modified: parsePdfDate(info?.ModDate),
324
+ title: info?.Title,
325
+ author: info?.Author,
326
+ subject: info?.Subject,
327
+ description: info?.Keywords, // Map Keywords to description as closest match
328
+ created: (0, dateUtils_js_1.parseOfficeDate)(info?.CreationDate),
329
+ modified: (0, dateUtils_js_1.parseOfficeDate)(info?.ModDate),
373
330
  // Note: lastModifiedBy is not available in PDF format - there's no concept of "last modifier"
374
331
  // The Author field only tracks original author.
375
332
  };
333
+ // Extract non-standard entries from the PDF Info dictionary as custom properties.
334
+ // The standard keys are defined by the PDF spec; anything else is user/tool-defined.
335
+ const standardPdfInfoKeys = new Set([
336
+ 'Title', 'Author', 'Subject', 'Keywords', 'Creator', 'Producer',
337
+ 'CreationDate', 'ModDate', 'Trapped', 'IsAcroFormPresent', 'IsXFAPresent',
338
+ 'IsCollectionPresent', 'IsSignaturesPresent', 'PDFFormatVersion'
339
+ ]);
340
+ if (info) {
341
+ const customProperties = {};
342
+ for (const key of Object.keys(info)) {
343
+ if (standardPdfInfoKeys.has(key))
344
+ continue;
345
+ const val = info[key];
346
+ if (val === null || val === undefined)
347
+ continue;
348
+ // pdf.js groups document-level custom metadata under a 'Custom' object.
349
+ // Flatten its entries directly into customProperties.
350
+ if (key === 'Custom' && typeof val === 'object' && !Array.isArray(val) && !(val instanceof Date)) {
351
+ for (const [customKey, customVal] of Object.entries(val)) {
352
+ if (customVal === null || customVal === undefined)
353
+ continue;
354
+ if (typeof customVal === 'string' || typeof customVal === 'number' || typeof customVal === 'boolean' || customVal instanceof Date) {
355
+ customProperties[customKey] = customVal;
356
+ }
357
+ }
358
+ continue;
359
+ }
360
+ if (typeof val === 'string' || typeof val === 'number' || typeof val === 'boolean' || val instanceof Date) {
361
+ customProperties[key] = val;
362
+ }
363
+ }
364
+ if (Object.keys(customProperties).length > 0) {
365
+ metadata.customProperties = customProperties;
366
+ }
367
+ }
376
368
  // --- Embedded File Attachment Extraction ---
377
369
  /**
378
370
  * PDF can contain embedded files (not images in content, but attached files).
@@ -384,7 +376,7 @@ const parsePdf = async (buffer, config) => {
384
376
  for (const name in embeddedFiles) {
385
377
  const file = embeddedFiles[name];
386
378
  const fileBuffer = Buffer.from(file.content);
387
- const attachment = (0, imageUtils_1.createAttachment)(file.filename, fileBuffer);
379
+ const attachment = (0, imageUtils_js_1.createAttachment)(file.filename, fileBuffer);
388
380
  attachments.push(attachment);
389
381
  }
390
382
  }
@@ -412,23 +404,30 @@ const parsePdf = async (buffer, config) => {
412
404
  }
413
405
  const commonObjs = page.commonObjs;
414
406
  for (const item of textContent.items) {
415
- const transform = item.transform;
407
+ // PDF.js 5.x: textContent.items can contain TextMarkedContent which lack
408
+ // 'str' and 'transform'. Skip these to avoid crashes and page skipping.
409
+ if (!isTextItem(item)) {
410
+ continue;
411
+ }
412
+ // At this point we know the item is a TextItem
413
+ const textItem = item;
414
+ const transform = textItem.transform;
416
415
  const x = transform[4];
417
416
  const y = transform[5];
418
- const width = item.width || 0;
419
- const height = item.height || Math.abs(transform[3]) || 12;
417
+ const width = textItem.width || 0;
418
+ const height = textItem.height || Math.abs(transform[3]) || 12;
420
419
  // Extract formatting from font
421
420
  const formatting = {};
422
421
  let fontName;
423
- if (item.fontName && commonObjs) {
422
+ if (textItem.fontName && commonObjs) {
424
423
  try {
425
- if (commonObjs.has(item.fontName)) {
424
+ if (commonObjs.has(textItem.fontName)) {
426
425
  // Use callback-based get to ensure safe resolution
427
426
  const fontData = await new Promise((resolve) => {
428
- // @ts-ignore
429
- commonObjs.get(item.fontName, (data) => resolve(data));
427
+ // @ts-ignore - commonObjs.get is callback-based in legacy builds
428
+ commonObjs.get(textItem.fontName, (data) => resolve(data));
430
429
  });
431
- if (fontData?.name) {
430
+ if (fontData?.name && typeof fontData.name === 'string') {
432
431
  // Remove PDF subset prefix (6 uppercase letters + '+')
433
432
  fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
434
433
  formatting.font = fontName;
@@ -454,7 +453,7 @@ const parsePdf = async (buffer, config) => {
454
453
  y,
455
454
  width,
456
455
  height,
457
- text: item.str,
456
+ text: textItem.str,
458
457
  fontName,
459
458
  formatting
460
459
  });
@@ -503,11 +502,10 @@ const parsePdf = async (buffer, config) => {
503
502
  }
504
503
  if (hasObj) {
505
504
  // Use callback-based get to ensure safe resolution
506
- const rawObj = await new Promise((resolve) => {
507
- // @ts-ignore
505
+ const imgObj = await new Promise((resolve) => {
506
+ // @ts-ignore - targetObjs.get is callback-based
508
507
  targetObjs.get(imgName, (data) => resolve(data));
509
508
  });
510
- const imgObj = rawObj;
511
509
  // Browser-specific: Handle ImageBitmap if data is missing
512
510
  if (typeof window !== 'undefined' && !imgObj.data && imgObj.bitmap) {
513
511
  try {
@@ -580,13 +578,16 @@ const parsePdf = async (buffer, config) => {
580
578
  const pageContent = [];
581
579
  // Extract link annotations for this page
582
580
  const annotations = [];
581
+ const matchedAnnotations = new Set();
583
582
  try {
584
583
  const annots = await page.getAnnotations();
585
584
  for (const annot of annots) {
586
585
  if (annot.subtype === 'Link' && annot.rect) {
586
+ // PDF.js 5.x compatibility: url might be in 'url', 'unsafeUrl', or 'data.url'
587
+ const url = annot.url || annot.unsafeUrl || annot.data?.url;
587
588
  annotations.push({
588
589
  rect: annot.rect,
589
- url: annot.url,
590
+ url: url,
590
591
  dest: annot.dest
591
592
  });
592
593
  }
@@ -598,7 +599,8 @@ const parsePdf = async (buffer, config) => {
598
599
  }
599
600
  // Sort items: Y descending (top to bottom), then X ascending (left to right)
600
601
  pageItems.sort((a, b) => {
601
- if (Math.abs(b.y - a.y) > 5)
602
+ // Relax tolerance slightly (5 -> 7) for PDF.js 5.x coordinate precision
603
+ if (Math.abs(b.y - a.y) > 7)
602
604
  return b.y - a.y;
603
605
  return a.x - b.x;
604
606
  });
@@ -672,6 +674,8 @@ const parsePdf = async (buffer, config) => {
672
674
  currentNode.text += text;
673
675
  // Check for link
674
676
  const link = findLinkForText(item, annotations);
677
+ if (link)
678
+ matchedAnnotations.add(link);
675
679
  let textMetadata;
676
680
  if (link) {
677
681
  if (link.url) {
@@ -733,14 +737,14 @@ const parsePdf = async (buffer, config) => {
733
737
  const imageBuffer = convertToRgbaBuffer(item.data, item.width, item.height, item.kind);
734
738
  // Encode as BMP
735
739
  const bmpBuffer = encodeBmp(item.width, item.height, new Uint8Array(imageBuffer));
736
- const attachment = (0, imageUtils_1.createAttachment)(attachmentName, bmpBuffer);
740
+ const attachment = (0, imageUtils_js_1.createAttachment)(attachmentName, bmpBuffer);
737
741
  attachment.mimeType = 'image/bmp';
738
742
  // Perform OCR if enabled
739
743
  if (config.ocr) {
740
744
  try {
741
745
  // Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
742
746
  if (item.width >= 10 && item.height >= 10) {
743
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(bmpBuffer, config.ocrLanguage)).trim();
747
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
744
748
  }
745
749
  }
746
750
  catch (e) {
@@ -760,7 +764,7 @@ const parsePdf = async (buffer, config) => {
760
764
  });
761
765
  }
762
766
  catch (e) {
763
- (0, errorUtils_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
767
+ (0, errorUtils_js_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
764
768
  }
765
769
  }
766
770
  }
@@ -21,7 +21,7 @@
21
21
  * @module PowerPointParser
22
22
  * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
23
23
  */
24
- import { OfficeParserAST, OfficeParserConfig } from '../types';
24
+ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
25
25
  /**
26
26
  * Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
27
27
  *