officeparser 7.0.3 → 7.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +152 -18
  2. package/dist/OfficeGenerator.d.ts +1 -1
  3. package/dist/OfficeGenerator.js +16 -7
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +4 -0
  6. package/dist/cli.js +12 -3
  7. package/dist/defaults.js +27 -1
  8. package/dist/generators/BaseGenerator.d.ts +3 -3
  9. package/dist/generators/ChunkingGenerator.js +31 -4
  10. package/dist/generators/CsvGenerator.d.ts +1 -1
  11. package/dist/generators/HtmlGenerator.d.ts +2 -1
  12. package/dist/generators/HtmlGenerator.js +462 -40
  13. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  14. package/dist/generators/MarkdownGenerator.js +3 -1
  15. package/dist/generators/PdfGenerator.d.ts +1 -1
  16. package/dist/generators/PdfGenerator.js +51 -10
  17. package/dist/generators/RtfGenerator.d.ts +2 -1
  18. package/dist/generators/RtfGenerator.js +43 -6
  19. package/dist/generators/TextGenerator.d.ts +1 -1
  20. package/dist/officeparser.browser.d.ts +377 -53
  21. package/dist/officeparser.browser.iife.js +380 -93
  22. package/dist/officeparser.browser.mjs +380 -93
  23. package/dist/parsers/CsvParser.js +6 -1
  24. package/dist/parsers/ExcelParser.js +69 -21
  25. package/dist/parsers/HtmlParser.js +15 -1
  26. package/dist/parsers/MarkdownParser.js +18 -10
  27. package/dist/parsers/OpenOfficeParser.js +61 -34
  28. package/dist/parsers/PdfParser.js +26 -1
  29. package/dist/parsers/PowerPointParser.js +168 -40
  30. package/dist/parsers/RtfParser.js +30 -24
  31. package/dist/parsers/WordParser.js +158 -11
  32. package/dist/sbom.cdx.json +100 -100
  33. package/dist/types.d.ts +383 -53
  34. package/dist/types.js +4 -0
  35. package/dist/utils/astUtils.d.ts +2 -2
  36. package/dist/utils/astUtils.js +2 -1
  37. package/dist/utils/configUtils.d.ts +5 -0
  38. package/dist/utils/configUtils.js +69 -2
  39. package/dist/utils/errorUtils.d.ts +20 -0
  40. package/dist/utils/errorUtils.js +39 -3
  41. package/dist/utils/moduleLoader.js +3 -3
  42. package/dist/utils/ocrUtils.js +271 -66
  43. package/dist/utils/xmlUtils.d.ts +17 -0
  44. package/dist/utils/xmlUtils.js +85 -1
  45. package/package.json +3 -2
@@ -2,6 +2,7 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseCsv = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
5
6
  /**
6
7
  * Parses a CSV file and extracts a single sheet with rows and cells.
7
8
  *
@@ -10,6 +11,10 @@ const astUtils_js_1 = require("../utils/astUtils.js");
10
11
  * @returns A promise resolving to the parsed AST
11
12
  */
12
13
  const parseCsv = async (buffer, config) => {
14
+ // Honour cancellation requests before the character-by-character parsing loop starts.
15
+ // CSV has no OCR or async I/O, but very large files can still occupy the thread for a
16
+ // noticeable duration, so short-circuiting on an aborted signal is still worthwhile.
17
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
13
18
  const textStr = buffer.toString('utf-8');
14
19
  const delimiter = config.csvDelimiter;
15
20
  const records = [];
@@ -105,6 +110,6 @@ const parseCsv = async (buffer, config) => {
105
110
  .join(config.newlineDelimiter)
106
111
  .replace(/\n{3,}/g, '\n\n');
107
112
  };
108
- return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, toTextSync);
113
+ return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, undefined, toTextSync);
109
114
  };
110
115
  exports.parseCsv = parseCsv;
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
40
40
  * @returns A promise resolving to the parsed AST
41
41
  */
42
42
  const parseExcel = async (buffer, config) => {
43
+ // Honour cancellation requests immediately — before extracting the ZIP archive.
44
+ // XLSX parsing involves decompressing multiple XML sheets and potentially running OCR
45
+ // on embedded chart images, so short-circuiting here saves significant work.
46
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
43
47
  const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
44
48
  const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
45
49
  const chartsRegex = /xl\/charts\/chart\d+.xml/g;
@@ -47,18 +51,23 @@ const parseExcel = async (buffer, config) => {
47
51
  const mediaFileRegex = /xl\/media\/.*/;
48
52
  const corePropsFileRegex = /docProps\/core\.xml/;
49
53
  const customPropsFileRegex = /docProps\/custom\.xml/;
54
+ const appPropsFileRegex = /docProps\/app\.xml/;
50
55
  const relsRegex = /xl\/worksheets\/_rels\/sheet\d+\.xml\.rels/g;
51
56
  const drawingRelsRegex = /xl\/drawings\/_rels\/drawing\d+\.xml\.rels/g;
57
+ const commentsRegex = /xl\/comments\d+\.xml/g;
52
58
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, (x) => !!x.match(sheetsRegex) ||
53
59
  !!x.match(drawingsRegex) ||
54
60
  !!x.match(chartsRegex) ||
61
+ (!config.ignoreComments && !!x.match(commentsRegex)) ||
55
62
  x === stringsFilePath ||
56
63
  x === 'xl/styles.xml' ||
57
64
  x === 'xl/workbook.xml' ||
58
65
  x === 'xl/_rels/workbook.xml.rels' ||
59
66
  !!x.match(corePropsFileRegex) ||
60
67
  !!x.match(customPropsFileRegex) ||
61
- (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(relsRegex) || !!x.match(drawingRelsRegex))));
68
+ !!x.match(appPropsFileRegex) ||
69
+ (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
70
+ ((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)));
62
71
  const sharedStringsFile = files.find(f => f.path === stringsFilePath);
63
72
  // Updated to store structured content (rich text runs) or simple string
64
73
  const sharedStrings = [];
@@ -411,6 +420,52 @@ const parseExcel = async (buffer, config) => {
411
420
  if (config.includeRawContent) {
412
421
  rawContents.push(file.content.toString());
413
422
  }
423
+ const sheetFilename = file.path.split('/').pop() || '';
424
+ const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
425
+ const relsFile = files.find(f => f.path === relsFilename);
426
+ const drawingMap = {}; // rId -> drawingPath
427
+ const sheetCommentsMap = {};
428
+ if (relsFile) {
429
+ const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
430
+ const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
431
+ for (const rel of relationships) {
432
+ const id = rel.getAttribute("Id");
433
+ const target = rel.getAttribute("Target");
434
+ const type = rel.getAttribute("Type");
435
+ if (id && target && type) {
436
+ if (config.extractAttachments && type.includes('drawing')) {
437
+ drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
438
+ }
439
+ else if (!config.ignoreComments && type.includes('comments')) {
440
+ const commentsPath = 'xl/' + target.replace('../', '');
441
+ const cFile = files.find(f => f.path === commentsPath);
442
+ if (cFile) {
443
+ const cXml = (0, xmlUtils_js_1.parseXmlString)(cFile.content.toString());
444
+ const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "comment");
445
+ const authorsList = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "author");
446
+ const authors = authorsList.map(a => a.textContent || '');
447
+ for (const cNode of commentNodes) {
448
+ const ref = cNode.getAttribute("ref");
449
+ const authorId = cNode.getAttribute("authorId");
450
+ const author = authorId !== null ? authors[parseInt(authorId)] : undefined;
451
+ const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(cNode, "t");
452
+ const text = tNodes.map(t => t.textContent || '').join('');
453
+ if (ref && text) {
454
+ if (!sheetCommentsMap[ref])
455
+ sheetCommentsMap[ref] = [];
456
+ sheetCommentsMap[ref].push({
457
+ type: 'comment',
458
+ text: text,
459
+ children: [{ type: 'text', text: text, formatting: {} }],
460
+ metadata: author ? { author } : undefined
461
+ });
462
+ }
463
+ }
464
+ }
465
+ }
466
+ }
467
+ }
468
+ }
414
469
  const rows = [];
415
470
  const sheetXml = file.content.toString();
416
471
  // regex to match <row> elements, capturing:
@@ -456,7 +511,7 @@ const parseExcel = async (buffer, config) => {
456
511
  const typeMatch = cAttrs.match(/t="([a-zA-Z]+)"/);
457
512
  const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
458
513
  const vMatch = cContent.match(/<v>([\s\S]*?)<\/v>/);
459
- const tMatch = cContent.match(/<t>([\s\S]*?)<\/t>/);
514
+ const tMatch = cContent.match(/<t\b[^>]*>([\s\S]*?)<\/t>/);
460
515
  let text = '';
461
516
  let cellNodes = [];
462
517
  if (type === 's' && vMatch) {
@@ -473,7 +528,7 @@ const parseExcel = async (buffer, config) => {
473
528
  }
474
529
  }
475
530
  else if (type === 'inlineStr' && tMatch) {
476
- text = tMatch[1].trim();
531
+ text = (0, xmlUtils_js_1.decodeXmlEntities)(tMatch[1].trim());
477
532
  }
478
533
  else if (vMatch) {
479
534
  text = vMatch[1].trim();
@@ -481,7 +536,9 @@ const parseExcel = async (buffer, config) => {
481
536
  // Parse cell coordinate
482
537
  const coordMatch = cAttrs.match(/r="([A-Z]+)(\d+)"/);
483
538
  let colIndex;
539
+ let ref;
484
540
  if (coordMatch) {
541
+ ref = coordMatch[1] + coordMatch[2];
485
542
  colIndex = colToNumber(coordMatch[1]);
486
543
  // If row index is missing in cell coord (unlikely but possible), use rowIndex
487
544
  }
@@ -521,10 +578,12 @@ const parseExcel = async (buffer, config) => {
521
578
  formatting: cellFormatting
522
579
  });
523
580
  }
581
+ const commentsNodeList = (ref && sheetCommentsMap[ref]) ? sheetCommentsMap[ref] : undefined;
524
582
  const cellNode = {
525
583
  type: 'cell',
526
584
  text: text,
527
585
  children: cellNodes,
586
+ comments: commentsNodeList,
528
587
  metadata: { row: rowIndex, col: colIndex }
529
588
  };
530
589
  if (config.includeRawContent) {
@@ -547,23 +606,6 @@ const parseExcel = async (buffer, config) => {
547
606
  }
548
607
  // Handle Drawings in Sheet (images and charts)
549
608
  if (config.extractAttachments) {
550
- // Parse Sheet Rels to map drawing rIds
551
- const sheetFilename = file.path.split('/').pop() || '';
552
- const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
553
- const relsFile = files.find(f => f.path === relsFilename);
554
- const drawingMap = {}; // rId -> drawingPath
555
- if (relsFile) {
556
- const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
557
- const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
558
- for (const rel of relationships) {
559
- const id = rel.getAttribute("Id");
560
- const target = rel.getAttribute("Target");
561
- const type = rel.getAttribute("Type");
562
- if (id && target && type && type.includes('drawing')) {
563
- drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
564
- }
565
- }
566
- }
567
609
  const drawingMatches = file.content.toString().match(/<drawing r:id="(.*?)"/g);
568
610
  if (drawingMatches) {
569
611
  for (const match of drawingMatches) {
@@ -633,6 +675,12 @@ const parseExcel = async (buffer, config) => {
633
675
  if (Object.keys(customProperties).length > 0)
634
676
  metadata.customProperties = customProperties;
635
677
  }
678
+ const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
679
+ if (appPropsFile) {
680
+ const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
681
+ if (Object.keys(appProperties).length > 0)
682
+ metadata.nativeProperties = appProperties;
683
+ }
636
684
  // Link OCR text and chart data to content nodes (like PPTX parser)
637
685
  const assignAttachmentData = (nodes) => {
638
686
  for (const node of nodes) {
@@ -677,6 +725,6 @@ const parseExcel = async (buffer, config) => {
677
725
  };
678
726
  return getText(c);
679
727
  }).filter(t => t != '').join(config.newlineDelimiter);
680
- return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, toTextSync);
728
+ return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, undefined, toTextSync);
681
729
  };
682
730
  exports.parseExcel = parseExcel;
@@ -2,6 +2,7 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseHtml = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
5
6
  const parseAttributes = (attrString) => {
6
7
  const attrs = {};
7
8
  const regex = /([a-zA-Z0-9\-:]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
@@ -95,6 +96,10 @@ const parseHtmlTree = (html) => {
95
96
  return root;
96
97
  };
97
98
  const parseHtml = async (buffer, config) => {
99
+ // Honour cancellation requests before the HTML tree is built and traversed.
100
+ // The custom recursive HTML parser can be expensive for large documents;
101
+ // rejecting early here prevents both the parsing and the subsequent AST construction.
102
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
98
103
  const textStr = buffer.toString('utf-8');
99
104
  const root = parseHtmlTree(textStr);
100
105
  // Find head and body
@@ -122,6 +127,15 @@ const parseHtml = async (buffer, config) => {
122
127
  if (titleNode && titleNode.children.length > 0 && titleNode.children[0].text) {
123
128
  metadata.title = titleNode.children[0].text;
124
129
  }
130
+ metadata.nativeProperties = {};
131
+ for (const child of head.children) {
132
+ if (child.tagName === 'meta') {
133
+ const name = child.attributes?.name || child.attributes?.property || child.attributes?.['http-equiv'];
134
+ if (name) {
135
+ metadata.nativeProperties[name] = child.attributes?.content || '';
136
+ }
137
+ }
138
+ }
125
139
  const extractMeta = (name) => {
126
140
  for (const child of head.children) {
127
141
  if (child.tagName === 'meta' && (child.attributes?.name === name || child.attributes?.property === name)) {
@@ -534,6 +548,6 @@ const parseHtml = async (buffer, config) => {
534
548
  return getText(n);
535
549
  }).join(config.newlineDelimiter)
536
550
  .replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
537
- return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, toTextSync);
551
+ return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, undefined, toTextSync);
538
552
  };
539
553
  exports.parseHtml = parseHtml;
@@ -2,7 +2,12 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseMarkdown = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
5
6
  const parseMarkdown = async (buffer, config) => {
7
+ // Honour cancellation requests before the line-by-line Markdown scanning loop begins.
8
+ // Markdown parsing is entirely synchronous and CPU-bound, so failing fast avoids
9
+ // processing content whose result will be discarded anyway.
10
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
6
11
  let textStr = buffer.toString('utf-8');
7
12
  textStr = textStr.replace(/\r\n/g, '\n');
8
13
  const content = [];
@@ -16,11 +21,20 @@ const parseMarkdown = async (buffer, config) => {
16
21
  textStr = textStr.substring(endIdx + 5);
17
22
  const lines = frontMatter.split('\n');
18
23
  const customProps = {};
24
+ const nativeProps = {};
19
25
  for (const line of lines) {
20
26
  const match = line.match(/^([^:]+):\s*(.*)$/);
21
27
  if (match) {
22
28
  const key = match[1].trim();
23
29
  let val = match[2].trim().replace(/^"(.*)"$/, '$1');
30
+ let parsedVal = val;
31
+ if (val === 'true')
32
+ parsedVal = true;
33
+ else if (val === 'false')
34
+ parsedVal = false;
35
+ else if (!isNaN(Number(val)) && val !== '')
36
+ parsedVal = Number(val);
37
+ nativeProps[key] = parsedVal;
24
38
  if (key === 'title')
25
39
  metadata.title = val;
26
40
  else if (key === 'author')
@@ -32,20 +46,14 @@ const parseMarkdown = async (buffer, config) => {
32
46
  else if (key === 'description')
33
47
  metadata.description = val;
34
48
  else {
35
- // Try to infer type for custom props
36
- if (val === 'true')
37
- customProps[key] = true;
38
- else if (val === 'false')
39
- customProps[key] = false;
40
- else if (!isNaN(Number(val)) && val !== '')
41
- customProps[key] = Number(val);
42
- else
43
- customProps[key] = val;
49
+ customProps[key] = parsedVal;
44
50
  }
45
51
  }
46
52
  }
47
53
  if (Object.keys(customProps).length > 0)
48
54
  metadata.customProperties = customProps;
55
+ if (Object.keys(nativeProps).length > 0)
56
+ metadata.nativeProperties = nativeProps;
49
57
  }
50
58
  }
51
59
  // Extract code blocks first to protect their contents
@@ -355,6 +363,6 @@ const parseMarkdown = async (buffer, config) => {
355
363
  return getText(n);
356
364
  }).join(config.newlineDelimiter)
357
365
  .replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
358
- return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, toTextSync);
366
+ return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, undefined, toTextSync);
359
367
  };
360
368
  exports.parseMarkdown = parseMarkdown;
@@ -39,6 +39,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
39
39
  * @returns A promise resolving to the parsed AST
40
40
  */
41
41
  const parseOpenOffice = async (buffer, config) => {
42
+ // Honour cancellation requests immediately — before extracting the ZIP archive.
43
+ // ODF containers (ODT/ODS/ODP) bundle content.xml, styles.xml, and media files;
44
+ // aborting early avoids needlessly inflating and parsing all of those resources.
45
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
42
46
  const contentFileRegex = /content\.xml/;
43
47
  const objectContentFileRegex = /Object \d+\/content\.xml/;
44
48
  const mediaFileRegex = /(Pictures|media)\/.*/;
@@ -153,9 +157,9 @@ const parseOpenOffice = async (buffer, config) => {
153
157
  if (textPosition.startsWith("super"))
154
158
  formatting.superscript = true;
155
159
  }
156
- if (Object.keys(formatting).length > 0)
157
- styleMap[name] = formatting;
158
160
  }
161
+ if (Object.keys(formatting).length > 0)
162
+ styleMap[name] = formatting;
159
163
  }
160
164
  };
161
165
  if (stylesDom) {
@@ -306,11 +310,17 @@ const parseOpenOffice = async (buffer, config) => {
306
310
  noteId: noteId
307
311
  }
308
312
  };
309
- if (config.putNotesAtLast) {
310
- notes.push(noteNode);
313
+ if (children.length > 0 && children[children.length - 1].type === 'text') {
314
+ const precedingNode = children[children.length - 1];
315
+ if (!precedingNode.notes) {
316
+ precedingNode.notes = [];
317
+ }
318
+ precedingNode.notes.push(noteNode);
311
319
  }
312
320
  else {
313
- children.push(noteNode);
321
+ const emptyTextNode = { type: 'text', text: '' };
322
+ emptyTextNode.notes = [noteNode];
323
+ children.push(emptyTextNode);
314
324
  }
315
325
  }
316
326
  }
@@ -488,22 +498,33 @@ const parseOpenOffice = async (buffer, config) => {
488
498
  const element = child;
489
499
  if (element.tagName === "text:p" || element.tagName === "text:h") {
490
500
  const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config, sourceXml);
491
- const pNode = {
492
- type: element.tagName === "text:h" ? 'heading' : 'paragraph',
493
- text: pContent.text,
494
- children: pContent.children,
495
- metadata: {
496
- ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
497
- ...(pContent.style ? { style: pContent.style } : {})
498
- }
499
- };
501
+ let pNode;
502
+ if (element.tagName === "text:h") {
503
+ pNode = {
504
+ type: 'heading',
505
+ text: pContent.text,
506
+ children: pContent.children,
507
+ metadata: {
508
+ level: parseInt(element.getAttribute("text:outline-level") || "1"),
509
+ ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
510
+ ...(pContent.style ? { style: pContent.style } : {})
511
+ }
512
+ };
513
+ }
514
+ else {
515
+ pNode = {
516
+ type: 'paragraph',
517
+ text: pContent.text,
518
+ children: pContent.children,
519
+ metadata: {
520
+ ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
521
+ ...(pContent.style ? { style: pContent.style } : {})
522
+ }
523
+ };
524
+ }
500
525
  // Clean up metadata if empty
501
- if (Object.keys(pNode.metadata || {}).length === 0)
526
+ if (pNode.type === 'paragraph' && Object.keys(pNode.metadata || {}).length === 0) {
502
527
  delete pNode.metadata;
503
- if (element.tagName === "text:h") {
504
- if (!pNode.metadata)
505
- pNode.metadata = {};
506
- pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
507
528
  }
508
529
  if (config.includeRawContent) {
509
530
  pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
@@ -535,11 +556,18 @@ const parseOpenOffice = async (buffer, config) => {
535
556
  }
536
557
  // Add cell(s) for repeated columns
537
558
  for (let k = 0; k < colsRepeated; k++) {
559
+ // Apply cell background color if defined in styleMap
560
+ const cellStyleName = cell.getAttribute("table:style-name");
561
+ const cellBgColor = cellStyleName && styleMap[cellStyleName]?.backgroundColor;
538
562
  const cellNode = {
539
563
  type: 'cell',
540
564
  text: cellText,
541
565
  children: cellChildren.length > 0 ? (k === 0 ? cellChildren : JSON.parse(JSON.stringify(cellChildren))) : [],
542
- metadata: { row: rowIndex, col: colIndex }
566
+ metadata: {
567
+ row: rowIndex,
568
+ col: colIndex,
569
+ ...(cellBgColor ? { backgroundColor: cellBgColor } : {})
570
+ }
543
571
  };
544
572
  const cellMetadata = cellNode.metadata;
545
573
  if (colSpan > 1)
@@ -617,9 +645,15 @@ const parseOpenOffice = async (buffer, config) => {
617
645
  if (Object.keys(styleInfo).length > 0) {
618
646
  paragraphStyleMap[name] = styleInfo;
619
647
  }
648
+ const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
649
+ const formatting = {};
650
+ if (cellProps) {
651
+ const bgColor = cellProps.getAttribute("fo:background-color");
652
+ if (bgColor && bgColor !== 'transparent')
653
+ formatting.backgroundColor = bgColor;
654
+ }
620
655
  const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
621
656
  if (textProps) {
622
- const formatting = {};
623
657
  if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
624
658
  formatting.bold = true;
625
659
  if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
@@ -650,9 +684,9 @@ const parseOpenOffice = async (buffer, config) => {
650
684
  if (textPosition.startsWith("super"))
651
685
  formatting.superscript = true;
652
686
  }
653
- if (Object.keys(formatting).length > 0)
654
- styleMap[name] = formatting;
655
687
  }
688
+ if (Object.keys(formatting).length > 0)
689
+ styleMap[name] = formatting;
656
690
  }
657
691
  }
658
692
  // Start traversal
@@ -1267,12 +1301,9 @@ const parseOpenOffice = async (buffer, config) => {
1267
1301
  }
1268
1302
  content.push(slideNode);
1269
1303
  if (noteNode && noteNode.children && noteNode.children.length > 0) {
1270
- if (config.putNotesAtLast) {
1271
- odpNotes.push(noteNode);
1272
- }
1273
- else {
1274
- content.push(noteNode);
1275
- }
1304
+ if (!slideNode.notes)
1305
+ slideNode.notes = [];
1306
+ slideNode.notes.push(noteNode);
1276
1307
  }
1277
1308
  }
1278
1309
  if (odpNotes.length > 0) {
@@ -1495,10 +1526,6 @@ const parseOpenOffice = async (buffer, config) => {
1495
1526
  };
1496
1527
  }
1497
1528
  }
1498
- // Append notes to content if configured
1499
- if (config.putNotesAtLast && notes.length > 0) {
1500
- content.push(...notes);
1501
- }
1502
1529
  const toTextSync = () => content.map(c => {
1503
1530
  const getText = (node) => {
1504
1531
  let t = '';
@@ -1520,6 +1547,6 @@ const parseOpenOffice = async (buffer, config) => {
1520
1547
  return (0, astUtils_js_1.createAST)(fileType, {
1521
1548
  ...metadata,
1522
1549
  styleMap: combinedStyleMap
1523
- }, content, attachments, config, toTextSync);
1550
+ }, content, attachments, config, undefined, toTextSync);
1524
1551
  };
1525
1552
  exports.parseOpenOffice = parseOpenOffice;
@@ -266,6 +266,7 @@ function convertToRgbaBuffer(data, width, height, kind) {
266
266
  * @returns Promise resolving to the parsed AST
267
267
  */
268
268
  const parsePdf = async (buffer, config) => {
269
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
269
270
  const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
270
271
  // Configure worker
271
272
  const workerSrc = config.pdfWorkerSrc;
@@ -350,6 +351,17 @@ const parsePdf = async (buffer, config) => {
350
351
  'IsCollectionPresent', 'IsSignaturesPresent', 'PDFFormatVersion'
351
352
  ]);
352
353
  if (info) {
354
+ metadata.nativeProperties = {};
355
+ for (const [key, val] of Object.entries(info)) {
356
+ if (key === 'Custom' && typeof val === 'object' && !Array.isArray(val) && !(val instanceof Date) && val !== null) {
357
+ for (const [customKey, customVal] of Object.entries(val)) {
358
+ metadata.nativeProperties[customKey] = customVal;
359
+ }
360
+ }
361
+ else {
362
+ metadata.nativeProperties[key] = val;
363
+ }
364
+ }
353
365
  const customProperties = {};
354
366
  for (const key of Object.keys(info)) {
355
367
  if (standardPdfInfoKeys.has(key))
@@ -377,6 +389,17 @@ const parsePdf = async (buffer, config) => {
377
389
  metadata.customProperties = customProperties;
378
390
  }
379
391
  }
392
+ if (meta.metadata) {
393
+ if (!metadata.nativeProperties)
394
+ metadata.nativeProperties = {};
395
+ const xmp = meta.metadata;
396
+ if (typeof xmp.getAll === 'function') {
397
+ metadata.nativeProperties['XMP'] = xmp.getAll();
398
+ }
399
+ else {
400
+ metadata.nativeProperties['XMP'] = xmp;
401
+ }
402
+ }
380
403
  // --- Embedded File Attachment Extraction ---
381
404
  /**
382
405
  * PDF can contain embedded files (not images in content, but attached files).
@@ -398,6 +421,7 @@ const parsePdf = async (buffer, config) => {
398
421
  }
399
422
  // --- First Pass: Collect all items for font statistics ---
400
423
  for (let i = 1; i <= numPages; i++) {
424
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
401
425
  let page;
402
426
  let textContent;
403
427
  const pageItems = [];
@@ -575,6 +599,7 @@ const parsePdf = async (buffer, config) => {
575
599
  const fontStats = calculateFontStats(allPageItems);
576
600
  // --- Second Pass: Process pages with font statistics ---
577
601
  for (let i = 0; i < allPageItems.length; i++) {
602
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
578
603
  const pageNum = i + 1;
579
604
  let page;
580
605
  try {
@@ -790,7 +815,7 @@ const parsePdf = async (buffer, config) => {
790
815
  });
791
816
  }
792
817
  const toTextSync = () => content.map(c => c.text).join(config.newlineDelimiter);
793
- return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, toTextSync);
818
+ return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, undefined, toTextSync);
794
819
  };
795
820
  exports.parsePdf = parsePdf;
796
821
  /**