officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -27,6 +27,71 @@ const types_js_1 = require("../types.js");
27
27
  const astUtils_js_1 = require("../utils/astUtils.js");
28
28
  const chartUtils_js_1 = require("../utils/chartUtils.js");
29
29
  const errorUtils_js_1 = require("../utils/errorUtils.js");
30
+ /**
31
+ * Tracks how many table cells a single document has been allowed to materialize.
32
+ *
33
+ * ODF encodes runs of identical cells/rows as `table:number-columns-repeated` and
34
+ * `table:number-rows-repeated` rather than repeating markup, so a few hundred bytes of XML can ask
35
+ * for an arbitrary number of nodes - and the two multiply, so a row repeat times a column repeat
36
+ * compounds it. The ZIP limits cannot catch this: the XML is tiny before decompression and the
37
+ * expansion happens afterwards, while building the AST.
38
+ *
39
+ * The budget bounds what gets *materialized*, never the attribute itself. Capping the attribute
40
+ * would break ordinary documents - LibreOffice routinely writes `number-rows-repeated="1048566"`
41
+ * to mean "the rest of the sheet is empty", and those runs are legitimate.
42
+ *
43
+ * Warns once per document rather than per clamp, so a wide sheet doesn't emit thousands of
44
+ * identical warnings.
45
+ */
46
+ class CellBudget {
47
+ limit;
48
+ config;
49
+ remaining;
50
+ warned = false;
51
+ constructor(limit, config) {
52
+ this.limit = limit;
53
+ this.config = config;
54
+ this.remaining = limit;
55
+ }
56
+ /** How many of `wanted` may be created; 0 once exhausted. */
57
+ take(wanted) {
58
+ // `!(wanted > 0)` rather than `wanted <= 0` so a NaN is rejected too: `NaN <= 0` is
59
+ // false, so a garbage repeat attribute (`parseInt("abc")`) would otherwise fall through
60
+ // and drain the entire remaining budget, dropping every legitimate cell that followed.
61
+ if (!(wanted > 0))
62
+ return 0;
63
+ if (this.remaining <= 0) {
64
+ this.warn();
65
+ return 0;
66
+ }
67
+ if (wanted <= this.remaining) {
68
+ this.remaining -= wanted;
69
+ return wanted;
70
+ }
71
+ const granted = this.remaining;
72
+ this.remaining = 0;
73
+ this.warn();
74
+ return granted;
75
+ }
76
+ get exhausted() { return this.remaining <= 0; }
77
+ warn() {
78
+ if (this.warned)
79
+ return;
80
+ this.warned = true;
81
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.TABLE_CELL_LIMIT_EXCEEDED, this.config, this.limit);
82
+ }
83
+ }
84
+ /** Resolves the configured cell budget, falling back to the documented default. */
85
+ const createCellBudget = (config) => new CellBudget(config.decompressionLimits?.maxTableCells ?? 1000000, config);
86
+ /**
87
+ * Coerces a `table:number-*-repeated` attribute to a usable repeat count. A missing, zero,
88
+ * negative or non-numeric value becomes 1 (the element renders once), so a garbage attribute
89
+ * can neither drop a cell nor poison arithmetic downstream (`colIndex += NaN`).
90
+ */
91
+ const toRepeatCount = (attr) => {
92
+ const n = parseInt(attr || "1");
93
+ return Number.isFinite(n) && n > 0 ? n : 1;
94
+ };
30
95
  const imageUtils_js_1 = require("../utils/imageUtils.js");
31
96
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
32
97
  const xmlUtils_js_1 = require("../utils/xmlUtils.js");
@@ -613,21 +678,22 @@ const parseOpenOffice = async (buffer, config) => {
613
678
  * @param config - Parser configuration
614
679
  * @returns Table content node with proper structure
615
680
  */
616
- const parseTable = (tableNode, paraStyleMap, styleMap, config, sourceXml) => {
681
+ const parseTable = (tableNode, paraStyleMap, styleMap, config, sourceXml, cellBudget) => {
617
682
  const rows = [];
618
683
  // Use getDirectChildren to avoid nested table rows
619
684
  const tableRows = (0, xmlUtils_js_1.getDirectChildren)(tableNode, "table:table-row");
620
685
  let rowIndex = 0;
621
686
  for (const row of tableRows) {
687
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
622
688
  const cells = [];
623
689
  // Use getDirectChildren to avoid nested table cells
624
690
  const tableCells = (0, xmlUtils_js_1.getDirectChildren)(row, "table:table-cell");
625
- const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
691
+ const rowsRepeated = toRepeatCount(row.getAttribute("table:number-rows-repeated"));
626
692
  let colIndex = 0;
627
693
  for (const cell of tableCells) {
628
694
  const cellChildren = [];
629
695
  let cellTextRef = { value: '' };
630
- const colsRepeated = parseInt(cell.getAttribute("table:number-columns-repeated") || "1");
696
+ const colsRepeated = toRepeatCount(cell.getAttribute("table:number-columns-repeated"));
631
697
  const colSpan = parseInt(cell.getAttribute("table:number-columns-spanned") || "1");
632
698
  const rowSpan = parseInt(cell.getAttribute("table:number-rows-spanned") || "1");
633
699
  // Helper to recursively process cell children (handles frames, text-boxes, etc. in ODP)
@@ -680,7 +746,7 @@ const parseOpenOffice = async (buffer, config) => {
680
746
  }
681
747
  else if (element.tagName === "table:table") {
682
748
  // Recursive call for nested table
683
- const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config, sourceXml);
749
+ const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config, sourceXml, cellBudget);
684
750
  cellChildren.push(nestedTableNode);
685
751
  }
686
752
  else if (element.tagName === "draw:frame" || element.tagName === "draw:text-box") {
@@ -697,7 +763,14 @@ const parseOpenOffice = async (buffer, config) => {
697
763
  cellText = cellText.slice(0, -1);
698
764
  }
699
765
  // Add cell(s) for repeated columns
700
- for (let k = 0; k < colsRepeated; k++) {
766
+ // Bounded by the document's cell budget, not by the attribute: the repeat count
767
+ // is attacker-influenced and this path materializes a node per iteration.
768
+ const allowedCols = cellBudget.take(colsRepeated);
769
+ for (let k = 0; k < allowedCols; k++) {
770
+ // Repeat expansion is the one place a small document produces a long loop,
771
+ // so it is also the one place a caller most needs to be able to cancel.
772
+ if ((k & 1023) === 0)
773
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
701
774
  // Apply cell background color if defined in styleMap
702
775
  const cellStyleName = cell.getAttribute("table:style-name");
703
776
  const cellBgColor = cellStyleName && styleMap[cellStyleName]?.backgroundColor;
@@ -723,8 +796,15 @@ const parseOpenOffice = async (buffer, config) => {
723
796
  colIndex++;
724
797
  }
725
798
  }
726
- // Add row(s) for repeated rows
727
- for (let k = 0; k < rowsRepeated; k++) {
799
+ // Add row(s) for repeated rows. Every repetition past the first deep-copies the
800
+ // whole cell array, so rows x cols is what actually exhausts memory; charge those
801
+ // copies against the same budget.
802
+ const allowedRows = cells.length === 0
803
+ ? rowsRepeated
804
+ : Math.min(rowsRepeated, 1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
805
+ for (let k = 0; k < allowedRows; k++) {
806
+ if ((k & 255) === 0)
807
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
728
808
  const rowNode = {
729
809
  type: 'row',
730
810
  children: k === 0 ? cells : JSON.parse(JSON.stringify(cells))
@@ -754,6 +834,11 @@ const parseOpenOffice = async (buffer, config) => {
754
834
  const body = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
755
835
  if (!body)
756
836
  return;
837
+ // One budget for the entire document. It has to span every table - spreadsheet sheets,
838
+ // ODT/ODP body tables, and nested tables alike - or a file sidesteps the cap simply by
839
+ // splitting a huge repeat expansion across many small tables. `traverse` and the
840
+ // spreadsheet branch below both close over this; `parseTable` receives it explicitly.
841
+ const cellBudget = createCellBudget(config);
757
842
  // Parse automatic styles (local to content.xml)
758
843
  const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
759
844
  if (automaticStyles) {
@@ -917,7 +1002,7 @@ const parseOpenOffice = async (buffer, config) => {
917
1002
  }
918
1003
  else if (node.tagName === "table:table") {
919
1004
  // Parse table with proper structure
920
- const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
1005
+ const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml, cellBudget);
921
1006
  if (asSheet) {
922
1007
  tableNode.type = 'sheet';
923
1008
  const sheetName = node.getAttribute("table:name");
@@ -1134,7 +1219,7 @@ const parseOpenOffice = async (buffer, config) => {
1134
1219
  traverse(textBox, targetArray, isHeading || forceHeading, sourceXml);
1135
1220
  }
1136
1221
  else if (table) {
1137
- const tableNode = parseTable(table, paragraphStyleMap, styleMap, config, sourceXml);
1222
+ const tableNode = parseTable(table, paragraphStyleMap, styleMap, config, sourceXml, cellBudget);
1138
1223
  if (config.includeRawContent)
1139
1224
  tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, sourceXml, config);
1140
1225
  targetArray.push(tableNode);
@@ -1253,14 +1338,15 @@ const parseOpenOffice = async (buffer, config) => {
1253
1338
  const tableRows = (0, xmlUtils_js_1.getElementsByTagName)(table, "table:table-row");
1254
1339
  let rowIndex = 0;
1255
1340
  for (let r = 0; r < tableRows.length; r++) {
1341
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
1256
1342
  const row = tableRows[r];
1257
1343
  const cells = [];
1258
1344
  const tableCells = (0, xmlUtils_js_1.getElementsByTagName)(row, "table:table-cell");
1259
1345
  let colIndex = 0;
1260
- const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
1346
+ const rowsRepeated = toRepeatCount(row.getAttribute("table:number-rows-repeated"));
1261
1347
  for (let c = 0; c < tableCells.length; c++) {
1262
1348
  const cell = tableCells[c];
1263
- const colsRepeated = parseInt(cell.getAttribute("table:number-columns-repeated") || "1");
1349
+ const colsRepeated = toRepeatCount(cell.getAttribute("table:number-columns-repeated"));
1264
1350
  // Extract text from cell (paragraphs inside cell)
1265
1351
  let cellText = "";
1266
1352
  const children = [];
@@ -1385,12 +1471,26 @@ const parseOpenOffice = async (buffer, config) => {
1385
1471
  children.push(chartNode);
1386
1472
  }
1387
1473
  }
1388
- // Add cell(s)
1389
- for (let k = 0; k < colsRepeated; k++) {
1390
- // For ODS (spreadsheets), we skip empty cells to avoid massive ASTs (millions of cells)
1391
- // but for ODP/ODT (presentation/text), cells are part of a defined table grid
1392
- // Also include cells that have children (e.g., image nodes) even if no text
1393
- if (cellText || children.length > 0 || fileType !== 'ods') {
1474
+ // Add cell(s). The repeat count is attacker-influenced, so the loop
1475
+ // is bounded by the document's remaining cell budget rather than by
1476
+ // the attribute. An empty ODS cell creates nothing, so it costs no
1477
+ // budget - which is what keeps the huge trailing-empty runs real
1478
+ // files carry (number-columns-repeated="16384") free.
1479
+ // For ODS an empty cell materializes nothing, so a huge
1480
+ // number-columns-repeated on a blank cell (the normal way ODF marks a
1481
+ // trailing empty run) is skipped in O(1) by advancing the column index
1482
+ // rather than spinning the loop colsRepeated times for zero output -
1483
+ // that spin was itself a CPU denial-of-service, unbounded by the cell
1484
+ // budget because it created no cells to charge against.
1485
+ const willMaterialize = (cellText || children.length > 0 || fileType !== 'ods');
1486
+ if (!willMaterialize) {
1487
+ colIndex += colsRepeated;
1488
+ }
1489
+ else {
1490
+ const allowedCols = cellBudget.take(colsRepeated);
1491
+ for (let k = 0; k < allowedCols; k++) {
1492
+ if ((k & 1023) === 0)
1493
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
1394
1494
  const cellNode = {
1395
1495
  type: 'cell',
1396
1496
  text: cellText,
@@ -1401,13 +1501,21 @@ const parseOpenOffice = async (buffer, config) => {
1401
1501
  cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, xmlString, config);
1402
1502
  }
1403
1503
  cells.push(cellNode);
1504
+ colIndex++;
1404
1505
  }
1405
- colIndex++;
1406
1506
  }
1407
1507
  }
1408
- // Add row(s)
1508
+ // Add row(s). This is where the two repeats multiply: each repetition
1509
+ // deep-copies the whole cell array, so rows x cols is what actually
1510
+ // exhausts memory. Charge the copies against the same budget.
1409
1511
  if (cells.length > 0) {
1410
- for (let k = 0; k < rowsRepeated; k++) {
1512
+ const allowedRows = Math.min(rowsRepeated,
1513
+ // The first row reuses `cells` rather than copying, so only the
1514
+ // repeats beyond it cost budget.
1515
+ 1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
1516
+ for (let k = 0; k < allowedRows; k++) {
1517
+ if ((k & 255) === 0)
1518
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
1411
1519
  const rowNode = {
1412
1520
  type: 'row',
1413
1521
  children: JSON.parse(JSON.stringify(cells)), // Deep copy for repeated rows
@@ -308,7 +308,10 @@ const parsePdf = async (buffer, config) => {
308
308
  const uint8Array = new Uint8Array(buffer);
309
309
  const loadingTask = pdfjs.getDocument({
310
310
  data: uint8Array,
311
- verbosity: 0 // ERRORS only, suppresses warnings
311
+ verbosity: 0, // ERRORS only, suppresses warnings
312
+ // Harden against untrusted PDFs: don't let pdf.js JIT font/CMap fast-paths
313
+ // compile via `new Function`.
314
+ isEvalSupported: false
312
315
  });
313
316
  // Handle loading errors, specifically missing worker in browser
314
317
  let pdfDocument;
@@ -724,6 +724,7 @@ const parsePowerPoint = async (buffer, config) => {
724
724
  const slideMasters = [];
725
725
  // Now for processing all the other files - slides and notes.
726
726
  for (const file of files) {
727
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
727
728
  if (file.path.match(mediaFileRegex))
728
729
  continue;
729
730
  if (file.path.match(chartFileRegex))
@@ -1080,6 +1080,7 @@ const parseWord = async (buffer, config) => {
1080
1080
  const bodyChildren = Array.from(body.childNodes);
1081
1081
  let pendingAnchorIds = [];
1082
1082
  for (const child of bodyChildren) {
1083
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
1083
1084
  if ((0, xmlUtils_js_1.isElement)(child)) {
1084
1085
  if (child.nodeName === 'w:p') {
1085
1086
  content.push(parseParagraph(child, documentContent, pendingAnchorIds));