officeparser 7.2.3 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +277 -17
- package/dist/OfficeConverter.d.ts +1 -1
- package/dist/OfficeConverter.js +3 -0
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +5 -2
- package/dist/defaults.js +12 -0
- package/dist/generators/BaseGenerator.d.ts +34 -1
- package/dist/generators/BaseGenerator.js +98 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +28 -16
- package/dist/generators/EpubGenerator.d.ts +43 -0
- package/dist/generators/EpubGenerator.js +312 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +378 -61
- package/dist/generators/MarkdownGenerator.d.ts +28 -5
- package/dist/generators/MarkdownGenerator.js +432 -51
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +47 -22
- package/dist/generators/TextGenerator.js +98 -11
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +427 -20
- package/dist/officeparser.browser.iife.js +338 -206
- package/dist/officeparser.browser.mjs +346 -214
- package/dist/officeparser.browser.slim.d.ts +427 -20
- package/dist/officeparser.browser.slim.iife.js +346 -214
- package/dist/officeparser.browser.slim.mjs +346 -214
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/ExcelParser.js +2 -0
- package/dist/parsers/HtmlParser.js +507 -48
- package/dist/parsers/MarkdownParser.js +704 -92
- package/dist/parsers/OpenOfficeParser.js +128 -20
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/PowerPointParser.js +1 -0
- package/dist/parsers/WordParser.js +1 -0
- package/dist/sbom.cdx.json +1695 -0
- package/dist/types.d.ts +427 -20
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +53 -4
- package/dist/utils/errorUtils.js +7 -3
- package/dist/utils/sanitize.d.ts +139 -0
- package/dist/utils/sanitize.js +318 -0
- package/dist/utils/xmlUtils.js +2 -2
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +16 -12
|
@@ -27,6 +27,71 @@ const types_js_1 = require("../types.js");
|
|
|
27
27
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
28
28
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
29
29
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
30
|
+
/**
|
|
31
|
+
* Tracks how many table cells a single document has been allowed to materialize.
|
|
32
|
+
*
|
|
33
|
+
* ODF encodes runs of identical cells/rows as `table:number-columns-repeated` and
|
|
34
|
+
* `table:number-rows-repeated` rather than repeating markup, so a few hundred bytes of XML can ask
|
|
35
|
+
* for an arbitrary number of nodes - and the two multiply, so a row repeat times a column repeat
|
|
36
|
+
* compounds it. The ZIP limits cannot catch this: the XML is tiny before decompression and the
|
|
37
|
+
* expansion happens afterwards, while building the AST.
|
|
38
|
+
*
|
|
39
|
+
* The budget bounds what gets *materialized*, never the attribute itself. Capping the attribute
|
|
40
|
+
* would break ordinary documents - LibreOffice routinely writes `number-rows-repeated="1048566"`
|
|
41
|
+
* to mean "the rest of the sheet is empty", and those runs are legitimate.
|
|
42
|
+
*
|
|
43
|
+
* Warns once per document rather than per clamp, so a wide sheet doesn't emit thousands of
|
|
44
|
+
* identical warnings.
|
|
45
|
+
*/
|
|
46
|
+
class CellBudget {
|
|
47
|
+
limit;
|
|
48
|
+
config;
|
|
49
|
+
remaining;
|
|
50
|
+
warned = false;
|
|
51
|
+
constructor(limit, config) {
|
|
52
|
+
this.limit = limit;
|
|
53
|
+
this.config = config;
|
|
54
|
+
this.remaining = limit;
|
|
55
|
+
}
|
|
56
|
+
/** How many of `wanted` may be created; 0 once exhausted. */
|
|
57
|
+
take(wanted) {
|
|
58
|
+
// `!(wanted > 0)` rather than `wanted <= 0` so a NaN is rejected too: `NaN <= 0` is
|
|
59
|
+
// false, so a garbage repeat attribute (`parseInt("abc")`) would otherwise fall through
|
|
60
|
+
// and drain the entire remaining budget, dropping every legitimate cell that followed.
|
|
61
|
+
if (!(wanted > 0))
|
|
62
|
+
return 0;
|
|
63
|
+
if (this.remaining <= 0) {
|
|
64
|
+
this.warn();
|
|
65
|
+
return 0;
|
|
66
|
+
}
|
|
67
|
+
if (wanted <= this.remaining) {
|
|
68
|
+
this.remaining -= wanted;
|
|
69
|
+
return wanted;
|
|
70
|
+
}
|
|
71
|
+
const granted = this.remaining;
|
|
72
|
+
this.remaining = 0;
|
|
73
|
+
this.warn();
|
|
74
|
+
return granted;
|
|
75
|
+
}
|
|
76
|
+
get exhausted() { return this.remaining <= 0; }
|
|
77
|
+
warn() {
|
|
78
|
+
if (this.warned)
|
|
79
|
+
return;
|
|
80
|
+
this.warned = true;
|
|
81
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.TABLE_CELL_LIMIT_EXCEEDED, this.config, this.limit);
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
/** Resolves the configured cell budget, falling back to the documented default. */
|
|
85
|
+
const createCellBudget = (config) => new CellBudget(config.decompressionLimits?.maxTableCells ?? 1000000, config);
|
|
86
|
+
/**
|
|
87
|
+
* Coerces a `table:number-*-repeated` attribute to a usable repeat count. A missing, zero,
|
|
88
|
+
* negative or non-numeric value becomes 1 (the element renders once), so a garbage attribute
|
|
89
|
+
* can neither drop a cell nor poison arithmetic downstream (`colIndex += NaN`).
|
|
90
|
+
*/
|
|
91
|
+
const toRepeatCount = (attr) => {
|
|
92
|
+
const n = parseInt(attr || "1");
|
|
93
|
+
return Number.isFinite(n) && n > 0 ? n : 1;
|
|
94
|
+
};
|
|
30
95
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
31
96
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
32
97
|
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
@@ -613,21 +678,22 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
613
678
|
* @param config - Parser configuration
|
|
614
679
|
* @returns Table content node with proper structure
|
|
615
680
|
*/
|
|
616
|
-
const parseTable = (tableNode, paraStyleMap, styleMap, config, sourceXml) => {
|
|
681
|
+
const parseTable = (tableNode, paraStyleMap, styleMap, config, sourceXml, cellBudget) => {
|
|
617
682
|
const rows = [];
|
|
618
683
|
// Use getDirectChildren to avoid nested table rows
|
|
619
684
|
const tableRows = (0, xmlUtils_js_1.getDirectChildren)(tableNode, "table:table-row");
|
|
620
685
|
let rowIndex = 0;
|
|
621
686
|
for (const row of tableRows) {
|
|
687
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
622
688
|
const cells = [];
|
|
623
689
|
// Use getDirectChildren to avoid nested table cells
|
|
624
690
|
const tableCells = (0, xmlUtils_js_1.getDirectChildren)(row, "table:table-cell");
|
|
625
|
-
const rowsRepeated =
|
|
691
|
+
const rowsRepeated = toRepeatCount(row.getAttribute("table:number-rows-repeated"));
|
|
626
692
|
let colIndex = 0;
|
|
627
693
|
for (const cell of tableCells) {
|
|
628
694
|
const cellChildren = [];
|
|
629
695
|
let cellTextRef = { value: '' };
|
|
630
|
-
const colsRepeated =
|
|
696
|
+
const colsRepeated = toRepeatCount(cell.getAttribute("table:number-columns-repeated"));
|
|
631
697
|
const colSpan = parseInt(cell.getAttribute("table:number-columns-spanned") || "1");
|
|
632
698
|
const rowSpan = parseInt(cell.getAttribute("table:number-rows-spanned") || "1");
|
|
633
699
|
// Helper to recursively process cell children (handles frames, text-boxes, etc. in ODP)
|
|
@@ -680,7 +746,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
680
746
|
}
|
|
681
747
|
else if (element.tagName === "table:table") {
|
|
682
748
|
// Recursive call for nested table
|
|
683
|
-
const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config, sourceXml);
|
|
749
|
+
const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config, sourceXml, cellBudget);
|
|
684
750
|
cellChildren.push(nestedTableNode);
|
|
685
751
|
}
|
|
686
752
|
else if (element.tagName === "draw:frame" || element.tagName === "draw:text-box") {
|
|
@@ -697,7 +763,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
697
763
|
cellText = cellText.slice(0, -1);
|
|
698
764
|
}
|
|
699
765
|
// Add cell(s) for repeated columns
|
|
700
|
-
|
|
766
|
+
// Bounded by the document's cell budget, not by the attribute: the repeat count
|
|
767
|
+
// is attacker-influenced and this path materializes a node per iteration.
|
|
768
|
+
const allowedCols = cellBudget.take(colsRepeated);
|
|
769
|
+
for (let k = 0; k < allowedCols; k++) {
|
|
770
|
+
// Repeat expansion is the one place a small document produces a long loop,
|
|
771
|
+
// so it is also the one place a caller most needs to be able to cancel.
|
|
772
|
+
if ((k & 1023) === 0)
|
|
773
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
701
774
|
// Apply cell background color if defined in styleMap
|
|
702
775
|
const cellStyleName = cell.getAttribute("table:style-name");
|
|
703
776
|
const cellBgColor = cellStyleName && styleMap[cellStyleName]?.backgroundColor;
|
|
@@ -723,8 +796,15 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
723
796
|
colIndex++;
|
|
724
797
|
}
|
|
725
798
|
}
|
|
726
|
-
// Add row(s) for repeated rows
|
|
727
|
-
|
|
799
|
+
// Add row(s) for repeated rows. Every repetition past the first deep-copies the
|
|
800
|
+
// whole cell array, so rows x cols is what actually exhausts memory; charge those
|
|
801
|
+
// copies against the same budget.
|
|
802
|
+
const allowedRows = cells.length === 0
|
|
803
|
+
? rowsRepeated
|
|
804
|
+
: Math.min(rowsRepeated, 1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
|
|
805
|
+
for (let k = 0; k < allowedRows; k++) {
|
|
806
|
+
if ((k & 255) === 0)
|
|
807
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
728
808
|
const rowNode = {
|
|
729
809
|
type: 'row',
|
|
730
810
|
children: k === 0 ? cells : JSON.parse(JSON.stringify(cells))
|
|
@@ -754,6 +834,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
754
834
|
const body = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
|
|
755
835
|
if (!body)
|
|
756
836
|
return;
|
|
837
|
+
// One budget for the entire document. It has to span every table - spreadsheet sheets,
|
|
838
|
+
// ODT/ODP body tables, and nested tables alike - or a file sidesteps the cap simply by
|
|
839
|
+
// splitting a huge repeat expansion across many small tables. `traverse` and the
|
|
840
|
+
// spreadsheet branch below both close over this; `parseTable` receives it explicitly.
|
|
841
|
+
const cellBudget = createCellBudget(config);
|
|
757
842
|
// Parse automatic styles (local to content.xml)
|
|
758
843
|
const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
|
|
759
844
|
if (automaticStyles) {
|
|
@@ -917,7 +1002,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
917
1002
|
}
|
|
918
1003
|
else if (node.tagName === "table:table") {
|
|
919
1004
|
// Parse table with proper structure
|
|
920
|
-
const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
1005
|
+
const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml, cellBudget);
|
|
921
1006
|
if (asSheet) {
|
|
922
1007
|
tableNode.type = 'sheet';
|
|
923
1008
|
const sheetName = node.getAttribute("table:name");
|
|
@@ -1134,7 +1219,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1134
1219
|
traverse(textBox, targetArray, isHeading || forceHeading, sourceXml);
|
|
1135
1220
|
}
|
|
1136
1221
|
else if (table) {
|
|
1137
|
-
const tableNode = parseTable(table, paragraphStyleMap, styleMap, config, sourceXml);
|
|
1222
|
+
const tableNode = parseTable(table, paragraphStyleMap, styleMap, config, sourceXml, cellBudget);
|
|
1138
1223
|
if (config.includeRawContent)
|
|
1139
1224
|
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, sourceXml, config);
|
|
1140
1225
|
targetArray.push(tableNode);
|
|
@@ -1253,14 +1338,15 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1253
1338
|
const tableRows = (0, xmlUtils_js_1.getElementsByTagName)(table, "table:table-row");
|
|
1254
1339
|
let rowIndex = 0;
|
|
1255
1340
|
for (let r = 0; r < tableRows.length; r++) {
|
|
1341
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
1256
1342
|
const row = tableRows[r];
|
|
1257
1343
|
const cells = [];
|
|
1258
1344
|
const tableCells = (0, xmlUtils_js_1.getElementsByTagName)(row, "table:table-cell");
|
|
1259
1345
|
let colIndex = 0;
|
|
1260
|
-
const rowsRepeated =
|
|
1346
|
+
const rowsRepeated = toRepeatCount(row.getAttribute("table:number-rows-repeated"));
|
|
1261
1347
|
for (let c = 0; c < tableCells.length; c++) {
|
|
1262
1348
|
const cell = tableCells[c];
|
|
1263
|
-
const colsRepeated =
|
|
1349
|
+
const colsRepeated = toRepeatCount(cell.getAttribute("table:number-columns-repeated"));
|
|
1264
1350
|
// Extract text from cell (paragraphs inside cell)
|
|
1265
1351
|
let cellText = "";
|
|
1266
1352
|
const children = [];
|
|
@@ -1385,12 +1471,26 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1385
1471
|
children.push(chartNode);
|
|
1386
1472
|
}
|
|
1387
1473
|
}
|
|
1388
|
-
// Add cell(s)
|
|
1389
|
-
|
|
1390
|
-
|
|
1391
|
-
|
|
1392
|
-
|
|
1393
|
-
|
|
1474
|
+
// Add cell(s). The repeat count is attacker-influenced, so the loop
|
|
1475
|
+
// is bounded by the document's remaining cell budget rather than by
|
|
1476
|
+
// the attribute. An empty ODS cell creates nothing, so it costs no
|
|
1477
|
+
// budget - which is what keeps the huge trailing-empty runs real
|
|
1478
|
+
// files carry (number-columns-repeated="16384") free.
|
|
1479
|
+
// For ODS an empty cell materializes nothing, so a huge
|
|
1480
|
+
// number-columns-repeated on a blank cell (the normal way ODF marks a
|
|
1481
|
+
// trailing empty run) is skipped in O(1) by advancing the column index
|
|
1482
|
+
// rather than spinning the loop colsRepeated times for zero output -
|
|
1483
|
+
// that spin was itself a CPU denial-of-service, unbounded by the cell
|
|
1484
|
+
// budget because it created no cells to charge against.
|
|
1485
|
+
const willMaterialize = (cellText || children.length > 0 || fileType !== 'ods');
|
|
1486
|
+
if (!willMaterialize) {
|
|
1487
|
+
colIndex += colsRepeated;
|
|
1488
|
+
}
|
|
1489
|
+
else {
|
|
1490
|
+
const allowedCols = cellBudget.take(colsRepeated);
|
|
1491
|
+
for (let k = 0; k < allowedCols; k++) {
|
|
1492
|
+
if ((k & 1023) === 0)
|
|
1493
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
1394
1494
|
const cellNode = {
|
|
1395
1495
|
type: 'cell',
|
|
1396
1496
|
text: cellText,
|
|
@@ -1401,13 +1501,21 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1401
1501
|
cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, xmlString, config);
|
|
1402
1502
|
}
|
|
1403
1503
|
cells.push(cellNode);
|
|
1504
|
+
colIndex++;
|
|
1404
1505
|
}
|
|
1405
|
-
colIndex++;
|
|
1406
1506
|
}
|
|
1407
1507
|
}
|
|
1408
|
-
// Add row(s)
|
|
1508
|
+
// Add row(s). This is where the two repeats multiply: each repetition
|
|
1509
|
+
// deep-copies the whole cell array, so rows x cols is what actually
|
|
1510
|
+
// exhausts memory. Charge the copies against the same budget.
|
|
1409
1511
|
if (cells.length > 0) {
|
|
1410
|
-
|
|
1512
|
+
const allowedRows = Math.min(rowsRepeated,
|
|
1513
|
+
// The first row reuses `cells` rather than copying, so only the
|
|
1514
|
+
// repeats beyond it cost budget.
|
|
1515
|
+
1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
|
|
1516
|
+
for (let k = 0; k < allowedRows; k++) {
|
|
1517
|
+
if ((k & 255) === 0)
|
|
1518
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
1411
1519
|
const rowNode = {
|
|
1412
1520
|
type: 'row',
|
|
1413
1521
|
children: JSON.parse(JSON.stringify(cells)), // Deep copy for repeated rows
|
|
@@ -308,7 +308,10 @@ const parsePdf = async (buffer, config) => {
|
|
|
308
308
|
const uint8Array = new Uint8Array(buffer);
|
|
309
309
|
const loadingTask = pdfjs.getDocument({
|
|
310
310
|
data: uint8Array,
|
|
311
|
-
verbosity: 0 // ERRORS only, suppresses warnings
|
|
311
|
+
verbosity: 0, // ERRORS only, suppresses warnings
|
|
312
|
+
// Harden against untrusted PDFs: don't let pdf.js JIT font/CMap fast-paths
|
|
313
|
+
// compile via `new Function`.
|
|
314
|
+
isEvalSupported: false
|
|
312
315
|
});
|
|
313
316
|
// Handle loading errors, specifically missing worker in browser
|
|
314
317
|
let pdfDocument;
|
|
@@ -724,6 +724,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
724
724
|
const slideMasters = [];
|
|
725
725
|
// Now for processing all the other files - slides and notes.
|
|
726
726
|
for (const file of files) {
|
|
727
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
727
728
|
if (file.path.match(mediaFileRegex))
|
|
728
729
|
continue;
|
|
729
730
|
if (file.path.match(chartFileRegex))
|
|
@@ -1080,6 +1080,7 @@ const parseWord = async (buffer, config) => {
|
|
|
1080
1080
|
const bodyChildren = Array.from(body.childNodes);
|
|
1081
1081
|
let pendingAnchorIds = [];
|
|
1082
1082
|
for (const child of bodyChildren) {
|
|
1083
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
1083
1084
|
if ((0, xmlUtils_js_1.isElement)(child)) {
|
|
1084
1085
|
if (child.nodeName === 'w:p') {
|
|
1085
1086
|
content.push(parseParagraph(child, documentContent, pendingAnchorIds));
|