officeparser 7.0.3 → 7.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -1
- package/dist/OfficeParser.js +6 -0
- package/dist/defaults.js +16 -1
- package/dist/generators/ChunkingGenerator.js +23 -3
- package/dist/generators/PdfGenerator.js +51 -4
- package/dist/officeparser.browser.d.ts +118 -1
- package/dist/officeparser.browser.iife.js +47 -47
- package/dist/officeparser.browser.mjs +47 -47
- package/dist/parsers/CsvParser.js +5 -0
- package/dist/parsers/ExcelParser.js +6 -2
- package/dist/parsers/HtmlParser.js +5 -0
- package/dist/parsers/MarkdownParser.js +5 -0
- package/dist/parsers/OpenOfficeParser.js +4 -0
- package/dist/parsers/PdfParser.js +3 -0
- package/dist/parsers/PowerPointParser.js +4 -0
- package/dist/parsers/RtfParser.js +2 -0
- package/dist/parsers/WordParser.js +4 -0
- package/dist/sbom.cdx.json +98 -98
- package/dist/types.d.ts +118 -1
- package/dist/types.js +2 -0
- package/dist/utils/configUtils.js +14 -1
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +37 -2
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +8 -0
- package/dist/utils/xmlUtils.js +33 -1
- package/package.json +3 -2
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseCsv = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
5
6
|
/**
|
|
6
7
|
* Parses a CSV file and extracts a single sheet with rows and cells.
|
|
7
8
|
*
|
|
@@ -10,6 +11,10 @@ const astUtils_js_1 = require("../utils/astUtils.js");
|
|
|
10
11
|
* @returns A promise resolving to the parsed AST
|
|
11
12
|
*/
|
|
12
13
|
const parseCsv = async (buffer, config) => {
|
|
14
|
+
// Honour cancellation requests before the character-by-character parsing loop starts.
|
|
15
|
+
// CSV has no OCR or async I/O, but very large files can still occupy the thread for a
|
|
16
|
+
// noticeable duration, so short-circuiting on an aborted signal is still worthwhile.
|
|
17
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
13
18
|
const textStr = buffer.toString('utf-8');
|
|
14
19
|
const delimiter = config.csvDelimiter;
|
|
15
20
|
const records = [];
|
|
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
40
40
|
* @returns A promise resolving to the parsed AST
|
|
41
41
|
*/
|
|
42
42
|
const parseExcel = async (buffer, config) => {
|
|
43
|
+
// Honour cancellation requests immediately — before extracting the ZIP archive.
|
|
44
|
+
// XLSX parsing involves decompressing multiple XML sheets and potentially running OCR
|
|
45
|
+
// on embedded chart images, so short-circuiting here saves significant work.
|
|
46
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
43
47
|
const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
|
|
44
48
|
const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
|
|
45
49
|
const chartsRegex = /xl\/charts\/chart\d+.xml/g;
|
|
@@ -456,7 +460,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
456
460
|
const typeMatch = cAttrs.match(/t="([a-zA-Z]+)"/);
|
|
457
461
|
const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
|
|
458
462
|
const vMatch = cContent.match(/<v>([\s\S]*?)<\/v>/);
|
|
459
|
-
const tMatch = cContent.match(/<t
|
|
463
|
+
const tMatch = cContent.match(/<t\b[^>]*>([\s\S]*?)<\/t>/);
|
|
460
464
|
let text = '';
|
|
461
465
|
let cellNodes = [];
|
|
462
466
|
if (type === 's' && vMatch) {
|
|
@@ -473,7 +477,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
473
477
|
}
|
|
474
478
|
}
|
|
475
479
|
else if (type === 'inlineStr' && tMatch) {
|
|
476
|
-
text = tMatch[1].trim();
|
|
480
|
+
text = (0, xmlUtils_js_1.decodeXmlEntities)(tMatch[1].trim());
|
|
477
481
|
}
|
|
478
482
|
else if (vMatch) {
|
|
479
483
|
text = vMatch[1].trim();
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseHtml = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
5
6
|
const parseAttributes = (attrString) => {
|
|
6
7
|
const attrs = {};
|
|
7
8
|
const regex = /([a-zA-Z0-9\-:]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
|
|
@@ -95,6 +96,10 @@ const parseHtmlTree = (html) => {
|
|
|
95
96
|
return root;
|
|
96
97
|
};
|
|
97
98
|
const parseHtml = async (buffer, config) => {
|
|
99
|
+
// Honour cancellation requests before the HTML tree is built and traversed.
|
|
100
|
+
// The custom recursive HTML parser can be expensive for large documents;
|
|
101
|
+
// rejecting early here prevents both the parsing and the subsequent AST construction.
|
|
102
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
98
103
|
const textStr = buffer.toString('utf-8');
|
|
99
104
|
const root = parseHtmlTree(textStr);
|
|
100
105
|
// Find head and body
|
|
@@ -2,7 +2,12 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseMarkdown = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
5
6
|
const parseMarkdown = async (buffer, config) => {
|
|
7
|
+
// Honour cancellation requests before the line-by-line Markdown scanning loop begins.
|
|
8
|
+
// Markdown parsing is entirely synchronous and CPU-bound, so failing fast avoids
|
|
9
|
+
// processing content whose result will be discarded anyway.
|
|
10
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
6
11
|
let textStr = buffer.toString('utf-8');
|
|
7
12
|
textStr = textStr.replace(/\r\n/g, '\n');
|
|
8
13
|
const content = [];
|
|
@@ -39,6 +39,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
39
39
|
* @returns A promise resolving to the parsed AST
|
|
40
40
|
*/
|
|
41
41
|
const parseOpenOffice = async (buffer, config) => {
|
|
42
|
+
// Honour cancellation requests immediately — before extracting the ZIP archive.
|
|
43
|
+
// ODF containers (ODT/ODS/ODP) bundle content.xml, styles.xml, and media files;
|
|
44
|
+
// aborting early avoids needlessly inflating and parsing all of those resources.
|
|
45
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
42
46
|
const contentFileRegex = /content\.xml/;
|
|
43
47
|
const objectContentFileRegex = /Object \d+\/content\.xml/;
|
|
44
48
|
const mediaFileRegex = /(Pictures|media)\/.*/;
|
|
@@ -266,6 +266,7 @@ function convertToRgbaBuffer(data, width, height, kind) {
|
|
|
266
266
|
* @returns Promise resolving to the parsed AST
|
|
267
267
|
*/
|
|
268
268
|
const parsePdf = async (buffer, config) => {
|
|
269
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
269
270
|
const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
|
|
270
271
|
// Configure worker
|
|
271
272
|
const workerSrc = config.pdfWorkerSrc;
|
|
@@ -398,6 +399,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
398
399
|
}
|
|
399
400
|
// --- First Pass: Collect all items for font statistics ---
|
|
400
401
|
for (let i = 1; i <= numPages; i++) {
|
|
402
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
401
403
|
let page;
|
|
402
404
|
let textContent;
|
|
403
405
|
const pageItems = [];
|
|
@@ -575,6 +577,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
575
577
|
const fontStats = calculateFontStats(allPageItems);
|
|
576
578
|
// --- Second Pass: Process pages with font statistics ---
|
|
577
579
|
for (let i = 0; i < allPageItems.length; i++) {
|
|
580
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
578
581
|
const pageNum = i + 1;
|
|
579
582
|
let page;
|
|
580
583
|
try {
|
|
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
40
40
|
* @returns A promise resolving to the parsed AST
|
|
41
41
|
*/
|
|
42
42
|
const parsePowerPoint = async (buffer, config) => {
|
|
43
|
+
// Honour cancellation requests immediately — before extracting the ZIP archive.
|
|
44
|
+
// PPTX presentations can have many slides with media/charts and optional OCR per image,
|
|
45
|
+
// so an early abort prevents decompressing and traversing data that will be discarded.
|
|
46
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
43
47
|
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
44
48
|
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
45
49
|
const slideRelsRegex = /ppt\/slides\/_rels\/slide\d+\.xml\.rels/;
|
|
@@ -359,6 +359,7 @@ exports.SimpleRtfParser = SimpleRtfParser;
|
|
|
359
359
|
* @returns The parsed AST.
|
|
360
360
|
*/
|
|
361
361
|
const parseRtf = async (buffer, config) => {
|
|
362
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
362
363
|
const parser = new SimpleRtfParser(buffer);
|
|
363
364
|
const doc = parser.parse();
|
|
364
365
|
// Extract font and color tables
|
|
@@ -1653,6 +1654,7 @@ const parseRtf = async (buffer, config) => {
|
|
|
1653
1654
|
// Perform OCR if enabled
|
|
1654
1655
|
if (config.ocr && config.extractAttachments) {
|
|
1655
1656
|
for (const attachment of attachments) {
|
|
1657
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
1656
1658
|
if (attachment.mimeType.startsWith('image/')) {
|
|
1657
1659
|
try {
|
|
1658
1660
|
// Convert base64 data back to Buffer for Tesseract.js
|
|
@@ -86,6 +86,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
86
86
|
* @returns A promise resolving to the parsed AST
|
|
87
87
|
*/
|
|
88
88
|
const parseWord = async (buffer, config) => {
|
|
89
|
+
// Honour cancellation requests immediately — before opening the ZIP archive, loading XML
|
|
90
|
+
// files, or kicking off any OCR work. DOCX files can be large and the inflate + XML-parse
|
|
91
|
+
// steps are synchronous-heavy, so failing fast here avoids wasted CPU time.
|
|
92
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
89
93
|
const documentFileRegex = /word\/document[\d+]?.xml/;
|
|
90
94
|
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
|
|
91
95
|
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
|