officeparser 7.0.3 → 7.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,6 +2,7 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseCsv = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
5
6
  /**
6
7
  * Parses a CSV file and extracts a single sheet with rows and cells.
7
8
  *
@@ -10,6 +11,10 @@ const astUtils_js_1 = require("../utils/astUtils.js");
10
11
  * @returns A promise resolving to the parsed AST
11
12
  */
12
13
  const parseCsv = async (buffer, config) => {
14
+ // Honour cancellation requests before the character-by-character parsing loop starts.
15
+ // CSV has no OCR or async I/O, but very large files can still occupy the thread for a
16
+ // noticeable duration, so short-circuiting on an aborted signal is still worthwhile.
17
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
13
18
  const textStr = buffer.toString('utf-8');
14
19
  const delimiter = config.csvDelimiter;
15
20
  const records = [];
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
40
40
  * @returns A promise resolving to the parsed AST
41
41
  */
42
42
  const parseExcel = async (buffer, config) => {
43
+ // Honour cancellation requests immediately — before extracting the ZIP archive.
44
+ // XLSX parsing involves decompressing multiple XML sheets and potentially running OCR
45
+ // on embedded chart images, so short-circuiting here saves significant work.
46
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
43
47
  const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
44
48
  const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
45
49
  const chartsRegex = /xl\/charts\/chart\d+.xml/g;
@@ -456,7 +460,7 @@ const parseExcel = async (buffer, config) => {
456
460
  const typeMatch = cAttrs.match(/t="([a-zA-Z]+)"/);
457
461
  const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
458
462
  const vMatch = cContent.match(/<v>([\s\S]*?)<\/v>/);
459
- const tMatch = cContent.match(/<t>([\s\S]*?)<\/t>/);
463
+ const tMatch = cContent.match(/<t\b[^>]*>([\s\S]*?)<\/t>/);
460
464
  let text = '';
461
465
  let cellNodes = [];
462
466
  if (type === 's' && vMatch) {
@@ -473,7 +477,7 @@ const parseExcel = async (buffer, config) => {
473
477
  }
474
478
  }
475
479
  else if (type === 'inlineStr' && tMatch) {
476
- text = tMatch[1].trim();
480
+ text = (0, xmlUtils_js_1.decodeXmlEntities)(tMatch[1].trim());
477
481
  }
478
482
  else if (vMatch) {
479
483
  text = vMatch[1].trim();
@@ -2,6 +2,7 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseHtml = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
5
6
  const parseAttributes = (attrString) => {
6
7
  const attrs = {};
7
8
  const regex = /([a-zA-Z0-9\-:]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
@@ -95,6 +96,10 @@ const parseHtmlTree = (html) => {
95
96
  return root;
96
97
  };
97
98
  const parseHtml = async (buffer, config) => {
99
+ // Honour cancellation requests before the HTML tree is built and traversed.
100
+ // The custom recursive HTML parser can be expensive for large documents;
101
+ // rejecting early here prevents both the parsing and the subsequent AST construction.
102
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
98
103
  const textStr = buffer.toString('utf-8');
99
104
  const root = parseHtmlTree(textStr);
100
105
  // Find head and body
@@ -2,7 +2,12 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseMarkdown = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
5
6
  const parseMarkdown = async (buffer, config) => {
7
+ // Honour cancellation requests before the line-by-line Markdown scanning loop begins.
8
+ // Markdown parsing is entirely synchronous and CPU-bound, so failing fast avoids
9
+ // processing content whose result will be discarded anyway.
10
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
6
11
  let textStr = buffer.toString('utf-8');
7
12
  textStr = textStr.replace(/\r\n/g, '\n');
8
13
  const content = [];
@@ -39,6 +39,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
39
39
  * @returns A promise resolving to the parsed AST
40
40
  */
41
41
  const parseOpenOffice = async (buffer, config) => {
42
+ // Honour cancellation requests immediately — before extracting the ZIP archive.
43
+ // ODF containers (ODT/ODS/ODP) bundle content.xml, styles.xml, and media files;
44
+ // aborting early avoids needlessly inflating and parsing all of those resources.
45
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
42
46
  const contentFileRegex = /content\.xml/;
43
47
  const objectContentFileRegex = /Object \d+\/content\.xml/;
44
48
  const mediaFileRegex = /(Pictures|media)\/.*/;
@@ -266,6 +266,7 @@ function convertToRgbaBuffer(data, width, height, kind) {
266
266
  * @returns Promise resolving to the parsed AST
267
267
  */
268
268
  const parsePdf = async (buffer, config) => {
269
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
269
270
  const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
270
271
  // Configure worker
271
272
  const workerSrc = config.pdfWorkerSrc;
@@ -398,6 +399,7 @@ const parsePdf = async (buffer, config) => {
398
399
  }
399
400
  // --- First Pass: Collect all items for font statistics ---
400
401
  for (let i = 1; i <= numPages; i++) {
402
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
401
403
  let page;
402
404
  let textContent;
403
405
  const pageItems = [];
@@ -575,6 +577,7 @@ const parsePdf = async (buffer, config) => {
575
577
  const fontStats = calculateFontStats(allPageItems);
576
578
  // --- Second Pass: Process pages with font statistics ---
577
579
  for (let i = 0; i < allPageItems.length; i++) {
580
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
578
581
  const pageNum = i + 1;
579
582
  let page;
580
583
  try {
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
40
40
  * @returns A promise resolving to the parsed AST
41
41
  */
42
42
  const parsePowerPoint = async (buffer, config) => {
43
+ // Honour cancellation requests immediately — before extracting the ZIP archive.
44
+ // PPTX presentations can have many slides with media/charts and optional OCR per image,
45
+ // so an early abort prevents decompressing and traversing data that will be discarded.
46
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
43
47
  const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
44
48
  const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
45
49
  const slideRelsRegex = /ppt\/slides\/_rels\/slide\d+\.xml\.rels/;
@@ -359,6 +359,7 @@ exports.SimpleRtfParser = SimpleRtfParser;
359
359
  * @returns The parsed AST.
360
360
  */
361
361
  const parseRtf = async (buffer, config) => {
362
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
362
363
  const parser = new SimpleRtfParser(buffer);
363
364
  const doc = parser.parse();
364
365
  // Extract font and color tables
@@ -1653,6 +1654,7 @@ const parseRtf = async (buffer, config) => {
1653
1654
  // Perform OCR if enabled
1654
1655
  if (config.ocr && config.extractAttachments) {
1655
1656
  for (const attachment of attachments) {
1657
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
1656
1658
  if (attachment.mimeType.startsWith('image/')) {
1657
1659
  try {
1658
1660
  // Convert base64 data back to Buffer for Tesseract.js
@@ -86,6 +86,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
86
86
  * @returns A promise resolving to the parsed AST
87
87
  */
88
88
  const parseWord = async (buffer, config) => {
89
+ // Honour cancellation requests immediately — before opening the ZIP archive, loading XML
90
+ // files, or kicking off any OCR work. DOCX files can be large and the inflate + XML-parse
91
+ // steps are synchronous-heavy, so failing fast here avoids wasted CPU time.
92
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
89
93
  const documentFileRegex = /word\/document[\d+]?.xml/;
90
94
  const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
91
95
  const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;