officeparser 4.1.1 → 4.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -13,6 +13,8 @@ A Node.js library to parse text out of any office file.
13
13
 
14
14
 
15
15
  #### Update
16
+ * 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
17
+ * 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
16
18
  * 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
17
19
  * 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
18
20
  * 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
package/officeParser.js CHANGED
@@ -1,11 +1,11 @@
1
1
  #!/usr/bin/env node
2
2
 
3
- const decompress = require('decompress');
4
- const fs = require('fs');
5
- const rimraf = require('rimraf');
6
- const fileType = require('file-type');
7
- const pdfjs = require('./pdfjs-dist-build/pdf.js');
8
- const { DOMParser } = require('@xmldom/xmldom');
3
+ const decompress = require('decompress');
4
+ const fs = require('fs');
5
+ const { rimrafSync } = require('rimraf');
6
+ const fileType = require('file-type');
7
+ const pdfjs = require('./pdfjs-dist-build/pdf.js');
8
+ const { DOMParser } = require('@xmldom/xmldom');
9
9
 
10
10
  /** Header for error messages */
11
11
  const ERRORHEADER = "[OfficeParser]: ";
@@ -126,6 +126,7 @@ function parsePowerPoint(filepath, callback, config) {
126
126
  // Files regex that hold our content of interest
127
127
  const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
128
128
  const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
129
+ const slideNumberRegex = /lide(\d+)\.xml/;
129
130
 
130
131
  /** The decompress location which contains the filename in it */
131
132
  const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
@@ -134,6 +135,17 @@ function parsePowerPoint(filepath, callback, config) {
134
135
  { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
135
136
  )
136
137
  .then(files => {
138
+ // Sort files by slide number and their notes (if any).
139
+ files.sort((a, b) => {
140
+ const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
141
+ const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
142
+
143
+ const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
144
+ const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
145
+
146
+ return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
147
+ });
148
+
137
149
  // Verify if atleast the slides xml files exist in the extracted files list.
138
150
  if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
139
151
  throw ERRORMSG.fileCorrupted(filepath);
@@ -218,18 +230,45 @@ function parseExcel(filepath, callback, config) {
218
230
  })
219
231
  // ********************************** excel xml files explanation ***************************************
220
232
  // Structure of xmlContent of an excel file is a bit complex.
221
- // We have a sharedStrings.xml file which has strings inside t tags
233
+ // We usually have a sharedStrings.xml file which has strings inside t tags
234
+ // However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
222
235
  // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
223
236
  // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
224
237
  // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
238
+ // However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
239
+ // We extract either the inline strings or use the value to get numbers of text from shared strings.
225
240
  // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
226
241
  // ******************************************************************************************************
227
242
  .then(xmlContentFilesObject => {
228
243
  /** Store all the text content to respond */
229
244
  let responseText = [];
230
245
 
231
- /** Find text nodes with t tags in sharedStrings xml file */
232
- const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
246
+ /** Function to check if the given c node is a valid inline string node. */
247
+ function isValidInlineStringCNode(cNode) {
248
+ // Initial check to see if the passed node is a cNode
249
+ if (cNode.tagName.toLowerCase() != 'c')
250
+ return false;
251
+ if (cNode.getAttribute("t") != 'inlineStr')
252
+ return false;
253
+ const childNodesNamedIs = cNode.getElementsByTagName('is');
254
+ if (childNodesNamedIs.length != 1)
255
+ return false;
256
+ const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
257
+ if (childNodesNamedT.length != 1)
258
+ return false;
259
+ return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
260
+ }
261
+
262
+ /** Function to check if the given c node has a valid v node */
263
+ function hasValidVNodeInCNode(cNode) {
264
+ return cNode.getElementsByTagName("v")[0]
265
+ && cNode.getElementsByTagName("v")[0].childNodes[0]
266
+ && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
267
+ }
268
+
269
+ /** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
270
+ const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
271
+ : [];
233
272
  /** Create shared string array. This will be used as a map to get strings from within sheet files. */
234
273
  const sharedStrings = Array.from(sharedStringsXmlTNodesList)
235
274
  .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
@@ -238,25 +277,33 @@ function parseExcel(filepath, callback, config) {
238
277
  xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
239
278
  /** Find text nodes with c tags in sharedStrings xml file */
240
279
  const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
241
- // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
280
+ // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
242
281
  responseText.push(
243
282
  Array.from(sheetsXmlCNodesList)
244
- // Filter c nodes than do not have any valid v nodes
245
- .filter(cNode => cNode.getElementsByTagName("v")[0]
246
- && cNode.getElementsByTagName("v")[0].childNodes[0]
247
- && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
283
+ // Filter out invalid c nodes
284
+ .filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
248
285
  .map(cNode => {
249
- /** Flag whether this node's value represents a string index */
250
- const isString = cNode.getAttribute("t") == "s";
251
- /** Find value nodes represented by v tags */
252
- const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
253
- // Validate text
254
- if (isString && value >= sharedStrings.length)
255
- throw ERRORMSG.fileCorrupted(filepath);
256
-
257
- return isString
258
- ? sharedStrings[value]
259
- : value;
286
+ // Processing if this is a valid inline string c node.
287
+ if (isValidInlineStringCNode(cNode))
288
+ return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
289
+
290
+ // Processing if this c node has a valid v node.
291
+ if (hasValidVNodeInCNode(cNode)) {
292
+ /** Flag whether this node's value represents an index in the shared string array */
293
+ const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
294
+ /** Find value nodes represented by v tags */
295
+ const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
296
+ // Validate text
297
+ if (isIndexInSharedStrings && value >= sharedStrings.length)
298
+ throw ERRORMSG.fileCorrupted(filepath);
299
+
300
+ return isIndexInSharedStrings
301
+ ? sharedStrings[value]
302
+ : value;
303
+ }
304
+ // TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
305
+ // Not the case now but it could happen and it is better to be safe.
306
+ return '';
260
307
  })
261
308
  // Join each cell text within a sheet with a space.
262
309
  .join(config.newlineDelimiter ?? "\n")
@@ -496,7 +543,7 @@ function parseOffice(file, callback, config = {}) {
496
543
  }
497
544
 
498
545
  // create temp file subdirectory if it does not exist
499
- fs.mkdirSync(`${internalConfig.tempFilesLocation}/tempfiles`, { recursive: true });
546
+ fs.mkdirSync(getTempFilesDirectory(internalConfig.tempFilesLocation), { recursive: true });
500
547
 
501
548
  // Check if buffer
502
549
  if (Buffer.isBuffer(file)) {
@@ -505,11 +552,11 @@ function parseOffice(file, callback, config = {}) {
505
552
  .then(data =>
506
553
  {
507
554
  // temp file name
508
- const newfilepath = getNewFileName(internalConfig.tempFilesLocation, data.ext.toLowerCase());
555
+ const newFileName = getNewFileName(data.ext.toLowerCase());
509
556
  // write new file
510
- fs.writeFileSync(newfilepath, file);
557
+ fs.writeFileSync(getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName), file);
511
558
  // resolve promise
512
- res(newfilepath);
559
+ res(newFileName);
513
560
  })
514
561
  .catch(() => rej(ERRORMSG.improperBuffers));
515
562
  return;
@@ -522,16 +569,18 @@ function parseOffice(file, callback, config = {}) {
522
569
  throw ERRORMSG.fileDoesNotExist(file);
523
570
 
524
571
  // temp file name
525
- const newfilepath = getNewFileName(internalConfig.tempFilesLocation, file.split(".").pop().toLowerCase());
572
+ const newFileName = getNewFileName(file.split(".").pop().toLowerCase());
526
573
  // Copy the file into a temp location with the temp name
527
- fs.copyFileSync(file, newfilepath)
574
+ fs.copyFileSync(file, getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName))
528
575
  // resolve promise
529
- res(newfilepath);
576
+ res(newFileName);
530
577
  });
531
578
 
532
579
  // Process filePreparedPromise resolution.
533
580
  filePreparedPromise
534
- .then(filepath => {
581
+ .then(filename => {
582
+ // The file path
583
+ const filepath = getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), filename);
535
584
  // File extension. Already in lowercase when we prepared the temp file above.
536
585
  const extension = filepath.split(".").pop();
537
586
 
@@ -562,9 +611,33 @@ function parseOffice(file, callback, config = {}) {
562
611
  /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
563
612
  function internalCallback(data, err) {
564
613
  // Check if we need to preserve unzipped content files or delete them.
565
- if (!internalConfig.preserveTempFiles)
566
- // Delete decompress sublocation.
567
- rimraf(internalConfig.tempFilesLocation, rimrafErr => consoleError(rimrafErr, internalConfig.outputErrorToConsole));
614
+ if (!internalConfig.preserveTempFiles) {
615
+ /** Safely delete location */
616
+ function safelyDeleteLocation(location, deleteDirIfEmpty) {
617
+ if (!fs.existsSync(location))
618
+ return;
619
+
620
+ if (!fs.lstatSync(location).isDirectory() // If not directory
621
+ || !deleteDirIfEmpty // or if the deleteDirIfEmpty is false or undefined
622
+ || fs.readdirSync(location).length == 0) { // or if it is true, we check if the contents are empty.
623
+ try {
624
+ rimrafSync(location);
625
+ }
626
+ catch(rimrafErr) {
627
+ consoleError(rimrafErr, internalConfig.outputErrorToConsole);
628
+ }
629
+ }
630
+ }
631
+
632
+ // We delete our file as well as the extracted files inside the office files.
633
+ // There is no extraction for pdf files because we don't support any unpacking of it.
634
+ // After removing files, we check if the folders containing them are empty.
635
+ // If yes, we remove those folders too.
636
+ safelyDeleteLocation(filepath);
637
+ safelyDeleteLocation(getFilePath(internalConfig.tempFilesLocation, filename));
638
+ safelyDeleteLocation(getTempFilesDirectory(internalConfig.tempFilesLocation), true);
639
+ safelyDeleteLocation(internalConfig.tempFilesLocation, true);
640
+ }
568
641
 
569
642
  // Check if there is an error. Throw if there is an error.
570
643
  if (err)
@@ -597,21 +670,30 @@ function parseOfficeAsync(file, config = {}) {
597
670
  let globalFileNameIterator = 0;
598
671
  /**
599
672
  * File Name generator that takes the extension as an input and returns a file name that comprises a timestamp and an incrementing number
600
- * to allow the files to be sorted in chronological order
601
- * @param {string} tempFilesLocation Directory whether this new file needs to be stored
602
- * @param {string} ext File extension for this new generated file name
673
+ * to allow the files to be sorted in chronological order. We also prefix them with ppid and pid of the process to support
674
+ * common destination from worker threads as well as multiple processes running them together.
675
+ * @param {string} ext File extension for this new generated file name
603
676
  * @returns {string}
604
677
  */
605
- function getNewFileName(tempFilesLocation, ext) {
678
+ function getNewFileName(ext) {
606
679
  // Get the iterator part of the file name
607
680
  let iteratorPart = (globalFileNameIterator++).toString().padStart(5, '0');
608
681
  // We want the iterator part of the file name to be of 5 digits.
609
682
  // Therefore, when the iterator crosses into 6 digits, we reset it to 0.
610
683
  if (globalFileNameIterator > 99999)
611
684
  globalFileNameIterator = 0;
685
+ // Return the file name with ppid and pid to allow unique names even with worker threads.
686
+ return `${process.ppid}_${process.pid}_${new Date().getTime().toString() + iteratorPart}.${ext}`;
687
+ }
688
+
689
+ /** Gets directory for storing files. */
690
+ function getTempFilesDirectory(root) {
691
+ return `${root}/tempfiles`;
692
+ }
612
693
 
613
- // Return the file name
614
- return `${tempFilesLocation}/tempfiles/${new Date().getTime().toString() + iteratorPart}.${ext}`;
694
+ /** Gets file path for the supplied directory and the file name. */
695
+ function getFilePath(directory, fileName) {
696
+ return `${directory}/${fileName}`;
615
697
  }
616
698
 
617
699
  /**
@@ -634,7 +716,7 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
634
716
 
635
717
 
636
718
  // Run this library on CLI
637
- if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
719
+ if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser")) {
638
720
  if (process.argv.length == 2) {
639
721
  // continue
640
722
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.1.1",
3
+ "version": "4.2.0",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [
@@ -44,10 +44,10 @@
44
44
  "homepage": "https://github.com/harshankur/officeParser#readme",
45
45
  "dependencies": {
46
46
  "@xmldom/xmldom": "^0.8.10",
47
- "decompress": "^4.2.0",
47
+ "decompress": "^4.2.1",
48
48
  "file-type": "^16.5.4",
49
49
  "node-ensure": "^0.0.0",
50
- "rimraf": "^2.6.3"
50
+ "rimraf": "^5.0.10"
51
51
  },
52
52
  "devDependencies": {
53
53
  "@types/decompress": "^4.2.6",