officeparser 4.1.2 → 4.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -13,6 +13,8 @@ A Node.js library to parse text out of any office file.
13
13
 
14
14
 
15
15
  #### Update
16
+ * 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
17
+ * 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
16
18
  * 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
17
19
  * 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
18
20
  * 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
package/officeParser.js CHANGED
@@ -1,11 +1,11 @@
1
1
  #!/usr/bin/env node
2
2
 
3
- const decompress = require('decompress');
4
- const fs = require('fs');
5
- const rimraf = require('rimraf');
6
- const fileType = require('file-type');
7
- const pdfjs = require('./pdfjs-dist-build/pdf.js');
8
- const { DOMParser } = require('@xmldom/xmldom');
3
+ const decompress = require('decompress');
4
+ const fs = require('fs');
5
+ const { rimrafSync } = require('rimraf');
6
+ const fileType = require('file-type');
7
+ const pdfjs = require('./pdfjs-dist-build/pdf.js');
8
+ const { DOMParser } = require('@xmldom/xmldom');
9
9
 
10
10
  /** Header for error messages */
11
11
  const ERRORHEADER = "[OfficeParser]: ";
@@ -543,7 +543,7 @@ function parseOffice(file, callback, config = {}) {
543
543
  }
544
544
 
545
545
  // create temp file subdirectory if it does not exist
546
- fs.mkdirSync(`${internalConfig.tempFilesLocation}/tempfiles`, { recursive: true });
546
+ fs.mkdirSync(getTempFilesDirectory(internalConfig.tempFilesLocation), { recursive: true });
547
547
 
548
548
  // Check if buffer
549
549
  if (Buffer.isBuffer(file)) {
@@ -552,11 +552,11 @@ function parseOffice(file, callback, config = {}) {
552
552
  .then(data =>
553
553
  {
554
554
  // temp file name
555
- const newfilepath = getNewFileName(internalConfig.tempFilesLocation, data.ext.toLowerCase());
555
+ const newFileName = getNewFileName(data.ext.toLowerCase());
556
556
  // write new file
557
- fs.writeFileSync(newfilepath, file);
557
+ fs.writeFileSync(getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName), file);
558
558
  // resolve promise
559
- res(newfilepath);
559
+ res(newFileName);
560
560
  })
561
561
  .catch(() => rej(ERRORMSG.improperBuffers));
562
562
  return;
@@ -569,16 +569,18 @@ function parseOffice(file, callback, config = {}) {
569
569
  throw ERRORMSG.fileDoesNotExist(file);
570
570
 
571
571
  // temp file name
572
- const newfilepath = getNewFileName(internalConfig.tempFilesLocation, file.split(".").pop().toLowerCase());
572
+ const newFileName = getNewFileName(file.split(".").pop().toLowerCase());
573
573
  // Copy the file into a temp location with the temp name
574
- fs.copyFileSync(file, newfilepath)
574
+ fs.copyFileSync(file, getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName))
575
575
  // resolve promise
576
- res(newfilepath);
576
+ res(newFileName);
577
577
  });
578
578
 
579
579
  // Process filePreparedPromise resolution.
580
580
  filePreparedPromise
581
- .then(filepath => {
581
+ .then(filename => {
582
+ // The file path
583
+ const filepath = getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), filename);
582
584
  // File extension. Already in lowercase when we prepared the temp file above.
583
585
  const extension = filepath.split(".").pop();
584
586
 
@@ -609,9 +611,33 @@ function parseOffice(file, callback, config = {}) {
609
611
  /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
610
612
  function internalCallback(data, err) {
611
613
  // Check if we need to preserve unzipped content files or delete them.
612
- if (!internalConfig.preserveTempFiles)
613
- // Delete decompress sublocation.
614
- rimraf(internalConfig.tempFilesLocation, rimrafErr => consoleError(rimrafErr, internalConfig.outputErrorToConsole));
614
+ if (!internalConfig.preserveTempFiles) {
615
+ /** Safely delete location */
616
+ function safelyDeleteLocation(location, deleteDirIfEmpty) {
617
+ if (!fs.existsSync(location))
618
+ return;
619
+
620
+ if (!fs.lstatSync(location).isDirectory() // If not directory
621
+ || !deleteDirIfEmpty // or if the deleteDirIfEmpty is false or undefined
622
+ || fs.readdirSync(location).length == 0) { // or if it is true, we check if the contents are empty.
623
+ try {
624
+ rimrafSync(location);
625
+ }
626
+ catch(rimrafErr) {
627
+ consoleError(rimrafErr, internalConfig.outputErrorToConsole);
628
+ }
629
+ }
630
+ }
631
+
632
+ // We delete our file as well as the extracted files inside the office files.
633
+ // There is no extraction for pdf files because we don't support any unpacking of it.
634
+ // After removing files, we check if the folders containing them are empty.
635
+ // If yes, we remove those folders too.
636
+ safelyDeleteLocation(filepath);
637
+ safelyDeleteLocation(getFilePath(internalConfig.tempFilesLocation, filename));
638
+ safelyDeleteLocation(getTempFilesDirectory(internalConfig.tempFilesLocation), true);
639
+ safelyDeleteLocation(internalConfig.tempFilesLocation, true);
640
+ }
615
641
 
616
642
  // Check if there is an error. Throw if there is an error.
617
643
  if (err)
@@ -644,21 +670,30 @@ function parseOfficeAsync(file, config = {}) {
644
670
  let globalFileNameIterator = 0;
645
671
  /**
646
672
  * File Name generator that takes the extension as an input and returns a file name that comprises a timestamp and an incrementing number
647
- * to allow the files to be sorted in chronological order
648
- * @param {string} tempFilesLocation Directory whether this new file needs to be stored
649
- * @param {string} ext File extension for this new generated file name
673
+ * to allow the files to be sorted in chronological order. We also prefix them with ppid and pid of the process to support
674
+ * common destination from worker threads as well as multiple processes running them together.
675
+ * @param {string} ext File extension for this new generated file name
650
676
  * @returns {string}
651
677
  */
652
- function getNewFileName(tempFilesLocation, ext) {
678
+ function getNewFileName(ext) {
653
679
  // Get the iterator part of the file name
654
680
  let iteratorPart = (globalFileNameIterator++).toString().padStart(5, '0');
655
681
  // We want the iterator part of the file name to be of 5 digits.
656
682
  // Therefore, when the iterator crosses into 6 digits, we reset it to 0.
657
683
  if (globalFileNameIterator > 99999)
658
684
  globalFileNameIterator = 0;
685
+ // Return the file name with ppid and pid to allow unique names even with worker threads.
686
+ return `${process.ppid}_${process.pid}_${new Date().getTime().toString() + iteratorPart}.${ext}`;
687
+ }
688
+
689
+ /** Gets directory for storing files. */
690
+ function getTempFilesDirectory(root) {
691
+ return `${root}/tempfiles`;
692
+ }
659
693
 
660
- // Return the file name
661
- return `${tempFilesLocation}/tempfiles/${new Date().getTime().toString() + iteratorPart}.${ext}`;
694
+ /** Gets file path for the supplied directory and the file name. */
695
+ function getFilePath(directory, fileName) {
696
+ return `${directory}/${fileName}`;
662
697
  }
663
698
 
664
699
  /**
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.1.2",
3
+ "version": "4.2.0",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [
@@ -44,10 +44,10 @@
44
44
  "homepage": "https://github.com/harshankur/officeParser#readme",
45
45
  "dependencies": {
46
46
  "@xmldom/xmldom": "^0.8.10",
47
- "decompress": "^4.2.0",
47
+ "decompress": "^4.2.1",
48
48
  "file-type": "^16.5.4",
49
49
  "node-ensure": "^0.0.0",
50
- "rimraf": "^2.6.3"
50
+ "rimraf": "^5.0.10"
51
51
  },
52
52
  "devDependencies": {
53
53
  "@types/decompress": "^4.2.6",