officeparser 4.0.0 → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -97,6 +97,7 @@ officeParser.parseOfficeAsync(fileBuffers);
97
97
  *Optionally add a config object as 3rd variable to parseOffice for the following configurations*
98
98
  | flag | datatype | explanation |
99
99
  |----------------------|----------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
100
+ | tempFilesLocation | string | The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. **Please ensure that this directory actually exists.** Default is officeParsertemp. |
100
101
  | preserveTempFiles | boolean | Flag to not delete the internal content files and the possible duplicate temp files that it uses after unzipping office files. Default is false. It always deletes all of those files. |
101
102
  | outputErrorToConsole | boolean | Flag to show all the logs to console in case of an error. |
102
103
  | newlineDelimiter | string | The delimiter used for every new line in places that allow multiline text like word. Default is \n. |
@@ -172,34 +173,9 @@ function searchForTermInOfficeFile(searchterm, filepath): Promise<boolean> {
172
173
  .then(data => data.indexOf(searchterm) != -1)
173
174
  }
174
175
  ```
175
-
176
-
177
- \
178
176
  \
179
177
  **Please take note: I have breached convention in placing err as second argument in my callback but please understand that I had to do it to not break other people's existing modules.**
180
178
 
181
- *Optionally change decompression location for office Files at personalised locations for environments with restricted write access*
182
-
183
- ```js
184
- const officeParser = require('officeparser');
185
-
186
- // Default decompress location for office Files is "officeDist" in the directory where Node is started.
187
- // Put this file before parseOffice method to take effect.
188
- officeParser.setDecompressionLocation("/tmp"); // New decompression location would be "/tmp/officeDist"
189
-
190
- // P.S.: Setting location on a Windows environment with '\' hierarchy requires to be entered twice '\\'
191
- officeParser.setDecompressionLocation("C:\\tmp"); // New decompression location would be "C:\tmp\officeDist"
192
-
193
-
194
- officeParser.parseOffice("/path/to/officeFile", function(data, err){
195
- // "data" string in the callback here is the text parsed from the office file passed in the first argument above
196
- if (err) {
197
- console.log(err);
198
- return;
199
- }
200
- console.log(data);
201
- })
202
- ```
203
179
 
204
180
  ## Known Bugs
205
181
  1. Inconsistency and incorrectness in the positioning of footnotes and endnotes in .docx files where the footnotes and endnotes would end up at the end of the parsed text whereas it would be positioned exactly after the referenced word in .odt files.
package/officeParser.js CHANGED
@@ -5,7 +5,7 @@ const fs = require('fs');
5
5
  const rimraf = require('rimraf');
6
6
  const fileType = require('file-type');
7
7
  const pdfParse = require('pdf-parse');
8
- const { DOMParser } = require('xmldom');
8
+ const { DOMParser } = require('@xmldom/xmldom');
9
9
 
10
10
  /** Header for error messages */
11
11
  const ERRORHEADER = "[OfficeParser]: ";
@@ -14,14 +14,12 @@ const ERRORMSG = {
14
14
  extensionUnsupported: (ext) => `Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods, pdf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
15
15
  fileCorrupted: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
16
16
  fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
17
- locationNotFound: (location) => `Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
17
+ locationNotFound: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
18
18
  improperArguments: `Improper arguments`,
19
19
  improperBuffers: `Error occured while reading the file buffers`
20
20
  }
21
21
  /** Default sublocation for decompressing files under the current directory. */
22
- const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
23
- /** Location for decompressing files. Default is "officeDist" */
24
- let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
22
+ const DEFAULTDECOMPRESSSUBLOCATION = "officeParserTemp";
25
23
 
26
24
  /** Console error if allowed
27
25
  * @param {string} errorMessage Error message to show on the console
@@ -44,11 +42,12 @@ const parseString = (xml) => {
44
42
  };
45
43
 
46
44
  /** @typedef {Object} OfficeParserConfig
47
- * @property {boolean} preserveTempFiles Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
48
- * @property {boolean} outputErrorToConsole Flag to show all the logs to console in case of an error irrespective of your own handling.
49
- * @property {string} newlineDelimiter The delimiter used for every new line in places that allow multiline text like word. Default is \n.
50
- * @property {boolean} ignoreNotes Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
51
- * @property {boolean} putNotesAtLast Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
45
+ * @property {string} [tempFilesLocation] The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. Please ensure that this directory actually exists. Default is officeParsertemp.
46
+ * @property {boolean} [preserveTempFiles] Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
47
+ * @property {boolean} [outputErrorToConsole] Flag to show all the logs to console in case of an error irrespective of your own handling.
48
+ * @property {string} [newlineDelimiter] The delimiter used for every new line in places that allow multiline text like word. Default is \n.
49
+ * @property {boolean} [ignoreNotes] Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
50
+ * @property {boolean} [putNotesAtLast] Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
52
51
  */
53
52
 
54
53
 
@@ -64,7 +63,7 @@ function parseWord(filepath, callback, config) {
64
63
  const footnotesFile = 'word/footnotes.xml';
65
64
  const endnotesFile = 'word/endnotes.xml';
66
65
  /** The decompress location which contains the filename in it */
67
- const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
66
+ const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
68
67
  decompress(filepath,
69
68
  decompressLocation,
70
69
  { filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
@@ -127,7 +126,7 @@ function parsePowerPoint(filepath, callback, config) {
127
126
  const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
128
127
 
129
128
  /** The decompress location which contains the filename in it */
130
- const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
129
+ const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
131
130
  decompress(filepath,
132
131
  decompressLocation,
133
132
  { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
@@ -195,7 +194,7 @@ function parseExcel(filepath, callback, config) {
195
194
  const stringsFilePath = 'xl/sharedStrings.xml';
196
195
 
197
196
  /** The decompress location which contains the filename in it */
198
- const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
197
+ const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
199
198
  decompress(filepath,
200
199
  decompressLocation,
201
200
  { filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
@@ -306,7 +305,7 @@ function parseOpenOffice(filepath, callback, config) {
306
305
  const objectContentFilesRegex = /Object \d+\/content.xml/g;
307
306
 
308
307
  /** The decompress location which contains the filename in it */
309
- const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
308
+ const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
310
309
  decompress(filepath,
311
310
  decompressLocation,
312
311
  { filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
@@ -442,16 +441,32 @@ function parsePdf(filepath, callback, config) {
442
441
  }
443
442
 
444
443
  /** Main async function with callback to execute parseOffice for supported files
445
- * @param {string | Buffer} file File path or file buffers
446
- * @param {function} callback Callback function that returns value or error
447
- * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
444
+ * @param {string | Buffer} file File path or file buffers
445
+ * @param {function} callback Callback function that returns value or error
446
+ * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
448
447
  * @returns {void}
449
448
  */
450
449
  function parseOffice(file, callback, config = {}) {
450
+ const internalConfig = { ...config };
451
+ // Check if decompress location in the config is present.
452
+ // If it is valid, we set the final decompression location in the config.
453
+ // If it is not valid, we reject the promise with appropriate error message.
454
+ if (!internalConfig.tempFilesLocation)
455
+ internalConfig.tempFilesLocation = DEFAULTDECOMPRESSSUBLOCATION;
456
+ else {
457
+ const tempFilesLocation = `${internalConfig.tempFilesLocation}${internalConfig.tempFilesLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`;
458
+ if (!fs.existsSync(internalConfig.tempFilesLocation))
459
+ {
460
+ callback(undefined, ERRORMSG.locationNotFound(internalConfig.tempFilesLocation));
461
+ return;
462
+ }
463
+ internalConfig.tempFilesLocation = tempFilesLocation;
464
+ }
465
+
451
466
  // Prepare file for processing
452
467
  const filePreparedPromise = new Promise((res, rej) => {
453
468
  // create temp file subdirectory if it does not exist
454
- fs.mkdirSync(`${decompressSubLocation}/tempfiles`, { recursive: true });
469
+ fs.mkdirSync(`${internalConfig.tempFilesLocation}/tempfiles`, { recursive: true });
455
470
 
456
471
  // Check if buffer
457
472
  if (Buffer.isBuffer(file)) {
@@ -460,7 +475,7 @@ function parseOffice(file, callback, config = {}) {
460
475
  .then(data =>
461
476
  {
462
477
  // temp file name
463
- const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${data.ext.toLowerCase()}`;
478
+ const newfilepath = `${internalConfig.tempFilesLocation}/tempfiles/${new Date().getTime().toString()}.${data.ext.toLowerCase()}`;
464
479
  // write new file
465
480
  fs.writeFileSync(newfilepath, file);
466
481
  // resolve promise
@@ -477,7 +492,7 @@ function parseOffice(file, callback, config = {}) {
477
492
  throw ERRORMSG.fileDoesNotExist(file);
478
493
 
479
494
  // temp file name
480
- const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${file.split(".").pop().toLowerCase()}`;
495
+ const newfilepath = `${internalConfig.tempFilesLocation}/tempfiles/${new Date().getTime().toString()}.${file.split(".").pop().toLowerCase()}`;
481
496
  // Copy the file into a temp location with the temp name
482
497
  fs.copyFileSync(file, newfilepath)
483
498
  // resolve promise
@@ -493,21 +508,21 @@ function parseOffice(file, callback, config = {}) {
493
508
  // Switch between parsing functions depending on extension.
494
509
  switch(extension) {
495
510
  case "docx":
496
- parseWord(filepath, internalCallback, config);
511
+ parseWord(filepath, internalCallback, internalConfig);
497
512
  break;
498
513
  case "pptx":
499
- parsePowerPoint(filepath, internalCallback, config);
514
+ parsePowerPoint(filepath, internalCallback, internalConfig);
500
515
  break;
501
516
  case "xlsx":
502
- parseExcel(filepath, internalCallback, config);
517
+ parseExcel(filepath, internalCallback, internalConfig);
503
518
  break;
504
519
  case "odt":
505
520
  case "odp":
506
521
  case "ods":
507
- parseOpenOffice(filepath, internalCallback, config);
522
+ parseOpenOffice(filepath, internalCallback, internalConfig);
508
523
  break;
509
524
  case "pdf":
510
- parsePdf(filepath, internalCallback, config);
525
+ parsePdf(filepath, internalCallback, internalConfig);
511
526
  break;
512
527
 
513
528
  default:
@@ -517,29 +532,29 @@ function parseOffice(file, callback, config = {}) {
517
532
  /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
518
533
  function internalCallback(data, err) {
519
534
  if (err)
520
- consoleError(err, config.outputErrorToConsole)
535
+ consoleError(err, internalConfig.outputErrorToConsole)
521
536
  // Call the original callback
522
537
  callback(data, err);
523
538
  // Check if we need to preserve unzipped content files or delete them.
524
- if (config.preserveTempFiles)
539
+ if (internalConfig.preserveTempFiles)
525
540
  return;
526
541
  // Delete decompress sublocation.
527
- rimraf(decompressSubLocation, rimrafErr => consoleError(rimrafErr, config.outputErrorToConsole));
542
+ rimraf(internalConfig.tempFilesLocation, rimrafErr => consoleError(rimrafErr, internalConfig.outputErrorToConsole));
528
543
  }
529
544
  })
530
545
  .catch(error => {
531
- consoleError(error, config.outputErrorToConsole);
546
+ consoleError(error, internalConfig.outputErrorToConsole);
532
547
  callback(undefined, error);
533
548
  });
534
549
  }
535
550
 
536
551
  /**
537
552
  * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
538
- * @param {string | Buffer} file File path or file buffers
539
- * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
553
+ * @param {string | Buffer} file File path or file buffers
554
+ * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
540
555
  * @returns {Promise<string>}
541
556
  */
542
- function parseOfficeAsync (file, config) {
557
+ function parseOfficeAsync (file, config = {}) {
543
558
  return new Promise((res, rej) => {
544
559
  parseOffice(file, function (data, err) {
545
560
  if (err)
@@ -549,26 +564,10 @@ function parseOfficeAsync (file, config) {
549
564
  });
550
565
  }
551
566
 
552
- /**
553
- * Set decompression directory. The final decompressed data will be put inside officeDist folder within your directory
554
- * @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
555
- * @returns {void}
556
- */
557
- function setDecompressionLocation(newLocation) {
558
- if (newLocation != undefined) {
559
- newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
560
- if (fs.existsSync(newLocation))
561
- decompressSubLocation = newLocation;
562
- return;
563
- }
564
- consoleError(ERRORMSG.locationNotFound(newLocation), config.outputErrorToConsole);
565
- decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
566
- }
567
567
 
568
568
  // Export functions
569
- module.exports.parseOffice = parseOffice;
570
- module.exports.parseOfficeAsync = parseOfficeAsync;
571
- module.exports.setDecompressionLocation = setDecompressionLocation;
569
+ module.exports.parseOffice = parseOffice;
570
+ module.exports.parseOfficeAsync = parseOfficeAsync;
572
571
 
573
572
 
574
573
  // Run this library on CLI
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.0.0",
3
+ "version": "4.0.2",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [
@@ -42,11 +42,11 @@
42
42
  },
43
43
  "homepage": "https://github.com/harshankur/officeParser#readme",
44
44
  "dependencies": {
45
+ "@xmldom/xmldom": "^0.8.10",
45
46
  "decompress": "^4.2.0",
46
47
  "file-type": "^16.5.4",
47
- "rimraf": "^2.6.3",
48
- "xmldom": "^0.6.0",
49
- "pdf-parse": "^1.1.1"
48
+ "pdf-parse": "^1.1.1",
49
+ "rimraf": "^2.6.3"
50
50
  },
51
51
  "devDependencies": {
52
52
  "@types/decompress": "^4.2.6",
@@ -1,44 +1,42 @@
1
1
  #!/usr/bin/env node
2
2
  export type OfficeParserConfig = {
3
+ /**
4
+ * The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. Please ensure that this directory actually exists. Default is officeParsertemp.
5
+ */
6
+ tempFilesLocation?: string;
3
7
  /**
4
8
  * Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
5
9
  */
6
- preserveTempFiles: boolean;
10
+ preserveTempFiles?: boolean;
7
11
  /**
8
12
  * Flag to show all the logs to console in case of an error irrespective of your own handling.
9
13
  */
10
- outputErrorToConsole: boolean;
14
+ outputErrorToConsole?: boolean;
11
15
  /**
12
16
  * The delimiter used for every new line in places that allow multiline text like word. Default is \n.
13
17
  */
14
- newlineDelimiter: string;
18
+ newlineDelimiter?: string;
15
19
  /**
16
20
  * Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
17
21
  */
18
- ignoreNotes: boolean;
22
+ ignoreNotes?: boolean;
19
23
  /**
20
24
  * Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
21
25
  */
22
- putNotesAtLast: boolean;
26
+ putNotesAtLast?: boolean;
23
27
  };
24
28
  /** Main async function with callback to execute parseOffice for supported files
25
- * @param {string | Buffer} file File path or file buffers
26
- * @param {function} callback Callback function that returns value or error
27
- * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
29
+ * @param {string | Buffer} file File path or file buffers
30
+ * @param {function} callback Callback function that returns value or error
31
+ * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
28
32
  * @returns {void}
29
33
  */
30
34
  export function parseOffice(file: string | Buffer, callback: Function, config?: OfficeParserConfig): void;
31
35
  /**
32
36
  * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
33
- * @param {string | Buffer} file File path or file buffers
34
- * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
37
+ * @param {string | Buffer} file File path or file buffers
38
+ * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
35
39
  * @returns {Promise<string>}
36
40
  */
37
- export function parseOfficeAsync(file: string | Buffer, config: OfficeParserConfig): Promise<string>;
38
- /**
39
- * Set decompression directory. The final decompressed data will be put inside officeDist folder within your directory
40
- * @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
41
- * @returns {void}
42
- */
43
- export function setDecompressionLocation(newLocation: string): void;
41
+ export function parseOfficeAsync(file: string | Buffer, config?: OfficeParserConfig): Promise<string>;
44
42
  //# sourceMappingURL=officeParser.d.ts.map