officeparser 3.0.0 → 3.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -12,14 +12,15 @@ A Node.js library to parse text out of any office file.
12
12
 
13
13
 
14
14
  #### Update
15
- * 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling
16
- * 2021/11/21 - Added promise way to existing callback functions
15
+ * 2022/12/24 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
16
+ * 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
17
+ * 2021/11/21 - Added promise way to existing callback functions.
17
18
  * 2020/06/01 - Added error handling and console.log enable/disable methods. Default is set at enabled. Everything backward compatible.
18
19
  * 2019/06/17 - Added method to change location for decompressing office files in places with restricted write access.
19
20
  * 2019/04/30 - Removed case sensitive file extension bug. File names with capital lettered extensions now supported.
20
21
  * 2019/04/23 - Added support for open office files *.odt, *.odp, *.ods through parseOffice function. Created a new method parseOpenOffice for those who prefer targetted functions.
21
- * 2019/04/23 - Added feature to delete the generated dist folder after function callback
22
- * 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension
22
+ * 2019/04/23 - Added feature to delete the generated dist folder after function callback.
23
+ * 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension.
23
24
  * 2019/04/22 - Added file extension validations. Removed errors for excel files with no drawing elements.
24
25
  * 2019/04/19 - Support added for *.xlsx files.
25
26
  * 2019/04/18 - Support added for *.pptx files.
@@ -33,9 +34,21 @@ A Node.js library to parse text out of any office file.
33
34
  npm i officeparser
34
35
  ```
35
36
 
37
+ ## Command Line usage
38
+ If you have already installed officeParser, then follow below command
39
+ for extracting content from a file sampleFile
40
+ ```
41
+ node officeParser.js <fileName>
42
+ ```
43
+
44
+ If you have not installed the library, you can use npx to instantly extract parsed data.
45
+ ```
46
+ npx officeparser <fileName>
47
+ ```
48
+
36
49
  ----------
37
50
 
38
- **Usage**
51
+ **Library Usage**
39
52
  ```js
40
53
  const officeParser = require('officeparser');
41
54
 
package/officeParser.js CHANGED
@@ -1,3 +1,5 @@
1
+ #!/usr/bin/env node
2
+
1
3
  const decompress = require('decompress');
2
4
  const xml2js = require('xml2js')
3
5
  const fs = require('fs')
@@ -10,11 +12,13 @@ const ERRORMSG = {
10
12
  extensionUnsupported: (ext) => `${ERRORHEADER}Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
11
13
  fileCorrupted: (filename) => `${ERRORHEADER}Your file ${filename} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
12
14
  fileDoesNotExist: (filename) => `${ERRORHEADER}File ${filename} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
13
- locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`
15
+ locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
16
+ improperArguments: `${ERRORHEADER}Improper arguments`
14
17
  }
18
+ /** Default sublocation for decompressing files under the current directory. */
19
+ const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
15
20
  /** Location for decompressing files. Default is "officeDist" */
16
- const DECOMPRESSSUBLOCATION = "officeDist"
17
- let decompressLocation = DECOMPRESSSUBLOCATION;
21
+ let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
18
22
  /** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
19
23
  let outputErrorToConsole = false;
20
24
 
@@ -74,7 +78,7 @@ function parseWord(filename, callback, deleteOfficeDist = true) {
74
78
 
75
79
  const contentFile = 'word/document.xml';
76
80
  decompress(filename,
77
- decompressLocation,
81
+ decompressSubLocation,
78
82
  { filter: x => x.path == contentFile }
79
83
  )
80
84
  .then(files => {
@@ -83,14 +87,14 @@ function parseWord(filename, callback, deleteOfficeDist = true) {
83
87
  return callback(undefined, ERRORMSG.fileCorrupted(filename));
84
88
  }
85
89
 
86
- return fs.readFileSync(`${decompressLocation}/${contentFile}`, 'utf8');
90
+ return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
87
91
  })
88
92
  .then(xmlContent => parseStringPromise(xmlContent))
89
93
  .then(xmlObjects => {
90
94
  extractTextFromWordXmlObjects(xmlObjects);
91
95
  const returnCallbackPromise = new Promise((res, rej) => {
92
96
  if (deleteOfficeDist)
93
- rimraf(decompressLocation, err => {
97
+ rimraf(decompressSubLocation, err => {
94
98
  if (err)
95
99
  consoleError(err);
96
100
  res();
@@ -152,7 +156,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
152
156
  ]
153
157
 
154
158
  decompress(filename,
155
- decompressLocation,
159
+ decompressSubLocation,
156
160
  { filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
157
161
  )
158
162
  .then(files => {
@@ -165,7 +169,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
165
169
  }
166
170
 
167
171
  // Returning an array of all the xml contents read using fs.readFileSync
168
- return files.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))
172
+ return files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8'))
169
173
  })
170
174
  .then(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent)))) // Returning an array of all parseStringPromise responses
171
175
  .then(xmlObjectsArray => {
@@ -173,7 +177,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
173
177
 
174
178
  const returnCallbackPromise = new Promise((res, rej) => {
175
179
  if (deleteOfficeDist)
176
- rimraf(decompressLocation, err => {
180
+ rimraf(decompressSubLocation, err => {
177
181
  if (err)
178
182
  consoleError(err);
179
183
  res();
@@ -289,7 +293,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
289
293
  ]
290
294
 
291
295
  decompress(filename,
292
- decompressLocation,
296
+ decompressSubLocation,
293
297
  { filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
294
298
  )
295
299
  .then(files => {
@@ -303,7 +307,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
303
307
  }
304
308
 
305
309
  // Returning a 2dArray of all the xml contents read using fs.readFileSync and separated by array elements
306
- return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')))
310
+ return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8')))
307
311
  })
308
312
  .then(xmlContent2dArray => Promise.all(xmlContent2dArray.map(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent, false)))))) // Returning a 2dArray of all parseStringPromise responses
309
313
  .then(xmlObjects2dArray => {
@@ -311,7 +315,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
311
315
 
312
316
  const returnCallbackPromise = new Promise((res, rej) => {
313
317
  if (deleteOfficeDist)
314
- rimraf(decompressLocation, err => {
318
+ rimraf(decompressSubLocation, err => {
315
319
  if (err)
316
320
  consoleError(err);
317
321
  res();
@@ -368,7 +372,7 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
368
372
 
369
373
  const contentFile = 'content.xml';
370
374
  decompress(filename,
371
- decompressLocation,
375
+ decompressSubLocation,
372
376
  { filter: x => x.path == contentFile }
373
377
  )
374
378
  .then(files => {
@@ -377,14 +381,14 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
377
381
  return callback(undefined, ERRORMSG.fileCorrupted(filename));
378
382
  }
379
383
 
380
- return fs.readFileSync(`${decompressLocation}/${contentFile}`, 'utf8');
384
+ return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
381
385
  })
382
386
  .then(xmlContent => parseStringPromise(xmlContent))
383
387
  .then(xmlObjects => {
384
388
  extractTextFromOpenOfficeXmlObjects(xmlObjects);
385
389
  const returnCallbackPromise = new Promise((res, rej) => {
386
390
  if (deleteOfficeDist)
387
- rimraf(decompressLocation, err => {
391
+ rimraf(decompressSubLocation, err => {
388
392
  if (err)
389
393
  consoleError(err);
390
394
  res();
@@ -406,6 +410,10 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
406
410
 
407
411
  /** Main async function with callback to execute parseOffice for supported files */
408
412
  function parseOffice(filename, callback, deleteOfficeDist = true) {
413
+ if (!fs.existsSync(filename)) {
414
+ consoleError(ERRORMSG.fileDoesNotExist(filename));
415
+ return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
416
+ }
409
417
  var extension = filename.split(".").pop().toLowerCase();
410
418
 
411
419
  switch(extension)
@@ -437,14 +445,14 @@ function parseOffice(filename, callback, deleteOfficeDist = true) {
437
445
  */
438
446
  function setDecompressionLocation(newLocation) {
439
447
  if (newLocation != undefined) {
440
- newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DECOMPRESSSUBLOCATION}`
448
+ newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
441
449
 
442
450
  if (fs.existsSync(newLocation))
443
- decompressLocation = newLocation;
451
+ decompressSubLocation = newLocation;
444
452
  return;
445
453
  }
446
454
  consoleError(ERRORMSG.locationNotFound(newLocation));
447
- decompressLocation = DECOMPRESSSUBLOCATION;
455
+ decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
448
456
  }
449
457
 
450
458
  /** Enable console output */
@@ -548,3 +556,15 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
548
556
  module.exports.setDecompressionLocation = setDecompressionLocation;
549
557
  module.exports.enableConsoleOutput = enableConsoleOutput;
550
558
  module.exports.disableConsoleOutput = disableConsoleOutput;
559
+
560
+ if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
561
+ if (process.argv.length == 2) {
562
+ // continue
563
+ }
564
+ else if (process.argv.length == 3)
565
+ parseOfficeAsync(process.argv[2])
566
+ .then(text => console.log(text))
567
+ .catch(error => console.error(error))
568
+ else
569
+ console.error(ERRORMSG.improperArguments)
570
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "3.0.0",
3
+ "version": "3.1.1",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp and ods files.",
5
5
  "main": "officeParser.js",
6
6
  "scripts": {