officeparser 3.0.0 → 3.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -5
- package/officeParser.js +38 -18
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -12,14 +12,15 @@ A Node.js library to parse text out of any office file.
|
|
|
12
12
|
|
|
13
13
|
|
|
14
14
|
#### Update
|
|
15
|
-
* 2022/12/
|
|
16
|
-
*
|
|
15
|
+
* 2022/12/24 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
|
|
16
|
+
* 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
|
|
17
|
+
* 2021/11/21 - Added promise way to existing callback functions.
|
|
17
18
|
* 2020/06/01 - Added error handling and console.log enable/disable methods. Default is set at enabled. Everything backward compatible.
|
|
18
19
|
* 2019/06/17 - Added method to change location for decompressing office files in places with restricted write access.
|
|
19
20
|
* 2019/04/30 - Removed case sensitive file extension bug. File names with capital lettered extensions now supported.
|
|
20
21
|
* 2019/04/23 - Added support for open office files *.odt, *.odp, *.ods through parseOffice function. Created a new method parseOpenOffice for those who prefer targetted functions.
|
|
21
|
-
* 2019/04/23 - Added feature to delete the generated dist folder after function callback
|
|
22
|
-
* 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension
|
|
22
|
+
* 2019/04/23 - Added feature to delete the generated dist folder after function callback.
|
|
23
|
+
* 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension.
|
|
23
24
|
* 2019/04/22 - Added file extension validations. Removed errors for excel files with no drawing elements.
|
|
24
25
|
* 2019/04/19 - Support added for *.xlsx files.
|
|
25
26
|
* 2019/04/18 - Support added for *.pptx files.
|
|
@@ -33,9 +34,21 @@ A Node.js library to parse text out of any office file.
|
|
|
33
34
|
npm i officeparser
|
|
34
35
|
```
|
|
35
36
|
|
|
37
|
+
## Command Line usage
|
|
38
|
+
If you have already installed officeParser, then follow below command
|
|
39
|
+
for extracting content from a file sampleFile
|
|
40
|
+
```
|
|
41
|
+
node officeParser.js <fileName>
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
If you have not installed the library, you can use npx to instantly extract parsed data.
|
|
45
|
+
```
|
|
46
|
+
npx officeparser <fileName>
|
|
47
|
+
```
|
|
48
|
+
|
|
36
49
|
----------
|
|
37
50
|
|
|
38
|
-
**Usage**
|
|
51
|
+
**Library Usage**
|
|
39
52
|
```js
|
|
40
53
|
const officeParser = require('officeparser');
|
|
41
54
|
|
package/officeParser.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
1
3
|
const decompress = require('decompress');
|
|
2
4
|
const xml2js = require('xml2js')
|
|
3
5
|
const fs = require('fs')
|
|
@@ -10,11 +12,13 @@ const ERRORMSG = {
|
|
|
10
12
|
extensionUnsupported: (ext) => `${ERRORHEADER}Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
11
13
|
fileCorrupted: (filename) => `${ERRORHEADER}Your file ${filename} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
12
14
|
fileDoesNotExist: (filename) => `${ERRORHEADER}File ${filename} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
13
|
-
locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location
|
|
15
|
+
locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
|
|
16
|
+
improperArguments: `${ERRORHEADER}Improper arguments`
|
|
14
17
|
}
|
|
18
|
+
/** Default sublocation for decompressing files under the current directory. */
|
|
19
|
+
const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
|
|
15
20
|
/** Location for decompressing files. Default is "officeDist" */
|
|
16
|
-
|
|
17
|
-
let decompressLocation = DECOMPRESSSUBLOCATION;
|
|
21
|
+
let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
18
22
|
/** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
|
|
19
23
|
let outputErrorToConsole = false;
|
|
20
24
|
|
|
@@ -74,7 +78,7 @@ function parseWord(filename, callback, deleteOfficeDist = true) {
|
|
|
74
78
|
|
|
75
79
|
const contentFile = 'word/document.xml';
|
|
76
80
|
decompress(filename,
|
|
77
|
-
|
|
81
|
+
decompressSubLocation,
|
|
78
82
|
{ filter: x => x.path == contentFile }
|
|
79
83
|
)
|
|
80
84
|
.then(files => {
|
|
@@ -83,14 +87,14 @@ function parseWord(filename, callback, deleteOfficeDist = true) {
|
|
|
83
87
|
return callback(undefined, ERRORMSG.fileCorrupted(filename));
|
|
84
88
|
}
|
|
85
89
|
|
|
86
|
-
return fs.readFileSync(`${
|
|
90
|
+
return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
|
|
87
91
|
})
|
|
88
92
|
.then(xmlContent => parseStringPromise(xmlContent))
|
|
89
93
|
.then(xmlObjects => {
|
|
90
94
|
extractTextFromWordXmlObjects(xmlObjects);
|
|
91
95
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
92
96
|
if (deleteOfficeDist)
|
|
93
|
-
rimraf(
|
|
97
|
+
rimraf(decompressSubLocation, err => {
|
|
94
98
|
if (err)
|
|
95
99
|
consoleError(err);
|
|
96
100
|
res();
|
|
@@ -152,7 +156,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
|
|
|
152
156
|
]
|
|
153
157
|
|
|
154
158
|
decompress(filename,
|
|
155
|
-
|
|
159
|
+
decompressSubLocation,
|
|
156
160
|
{ filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
|
|
157
161
|
)
|
|
158
162
|
.then(files => {
|
|
@@ -165,7 +169,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
|
|
|
165
169
|
}
|
|
166
170
|
|
|
167
171
|
// Returning an array of all the xml contents read using fs.readFileSync
|
|
168
|
-
return files.map(file => fs.readFileSync(`${
|
|
172
|
+
return files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8'))
|
|
169
173
|
})
|
|
170
174
|
.then(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent)))) // Returning an array of all parseStringPromise responses
|
|
171
175
|
.then(xmlObjectsArray => {
|
|
@@ -173,7 +177,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
|
|
|
173
177
|
|
|
174
178
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
175
179
|
if (deleteOfficeDist)
|
|
176
|
-
rimraf(
|
|
180
|
+
rimraf(decompressSubLocation, err => {
|
|
177
181
|
if (err)
|
|
178
182
|
consoleError(err);
|
|
179
183
|
res();
|
|
@@ -289,7 +293,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
|
|
|
289
293
|
]
|
|
290
294
|
|
|
291
295
|
decompress(filename,
|
|
292
|
-
|
|
296
|
+
decompressSubLocation,
|
|
293
297
|
{ filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
|
|
294
298
|
)
|
|
295
299
|
.then(files => {
|
|
@@ -303,7 +307,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
|
|
|
303
307
|
}
|
|
304
308
|
|
|
305
309
|
// Returning a 2dArray of all the xml contents read using fs.readFileSync and separated by array elements
|
|
306
|
-
return files2dArray.map(files => files.map(file => fs.readFileSync(`${
|
|
310
|
+
return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8')))
|
|
307
311
|
})
|
|
308
312
|
.then(xmlContent2dArray => Promise.all(xmlContent2dArray.map(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent, false)))))) // Returning a 2dArray of all parseStringPromise responses
|
|
309
313
|
.then(xmlObjects2dArray => {
|
|
@@ -311,7 +315,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
|
|
|
311
315
|
|
|
312
316
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
313
317
|
if (deleteOfficeDist)
|
|
314
|
-
rimraf(
|
|
318
|
+
rimraf(decompressSubLocation, err => {
|
|
315
319
|
if (err)
|
|
316
320
|
consoleError(err);
|
|
317
321
|
res();
|
|
@@ -368,7 +372,7 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
368
372
|
|
|
369
373
|
const contentFile = 'content.xml';
|
|
370
374
|
decompress(filename,
|
|
371
|
-
|
|
375
|
+
decompressSubLocation,
|
|
372
376
|
{ filter: x => x.path == contentFile }
|
|
373
377
|
)
|
|
374
378
|
.then(files => {
|
|
@@ -377,14 +381,14 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
377
381
|
return callback(undefined, ERRORMSG.fileCorrupted(filename));
|
|
378
382
|
}
|
|
379
383
|
|
|
380
|
-
return fs.readFileSync(`${
|
|
384
|
+
return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
|
|
381
385
|
})
|
|
382
386
|
.then(xmlContent => parseStringPromise(xmlContent))
|
|
383
387
|
.then(xmlObjects => {
|
|
384
388
|
extractTextFromOpenOfficeXmlObjects(xmlObjects);
|
|
385
389
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
386
390
|
if (deleteOfficeDist)
|
|
387
|
-
rimraf(
|
|
391
|
+
rimraf(decompressSubLocation, err => {
|
|
388
392
|
if (err)
|
|
389
393
|
consoleError(err);
|
|
390
394
|
res();
|
|
@@ -406,6 +410,10 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
406
410
|
|
|
407
411
|
/** Main async function with callback to execute parseOffice for supported files */
|
|
408
412
|
function parseOffice(filename, callback, deleteOfficeDist = true) {
|
|
413
|
+
if (!fs.existsSync(filename)) {
|
|
414
|
+
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
415
|
+
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
416
|
+
}
|
|
409
417
|
var extension = filename.split(".").pop().toLowerCase();
|
|
410
418
|
|
|
411
419
|
switch(extension)
|
|
@@ -437,14 +445,14 @@ function parseOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
437
445
|
*/
|
|
438
446
|
function setDecompressionLocation(newLocation) {
|
|
439
447
|
if (newLocation != undefined) {
|
|
440
|
-
newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${
|
|
448
|
+
newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
|
|
441
449
|
|
|
442
450
|
if (fs.existsSync(newLocation))
|
|
443
|
-
|
|
451
|
+
decompressSubLocation = newLocation;
|
|
444
452
|
return;
|
|
445
453
|
}
|
|
446
454
|
consoleError(ERRORMSG.locationNotFound(newLocation));
|
|
447
|
-
|
|
455
|
+
decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
448
456
|
}
|
|
449
457
|
|
|
450
458
|
/** Enable console output */
|
|
@@ -548,3 +556,15 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
|
548
556
|
module.exports.setDecompressionLocation = setDecompressionLocation;
|
|
549
557
|
module.exports.enableConsoleOutput = enableConsoleOutput;
|
|
550
558
|
module.exports.disableConsoleOutput = disableConsoleOutput;
|
|
559
|
+
|
|
560
|
+
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
|
|
561
|
+
if (process.argv.length == 2) {
|
|
562
|
+
// continue
|
|
563
|
+
}
|
|
564
|
+
else if (process.argv.length == 3)
|
|
565
|
+
parseOfficeAsync(process.argv[2])
|
|
566
|
+
.then(text => console.log(text))
|
|
567
|
+
.catch(error => console.error(error))
|
|
568
|
+
else
|
|
569
|
+
console.error(ERRORMSG.improperArguments)
|
|
570
|
+
}
|
package/package.json
CHANGED