officeparser 3.0.0 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -5
- package/officeParser.js +36 -18
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -12,14 +12,15 @@ A Node.js library to parse text out of any office file.
|
|
|
12
12
|
|
|
13
13
|
|
|
14
14
|
#### Update
|
|
15
|
-
* 2022/12/
|
|
16
|
-
*
|
|
15
|
+
* 2022/12/24 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
|
|
16
|
+
* 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
|
|
17
|
+
* 2021/11/21 - Added promise way to existing callback functions.
|
|
17
18
|
* 2020/06/01 - Added error handling and console.log enable/disable methods. Default is set at enabled. Everything backward compatible.
|
|
18
19
|
* 2019/06/17 - Added method to change location for decompressing office files in places with restricted write access.
|
|
19
20
|
* 2019/04/30 - Removed case sensitive file extension bug. File names with capital lettered extensions now supported.
|
|
20
21
|
* 2019/04/23 - Added support for open office files *.odt, *.odp, *.ods through parseOffice function. Created a new method parseOpenOffice for those who prefer targetted functions.
|
|
21
|
-
* 2019/04/23 - Added feature to delete the generated dist folder after function callback
|
|
22
|
-
* 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension
|
|
22
|
+
* 2019/04/23 - Added feature to delete the generated dist folder after function callback.
|
|
23
|
+
* 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension.
|
|
23
24
|
* 2019/04/22 - Added file extension validations. Removed errors for excel files with no drawing elements.
|
|
24
25
|
* 2019/04/19 - Support added for *.xlsx files.
|
|
25
26
|
* 2019/04/18 - Support added for *.pptx files.
|
|
@@ -33,9 +34,21 @@ A Node.js library to parse text out of any office file.
|
|
|
33
34
|
npm i officeparser
|
|
34
35
|
```
|
|
35
36
|
|
|
37
|
+
## Command Line usage
|
|
38
|
+
If you have already installed officeParser, then follow below command
|
|
39
|
+
for extracting content from a file sampleFile
|
|
40
|
+
```
|
|
41
|
+
node officeParser.js <fileName>
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
If you have not installed the library, you can use npx to instantly extract parsed data.
|
|
45
|
+
```
|
|
46
|
+
npx officeparser <fileName>
|
|
47
|
+
```
|
|
48
|
+
|
|
36
49
|
----------
|
|
37
50
|
|
|
38
|
-
**Usage**
|
|
51
|
+
**Library Usage**
|
|
39
52
|
```js
|
|
40
53
|
const officeParser = require('officeparser');
|
|
41
54
|
|
package/officeParser.js
CHANGED
|
@@ -10,11 +10,13 @@ const ERRORMSG = {
|
|
|
10
10
|
extensionUnsupported: (ext) => `${ERRORHEADER}Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
11
11
|
fileCorrupted: (filename) => `${ERRORHEADER}Your file ${filename} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
12
12
|
fileDoesNotExist: (filename) => `${ERRORHEADER}File ${filename} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
13
|
-
locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location
|
|
13
|
+
locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
|
|
14
|
+
improperArguments: `${ERRORHEADER}Improper arguments`
|
|
14
15
|
}
|
|
16
|
+
/** Default sublocation for decompressing files under the current directory. */
|
|
17
|
+
const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
|
|
15
18
|
/** Location for decompressing files. Default is "officeDist" */
|
|
16
|
-
|
|
17
|
-
let decompressLocation = DECOMPRESSSUBLOCATION;
|
|
19
|
+
let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
18
20
|
/** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
|
|
19
21
|
let outputErrorToConsole = false;
|
|
20
22
|
|
|
@@ -74,7 +76,7 @@ function parseWord(filename, callback, deleteOfficeDist = true) {
|
|
|
74
76
|
|
|
75
77
|
const contentFile = 'word/document.xml';
|
|
76
78
|
decompress(filename,
|
|
77
|
-
|
|
79
|
+
decompressSubLocation,
|
|
78
80
|
{ filter: x => x.path == contentFile }
|
|
79
81
|
)
|
|
80
82
|
.then(files => {
|
|
@@ -83,14 +85,14 @@ function parseWord(filename, callback, deleteOfficeDist = true) {
|
|
|
83
85
|
return callback(undefined, ERRORMSG.fileCorrupted(filename));
|
|
84
86
|
}
|
|
85
87
|
|
|
86
|
-
return fs.readFileSync(`${
|
|
88
|
+
return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
|
|
87
89
|
})
|
|
88
90
|
.then(xmlContent => parseStringPromise(xmlContent))
|
|
89
91
|
.then(xmlObjects => {
|
|
90
92
|
extractTextFromWordXmlObjects(xmlObjects);
|
|
91
93
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
92
94
|
if (deleteOfficeDist)
|
|
93
|
-
rimraf(
|
|
95
|
+
rimraf(decompressSubLocation, err => {
|
|
94
96
|
if (err)
|
|
95
97
|
consoleError(err);
|
|
96
98
|
res();
|
|
@@ -152,7 +154,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
|
|
|
152
154
|
]
|
|
153
155
|
|
|
154
156
|
decompress(filename,
|
|
155
|
-
|
|
157
|
+
decompressSubLocation,
|
|
156
158
|
{ filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
|
|
157
159
|
)
|
|
158
160
|
.then(files => {
|
|
@@ -165,7 +167,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
|
|
|
165
167
|
}
|
|
166
168
|
|
|
167
169
|
// Returning an array of all the xml contents read using fs.readFileSync
|
|
168
|
-
return files.map(file => fs.readFileSync(`${
|
|
170
|
+
return files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8'))
|
|
169
171
|
})
|
|
170
172
|
.then(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent)))) // Returning an array of all parseStringPromise responses
|
|
171
173
|
.then(xmlObjectsArray => {
|
|
@@ -173,7 +175,7 @@ function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
|
|
|
173
175
|
|
|
174
176
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
175
177
|
if (deleteOfficeDist)
|
|
176
|
-
rimraf(
|
|
178
|
+
rimraf(decompressSubLocation, err => {
|
|
177
179
|
if (err)
|
|
178
180
|
consoleError(err);
|
|
179
181
|
res();
|
|
@@ -289,7 +291,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
|
|
|
289
291
|
]
|
|
290
292
|
|
|
291
293
|
decompress(filename,
|
|
292
|
-
|
|
294
|
+
decompressSubLocation,
|
|
293
295
|
{ filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
|
|
294
296
|
)
|
|
295
297
|
.then(files => {
|
|
@@ -303,7 +305,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
|
|
|
303
305
|
}
|
|
304
306
|
|
|
305
307
|
// Returning a 2dArray of all the xml contents read using fs.readFileSync and separated by array elements
|
|
306
|
-
return files2dArray.map(files => files.map(file => fs.readFileSync(`${
|
|
308
|
+
return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8')))
|
|
307
309
|
})
|
|
308
310
|
.then(xmlContent2dArray => Promise.all(xmlContent2dArray.map(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent, false)))))) // Returning a 2dArray of all parseStringPromise responses
|
|
309
311
|
.then(xmlObjects2dArray => {
|
|
@@ -311,7 +313,7 @@ function parseExcel(filename, callback, deleteOfficeDist = true) {
|
|
|
311
313
|
|
|
312
314
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
313
315
|
if (deleteOfficeDist)
|
|
314
|
-
rimraf(
|
|
316
|
+
rimraf(decompressSubLocation, err => {
|
|
315
317
|
if (err)
|
|
316
318
|
consoleError(err);
|
|
317
319
|
res();
|
|
@@ -368,7 +370,7 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
368
370
|
|
|
369
371
|
const contentFile = 'content.xml';
|
|
370
372
|
decompress(filename,
|
|
371
|
-
|
|
373
|
+
decompressSubLocation,
|
|
372
374
|
{ filter: x => x.path == contentFile }
|
|
373
375
|
)
|
|
374
376
|
.then(files => {
|
|
@@ -377,14 +379,14 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
377
379
|
return callback(undefined, ERRORMSG.fileCorrupted(filename));
|
|
378
380
|
}
|
|
379
381
|
|
|
380
|
-
return fs.readFileSync(`${
|
|
382
|
+
return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
|
|
381
383
|
})
|
|
382
384
|
.then(xmlContent => parseStringPromise(xmlContent))
|
|
383
385
|
.then(xmlObjects => {
|
|
384
386
|
extractTextFromOpenOfficeXmlObjects(xmlObjects);
|
|
385
387
|
const returnCallbackPromise = new Promise((res, rej) => {
|
|
386
388
|
if (deleteOfficeDist)
|
|
387
|
-
rimraf(
|
|
389
|
+
rimraf(decompressSubLocation, err => {
|
|
388
390
|
if (err)
|
|
389
391
|
consoleError(err);
|
|
390
392
|
res();
|
|
@@ -406,6 +408,10 @@ function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
406
408
|
|
|
407
409
|
/** Main async function with callback to execute parseOffice for supported files */
|
|
408
410
|
function parseOffice(filename, callback, deleteOfficeDist = true) {
|
|
411
|
+
if (!fs.existsSync(filename)) {
|
|
412
|
+
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
413
|
+
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
414
|
+
}
|
|
409
415
|
var extension = filename.split(".").pop().toLowerCase();
|
|
410
416
|
|
|
411
417
|
switch(extension)
|
|
@@ -437,14 +443,14 @@ function parseOffice(filename, callback, deleteOfficeDist = true) {
|
|
|
437
443
|
*/
|
|
438
444
|
function setDecompressionLocation(newLocation) {
|
|
439
445
|
if (newLocation != undefined) {
|
|
440
|
-
newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${
|
|
446
|
+
newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
|
|
441
447
|
|
|
442
448
|
if (fs.existsSync(newLocation))
|
|
443
|
-
|
|
449
|
+
decompressSubLocation = newLocation;
|
|
444
450
|
return;
|
|
445
451
|
}
|
|
446
452
|
consoleError(ERRORMSG.locationNotFound(newLocation));
|
|
447
|
-
|
|
453
|
+
decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
448
454
|
}
|
|
449
455
|
|
|
450
456
|
/** Enable console output */
|
|
@@ -548,3 +554,15 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
|
548
554
|
module.exports.setDecompressionLocation = setDecompressionLocation;
|
|
549
555
|
module.exports.enableConsoleOutput = enableConsoleOutput;
|
|
550
556
|
module.exports.disableConsoleOutput = disableConsoleOutput;
|
|
557
|
+
|
|
558
|
+
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
|
|
559
|
+
if (process.argv.length == 2) {
|
|
560
|
+
// continue
|
|
561
|
+
}
|
|
562
|
+
else if (process.argv.length == 3)
|
|
563
|
+
parseOfficeAsync(process.argv[2])
|
|
564
|
+
.then(text => console.log(text))
|
|
565
|
+
.catch(error => console.error(error))
|
|
566
|
+
else
|
|
567
|
+
console.error(ERRORMSG.improperArguments)
|
|
568
|
+
}
|
package/package.json
CHANGED