officeparser 4.1.1 → 4.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/officeParser.js +125 -43
- package/package.json +3 -3
package/README.md
CHANGED
|
@@ -13,6 +13,8 @@ A Node.js library to parse text out of any office file.
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
#### Update
|
|
16
|
+
* 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
|
|
17
|
+
* 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
|
|
16
18
|
* 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
|
|
17
19
|
* 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
|
|
18
20
|
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
package/officeParser.js
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
|
-
const decompress
|
|
4
|
-
const fs
|
|
5
|
-
const
|
|
6
|
-
const fileType
|
|
7
|
-
const pdfjs
|
|
8
|
-
const { DOMParser }
|
|
3
|
+
const decompress = require('decompress');
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const { rimrafSync } = require('rimraf');
|
|
6
|
+
const fileType = require('file-type');
|
|
7
|
+
const pdfjs = require('./pdfjs-dist-build/pdf.js');
|
|
8
|
+
const { DOMParser } = require('@xmldom/xmldom');
|
|
9
9
|
|
|
10
10
|
/** Header for error messages */
|
|
11
11
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
@@ -126,6 +126,7 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
126
126
|
// Files regex that hold our content of interest
|
|
127
127
|
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
128
128
|
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
129
|
+
const slideNumberRegex = /lide(\d+)\.xml/;
|
|
129
130
|
|
|
130
131
|
/** The decompress location which contains the filename in it */
|
|
131
132
|
const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
|
|
@@ -134,6 +135,17 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
134
135
|
{ filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
|
|
135
136
|
)
|
|
136
137
|
.then(files => {
|
|
138
|
+
// Sort files by slide number and their notes (if any).
|
|
139
|
+
files.sort((a, b) => {
|
|
140
|
+
const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
|
|
141
|
+
const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
|
|
142
|
+
|
|
143
|
+
const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
|
|
144
|
+
const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
|
|
145
|
+
|
|
146
|
+
return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
|
|
147
|
+
});
|
|
148
|
+
|
|
137
149
|
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
138
150
|
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
|
|
139
151
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
@@ -218,18 +230,45 @@ function parseExcel(filepath, callback, config) {
|
|
|
218
230
|
})
|
|
219
231
|
// ********************************** excel xml files explanation ***************************************
|
|
220
232
|
// Structure of xmlContent of an excel file is a bit complex.
|
|
221
|
-
// We have a sharedStrings.xml file which has strings inside t tags
|
|
233
|
+
// We usually have a sharedStrings.xml file which has strings inside t tags
|
|
234
|
+
// However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
|
|
222
235
|
// Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
|
|
223
236
|
// Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
|
|
224
237
|
// If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
|
|
238
|
+
// However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
|
|
239
|
+
// We extract either the inline strings or use the value to get numbers of text from shared strings.
|
|
225
240
|
// Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
|
|
226
241
|
// ******************************************************************************************************
|
|
227
242
|
.then(xmlContentFilesObject => {
|
|
228
243
|
/** Store all the text content to respond */
|
|
229
244
|
let responseText = [];
|
|
230
245
|
|
|
231
|
-
/**
|
|
232
|
-
|
|
246
|
+
/** Function to check if the given c node is a valid inline string node. */
|
|
247
|
+
function isValidInlineStringCNode(cNode) {
|
|
248
|
+
// Initial check to see if the passed node is a cNode
|
|
249
|
+
if (cNode.tagName.toLowerCase() != 'c')
|
|
250
|
+
return false;
|
|
251
|
+
if (cNode.getAttribute("t") != 'inlineStr')
|
|
252
|
+
return false;
|
|
253
|
+
const childNodesNamedIs = cNode.getElementsByTagName('is');
|
|
254
|
+
if (childNodesNamedIs.length != 1)
|
|
255
|
+
return false;
|
|
256
|
+
const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
|
|
257
|
+
if (childNodesNamedT.length != 1)
|
|
258
|
+
return false;
|
|
259
|
+
return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/** Function to check if the given c node has a valid v node */
|
|
263
|
+
function hasValidVNodeInCNode(cNode) {
|
|
264
|
+
return cNode.getElementsByTagName("v")[0]
|
|
265
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
266
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
|
|
270
|
+
const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
|
|
271
|
+
: [];
|
|
233
272
|
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
234
273
|
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
235
274
|
.map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
|
|
@@ -238,25 +277,33 @@ function parseExcel(filepath, callback, config) {
|
|
|
238
277
|
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
239
278
|
/** Find text nodes with c tags in sharedStrings xml file */
|
|
240
279
|
const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
|
|
241
|
-
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
|
|
280
|
+
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
|
|
242
281
|
responseText.push(
|
|
243
282
|
Array.from(sheetsXmlCNodesList)
|
|
244
|
-
// Filter
|
|
245
|
-
.filter(cNode => cNode
|
|
246
|
-
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
247
|
-
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
|
|
283
|
+
// Filter out invalid c nodes
|
|
284
|
+
.filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
|
|
248
285
|
.map(cNode => {
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
//
|
|
254
|
-
if (
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
286
|
+
// Processing if this is a valid inline string c node.
|
|
287
|
+
if (isValidInlineStringCNode(cNode))
|
|
288
|
+
return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
|
|
289
|
+
|
|
290
|
+
// Processing if this c node has a valid v node.
|
|
291
|
+
if (hasValidVNodeInCNode(cNode)) {
|
|
292
|
+
/** Flag whether this node's value represents an index in the shared string array */
|
|
293
|
+
const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
|
|
294
|
+
/** Find value nodes represented by v tags */
|
|
295
|
+
const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
|
|
296
|
+
// Validate text
|
|
297
|
+
if (isIndexInSharedStrings && value >= sharedStrings.length)
|
|
298
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
299
|
+
|
|
300
|
+
return isIndexInSharedStrings
|
|
301
|
+
? sharedStrings[value]
|
|
302
|
+
: value;
|
|
303
|
+
}
|
|
304
|
+
// TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
|
|
305
|
+
// Not the case now but it could happen and it is better to be safe.
|
|
306
|
+
return '';
|
|
260
307
|
})
|
|
261
308
|
// Join each cell text within a sheet with a space.
|
|
262
309
|
.join(config.newlineDelimiter ?? "\n")
|
|
@@ -496,7 +543,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
496
543
|
}
|
|
497
544
|
|
|
498
545
|
// create temp file subdirectory if it does not exist
|
|
499
|
-
fs.mkdirSync(
|
|
546
|
+
fs.mkdirSync(getTempFilesDirectory(internalConfig.tempFilesLocation), { recursive: true });
|
|
500
547
|
|
|
501
548
|
// Check if buffer
|
|
502
549
|
if (Buffer.isBuffer(file)) {
|
|
@@ -505,11 +552,11 @@ function parseOffice(file, callback, config = {}) {
|
|
|
505
552
|
.then(data =>
|
|
506
553
|
{
|
|
507
554
|
// temp file name
|
|
508
|
-
const
|
|
555
|
+
const newFileName = getNewFileName(data.ext.toLowerCase());
|
|
509
556
|
// write new file
|
|
510
|
-
fs.writeFileSync(
|
|
557
|
+
fs.writeFileSync(getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName), file);
|
|
511
558
|
// resolve promise
|
|
512
|
-
res(
|
|
559
|
+
res(newFileName);
|
|
513
560
|
})
|
|
514
561
|
.catch(() => rej(ERRORMSG.improperBuffers));
|
|
515
562
|
return;
|
|
@@ -522,16 +569,18 @@ function parseOffice(file, callback, config = {}) {
|
|
|
522
569
|
throw ERRORMSG.fileDoesNotExist(file);
|
|
523
570
|
|
|
524
571
|
// temp file name
|
|
525
|
-
const
|
|
572
|
+
const newFileName = getNewFileName(file.split(".").pop().toLowerCase());
|
|
526
573
|
// Copy the file into a temp location with the temp name
|
|
527
|
-
fs.copyFileSync(file,
|
|
574
|
+
fs.copyFileSync(file, getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName))
|
|
528
575
|
// resolve promise
|
|
529
|
-
res(
|
|
576
|
+
res(newFileName);
|
|
530
577
|
});
|
|
531
578
|
|
|
532
579
|
// Process filePreparedPromise resolution.
|
|
533
580
|
filePreparedPromise
|
|
534
|
-
.then(
|
|
581
|
+
.then(filename => {
|
|
582
|
+
// The file path
|
|
583
|
+
const filepath = getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), filename);
|
|
535
584
|
// File extension. Already in lowercase when we prepared the temp file above.
|
|
536
585
|
const extension = filepath.split(".").pop();
|
|
537
586
|
|
|
@@ -562,9 +611,33 @@ function parseOffice(file, callback, config = {}) {
|
|
|
562
611
|
/** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
|
|
563
612
|
function internalCallback(data, err) {
|
|
564
613
|
// Check if we need to preserve unzipped content files or delete them.
|
|
565
|
-
if (!internalConfig.preserveTempFiles)
|
|
566
|
-
|
|
567
|
-
|
|
614
|
+
if (!internalConfig.preserveTempFiles) {
|
|
615
|
+
/** Safely delete location */
|
|
616
|
+
function safelyDeleteLocation(location, deleteDirIfEmpty) {
|
|
617
|
+
if (!fs.existsSync(location))
|
|
618
|
+
return;
|
|
619
|
+
|
|
620
|
+
if (!fs.lstatSync(location).isDirectory() // If not directory
|
|
621
|
+
|| !deleteDirIfEmpty // or if the deleteDirIfEmpty is false or undefined
|
|
622
|
+
|| fs.readdirSync(location).length == 0) { // or if it is true, we check if the contents are empty.
|
|
623
|
+
try {
|
|
624
|
+
rimrafSync(location);
|
|
625
|
+
}
|
|
626
|
+
catch(rimrafErr) {
|
|
627
|
+
consoleError(rimrafErr, internalConfig.outputErrorToConsole);
|
|
628
|
+
}
|
|
629
|
+
}
|
|
630
|
+
}
|
|
631
|
+
|
|
632
|
+
// We delete our file as well as the extracted files inside the office files.
|
|
633
|
+
// There is no extraction for pdf files because we don't support any unpacking of it.
|
|
634
|
+
// After removing files, we check if the folders containing them are empty.
|
|
635
|
+
// If yes, we remove those folders too.
|
|
636
|
+
safelyDeleteLocation(filepath);
|
|
637
|
+
safelyDeleteLocation(getFilePath(internalConfig.tempFilesLocation, filename));
|
|
638
|
+
safelyDeleteLocation(getTempFilesDirectory(internalConfig.tempFilesLocation), true);
|
|
639
|
+
safelyDeleteLocation(internalConfig.tempFilesLocation, true);
|
|
640
|
+
}
|
|
568
641
|
|
|
569
642
|
// Check if there is an error. Throw if there is an error.
|
|
570
643
|
if (err)
|
|
@@ -597,21 +670,30 @@ function parseOfficeAsync(file, config = {}) {
|
|
|
597
670
|
let globalFileNameIterator = 0;
|
|
598
671
|
/**
|
|
599
672
|
* File Name generator that takes the extension as an input and returns a file name that comprises a timestamp and an incrementing number
|
|
600
|
-
* to allow the files to be sorted in chronological order
|
|
601
|
-
*
|
|
602
|
-
* @param {string} ext
|
|
673
|
+
* to allow the files to be sorted in chronological order. We also prefix them with ppid and pid of the process to support
|
|
674
|
+
* common destination from worker threads as well as multiple processes running them together.
|
|
675
|
+
* @param {string} ext File extension for this new generated file name
|
|
603
676
|
* @returns {string}
|
|
604
677
|
*/
|
|
605
|
-
function getNewFileName(
|
|
678
|
+
function getNewFileName(ext) {
|
|
606
679
|
// Get the iterator part of the file name
|
|
607
680
|
let iteratorPart = (globalFileNameIterator++).toString().padStart(5, '0');
|
|
608
681
|
// We want the iterator part of the file name to be of 5 digits.
|
|
609
682
|
// Therefore, when the iterator crosses into 6 digits, we reset it to 0.
|
|
610
683
|
if (globalFileNameIterator > 99999)
|
|
611
684
|
globalFileNameIterator = 0;
|
|
685
|
+
// Return the file name with ppid and pid to allow unique names even with worker threads.
|
|
686
|
+
return `${process.ppid}_${process.pid}_${new Date().getTime().toString() + iteratorPart}.${ext}`;
|
|
687
|
+
}
|
|
688
|
+
|
|
689
|
+
/** Gets directory for storing files. */
|
|
690
|
+
function getTempFilesDirectory(root) {
|
|
691
|
+
return `${root}/tempfiles`;
|
|
692
|
+
}
|
|
612
693
|
|
|
613
|
-
|
|
614
|
-
|
|
694
|
+
/** Gets file path for the supplied directory and the file name. */
|
|
695
|
+
function getFilePath(directory, fileName) {
|
|
696
|
+
return `${directory}/${fileName}`;
|
|
615
697
|
}
|
|
616
698
|
|
|
617
699
|
/**
|
|
@@ -634,7 +716,7 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
|
634
716
|
|
|
635
717
|
|
|
636
718
|
// Run this library on CLI
|
|
637
|
-
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
|
|
719
|
+
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser")) {
|
|
638
720
|
if (process.argv.length == 2) {
|
|
639
721
|
// continue
|
|
640
722
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "4.
|
|
3
|
+
"version": "4.2.0",
|
|
4
4
|
"description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
|
|
5
5
|
"main": "officeParser.js",
|
|
6
6
|
"files": [
|
|
@@ -44,10 +44,10 @@
|
|
|
44
44
|
"homepage": "https://github.com/harshankur/officeParser#readme",
|
|
45
45
|
"dependencies": {
|
|
46
46
|
"@xmldom/xmldom": "^0.8.10",
|
|
47
|
-
"decompress": "^4.2.
|
|
47
|
+
"decompress": "^4.2.1",
|
|
48
48
|
"file-type": "^16.5.4",
|
|
49
49
|
"node-ensure": "^0.0.0",
|
|
50
|
-
"rimraf": "^
|
|
50
|
+
"rimraf": "^5.0.10"
|
|
51
51
|
},
|
|
52
52
|
"devDependencies": {
|
|
53
53
|
"@types/decompress": "^4.2.6",
|