officeparser 4.1.0 → 4.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/officeParser.js +69 -21
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -13,6 +13,7 @@ A Node.js library to parse text out of any office file.
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
#### Update
|
|
16
|
+
* 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
|
|
16
17
|
* 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
|
|
17
18
|
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
|
18
19
|
* 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
|
package/officeParser.js
CHANGED
|
@@ -126,6 +126,7 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
126
126
|
// Files regex that hold our content of interest
|
|
127
127
|
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
128
128
|
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
129
|
+
const slideNumberRegex = /lide(\d+)\.xml/;
|
|
129
130
|
|
|
130
131
|
/** The decompress location which contains the filename in it */
|
|
131
132
|
const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
|
|
@@ -134,6 +135,17 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
134
135
|
{ filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
|
|
135
136
|
)
|
|
136
137
|
.then(files => {
|
|
138
|
+
// Sort files by slide number and their notes (if any).
|
|
139
|
+
files.sort((a, b) => {
|
|
140
|
+
const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
|
|
141
|
+
const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
|
|
142
|
+
|
|
143
|
+
const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
|
|
144
|
+
const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
|
|
145
|
+
|
|
146
|
+
return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
|
|
147
|
+
});
|
|
148
|
+
|
|
137
149
|
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
138
150
|
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
|
|
139
151
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
@@ -218,18 +230,45 @@ function parseExcel(filepath, callback, config) {
|
|
|
218
230
|
})
|
|
219
231
|
// ********************************** excel xml files explanation ***************************************
|
|
220
232
|
// Structure of xmlContent of an excel file is a bit complex.
|
|
221
|
-
// We have a sharedStrings.xml file which has strings inside t tags
|
|
233
|
+
// We usually have a sharedStrings.xml file which has strings inside t tags
|
|
234
|
+
// However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
|
|
222
235
|
// Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
|
|
223
236
|
// Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
|
|
224
237
|
// If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
|
|
238
|
+
// However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
|
|
239
|
+
// We extract either the inline strings or use the value to get numbers of text from shared strings.
|
|
225
240
|
// Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
|
|
226
241
|
// ******************************************************************************************************
|
|
227
242
|
.then(xmlContentFilesObject => {
|
|
228
243
|
/** Store all the text content to respond */
|
|
229
244
|
let responseText = [];
|
|
230
245
|
|
|
231
|
-
/**
|
|
232
|
-
|
|
246
|
+
/** Function to check if the given c node is a valid inline string node. */
|
|
247
|
+
function isValidInlineStringCNode(cNode) {
|
|
248
|
+
// Initial check to see if the passed node is a cNode
|
|
249
|
+
if (cNode.tagName.toLowerCase() != 'c')
|
|
250
|
+
return false;
|
|
251
|
+
if (cNode.getAttribute("t") != 'inlineStr')
|
|
252
|
+
return false;
|
|
253
|
+
const childNodesNamedIs = cNode.getElementsByTagName('is');
|
|
254
|
+
if (childNodesNamedIs.length != 1)
|
|
255
|
+
return false;
|
|
256
|
+
const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
|
|
257
|
+
if (childNodesNamedT.length != 1)
|
|
258
|
+
return false;
|
|
259
|
+
return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/** Function to check if the given c node has a valid v node */
|
|
263
|
+
function hasValidVNodeInCNode(cNode) {
|
|
264
|
+
return cNode.getElementsByTagName("v")[0]
|
|
265
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
266
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
|
|
270
|
+
const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
|
|
271
|
+
: [];
|
|
233
272
|
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
234
273
|
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
235
274
|
.map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
|
|
@@ -238,25 +277,33 @@ function parseExcel(filepath, callback, config) {
|
|
|
238
277
|
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
239
278
|
/** Find text nodes with c tags in sharedStrings xml file */
|
|
240
279
|
const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
|
|
241
|
-
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
|
|
280
|
+
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
|
|
242
281
|
responseText.push(
|
|
243
282
|
Array.from(sheetsXmlCNodesList)
|
|
244
|
-
// Filter
|
|
245
|
-
.filter(cNode => cNode
|
|
246
|
-
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
247
|
-
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
|
|
283
|
+
// Filter out invalid c nodes
|
|
284
|
+
.filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
|
|
248
285
|
.map(cNode => {
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
//
|
|
254
|
-
if (
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
286
|
+
// Processing if this is a valid inline string c node.
|
|
287
|
+
if (isValidInlineStringCNode(cNode))
|
|
288
|
+
return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
|
|
289
|
+
|
|
290
|
+
// Processing if this c node has a valid v node.
|
|
291
|
+
if (hasValidVNodeInCNode(cNode)) {
|
|
292
|
+
/** Flag whether this node's value represents an index in the shared string array */
|
|
293
|
+
const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
|
|
294
|
+
/** Find value nodes represented by v tags */
|
|
295
|
+
const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
|
|
296
|
+
// Validate text
|
|
297
|
+
if (isIndexInSharedStrings && value >= sharedStrings.length)
|
|
298
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
299
|
+
|
|
300
|
+
return isIndexInSharedStrings
|
|
301
|
+
? sharedStrings[value]
|
|
302
|
+
: value;
|
|
303
|
+
}
|
|
304
|
+
// TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
|
|
305
|
+
// Not the case now but it could happen and it is better to be safe.
|
|
306
|
+
return '';
|
|
260
307
|
})
|
|
261
308
|
// Join each cell text within a sheet with a space.
|
|
262
309
|
.join(config.newlineDelimiter ?? "\n")
|
|
@@ -454,6 +501,7 @@ function parsePdf(filepath, callback, config) {
|
|
|
454
501
|
const responseText = textContentArray
|
|
455
502
|
.map(textContent => textContent.items) // Get all the items
|
|
456
503
|
.flat() // Flatten all the items object
|
|
504
|
+
.filter(item => item.str != '') // Ignore the empty string items.
|
|
457
505
|
.reduce((a, v) => (
|
|
458
506
|
{
|
|
459
507
|
text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
|
|
@@ -555,7 +603,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
555
603
|
break;
|
|
556
604
|
|
|
557
605
|
default:
|
|
558
|
-
|
|
606
|
+
internalCallback(undefined, ERRORMSG.extensionUnsupported(extension)); // Call the internalCallback function which removes the temp files if required.
|
|
559
607
|
}
|
|
560
608
|
|
|
561
609
|
/** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
|
|
@@ -633,7 +681,7 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
|
633
681
|
|
|
634
682
|
|
|
635
683
|
// Run this library on CLI
|
|
636
|
-
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
|
|
684
|
+
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser")) {
|
|
637
685
|
if (process.argv.length == 2) {
|
|
638
686
|
// continue
|
|
639
687
|
}
|
package/package.json
CHANGED