officeparser 4.1.1 → 4.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/officeParser.js +67 -20
- package/package.json +1 -1
package/officeParser.js
CHANGED
|
@@ -126,6 +126,7 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
126
126
|
// Files regex that hold our content of interest
|
|
127
127
|
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
128
128
|
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
129
|
+
const slideNumberRegex = /lide(\d+)\.xml/;
|
|
129
130
|
|
|
130
131
|
/** The decompress location which contains the filename in it */
|
|
131
132
|
const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
|
|
@@ -134,6 +135,17 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
134
135
|
{ filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
|
|
135
136
|
)
|
|
136
137
|
.then(files => {
|
|
138
|
+
// Sort files by slide number and their notes (if any).
|
|
139
|
+
files.sort((a, b) => {
|
|
140
|
+
const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
|
|
141
|
+
const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
|
|
142
|
+
|
|
143
|
+
const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
|
|
144
|
+
const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
|
|
145
|
+
|
|
146
|
+
return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
|
|
147
|
+
});
|
|
148
|
+
|
|
137
149
|
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
138
150
|
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
|
|
139
151
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
@@ -218,18 +230,45 @@ function parseExcel(filepath, callback, config) {
|
|
|
218
230
|
})
|
|
219
231
|
// ********************************** excel xml files explanation ***************************************
|
|
220
232
|
// Structure of xmlContent of an excel file is a bit complex.
|
|
221
|
-
// We have a sharedStrings.xml file which has strings inside t tags
|
|
233
|
+
// We usually have a sharedStrings.xml file which has strings inside t tags
|
|
234
|
+
// However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
|
|
222
235
|
// Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
|
|
223
236
|
// Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
|
|
224
237
|
// If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
|
|
238
|
+
// However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
|
|
239
|
+
// We extract either the inline strings or use the value to get numbers of text from shared strings.
|
|
225
240
|
// Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
|
|
226
241
|
// ******************************************************************************************************
|
|
227
242
|
.then(xmlContentFilesObject => {
|
|
228
243
|
/** Store all the text content to respond */
|
|
229
244
|
let responseText = [];
|
|
230
245
|
|
|
231
|
-
/**
|
|
232
|
-
|
|
246
|
+
/** Function to check if the given c node is a valid inline string node. */
|
|
247
|
+
function isValidInlineStringCNode(cNode) {
|
|
248
|
+
// Initial check to see if the passed node is a cNode
|
|
249
|
+
if (cNode.tagName.toLowerCase() != 'c')
|
|
250
|
+
return false;
|
|
251
|
+
if (cNode.getAttribute("t") != 'inlineStr')
|
|
252
|
+
return false;
|
|
253
|
+
const childNodesNamedIs = cNode.getElementsByTagName('is');
|
|
254
|
+
if (childNodesNamedIs.length != 1)
|
|
255
|
+
return false;
|
|
256
|
+
const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
|
|
257
|
+
if (childNodesNamedT.length != 1)
|
|
258
|
+
return false;
|
|
259
|
+
return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/** Function to check if the given c node has a valid v node */
|
|
263
|
+
function hasValidVNodeInCNode(cNode) {
|
|
264
|
+
return cNode.getElementsByTagName("v")[0]
|
|
265
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
266
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
|
|
270
|
+
const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
|
|
271
|
+
: [];
|
|
233
272
|
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
234
273
|
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
235
274
|
.map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
|
|
@@ -238,25 +277,33 @@ function parseExcel(filepath, callback, config) {
|
|
|
238
277
|
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
239
278
|
/** Find text nodes with c tags in sharedStrings xml file */
|
|
240
279
|
const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
|
|
241
|
-
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
|
|
280
|
+
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
|
|
242
281
|
responseText.push(
|
|
243
282
|
Array.from(sheetsXmlCNodesList)
|
|
244
|
-
// Filter
|
|
245
|
-
.filter(cNode => cNode
|
|
246
|
-
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
247
|
-
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
|
|
283
|
+
// Filter out invalid c nodes
|
|
284
|
+
.filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
|
|
248
285
|
.map(cNode => {
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
//
|
|
254
|
-
if (
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
286
|
+
// Processing if this is a valid inline string c node.
|
|
287
|
+
if (isValidInlineStringCNode(cNode))
|
|
288
|
+
return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
|
|
289
|
+
|
|
290
|
+
// Processing if this c node has a valid v node.
|
|
291
|
+
if (hasValidVNodeInCNode(cNode)) {
|
|
292
|
+
/** Flag whether this node's value represents an index in the shared string array */
|
|
293
|
+
const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
|
|
294
|
+
/** Find value nodes represented by v tags */
|
|
295
|
+
const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
|
|
296
|
+
// Validate text
|
|
297
|
+
if (isIndexInSharedStrings && value >= sharedStrings.length)
|
|
298
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
299
|
+
|
|
300
|
+
return isIndexInSharedStrings
|
|
301
|
+
? sharedStrings[value]
|
|
302
|
+
: value;
|
|
303
|
+
}
|
|
304
|
+
// TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
|
|
305
|
+
// Not the case now but it could happen and it is better to be safe.
|
|
306
|
+
return '';
|
|
260
307
|
})
|
|
261
308
|
// Join each cell text within a sheet with a space.
|
|
262
309
|
.join(config.newlineDelimiter ?? "\n")
|
|
@@ -634,7 +681,7 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
|
634
681
|
|
|
635
682
|
|
|
636
683
|
// Run this library on CLI
|
|
637
|
-
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
|
|
684
|
+
if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser")) {
|
|
638
685
|
if (process.argv.length == 2) {
|
|
639
686
|
// continue
|
|
640
687
|
}
|
package/package.json
CHANGED