officeparser 4.1.1 → 4.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/officeParser.js +67 -20
  2. package/package.json +1 -1
package/officeParser.js CHANGED
@@ -126,6 +126,7 @@ function parsePowerPoint(filepath, callback, config) {
126
126
  // Files regex that hold our content of interest
127
127
  const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
128
128
  const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
129
+ const slideNumberRegex = /lide(\d+)\.xml/;
129
130
 
130
131
  /** The decompress location which contains the filename in it */
131
132
  const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
@@ -134,6 +135,17 @@ function parsePowerPoint(filepath, callback, config) {
134
135
  { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
135
136
  )
136
137
  .then(files => {
138
+ // Sort files by slide number and their notes (if any).
139
+ files.sort((a, b) => {
140
+ const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
141
+ const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
142
+
143
+ const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
144
+ const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
145
+
146
+ return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
147
+ });
148
+
137
149
  // Verify if atleast the slides xml files exist in the extracted files list.
138
150
  if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
139
151
  throw ERRORMSG.fileCorrupted(filepath);
@@ -218,18 +230,45 @@ function parseExcel(filepath, callback, config) {
218
230
  })
219
231
  // ********************************** excel xml files explanation ***************************************
220
232
  // Structure of xmlContent of an excel file is a bit complex.
221
- // We have a sharedStrings.xml file which has strings inside t tags
233
+ // We usually have a sharedStrings.xml file which has strings inside t tags
234
+ // However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
222
235
  // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
223
236
  // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
224
237
  // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
238
+ // However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
239
+ // We extract either the inline strings or use the value to get numbers of text from shared strings.
225
240
  // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
226
241
  // ******************************************************************************************************
227
242
  .then(xmlContentFilesObject => {
228
243
  /** Store all the text content to respond */
229
244
  let responseText = [];
230
245
 
231
- /** Find text nodes with t tags in sharedStrings xml file */
232
- const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
246
+ /** Function to check if the given c node is a valid inline string node. */
247
+ function isValidInlineStringCNode(cNode) {
248
+ // Initial check to see if the passed node is a cNode
249
+ if (cNode.tagName.toLowerCase() != 'c')
250
+ return false;
251
+ if (cNode.getAttribute("t") != 'inlineStr')
252
+ return false;
253
+ const childNodesNamedIs = cNode.getElementsByTagName('is');
254
+ if (childNodesNamedIs.length != 1)
255
+ return false;
256
+ const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
257
+ if (childNodesNamedT.length != 1)
258
+ return false;
259
+ return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
260
+ }
261
+
262
+ /** Function to check if the given c node has a valid v node */
263
+ function hasValidVNodeInCNode(cNode) {
264
+ return cNode.getElementsByTagName("v")[0]
265
+ && cNode.getElementsByTagName("v")[0].childNodes[0]
266
+ && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
267
+ }
268
+
269
+ /** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
270
+ const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
271
+ : [];
233
272
  /** Create shared string array. This will be used as a map to get strings from within sheet files. */
234
273
  const sharedStrings = Array.from(sharedStringsXmlTNodesList)
235
274
  .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
@@ -238,25 +277,33 @@ function parseExcel(filepath, callback, config) {
238
277
  xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
239
278
  /** Find text nodes with c tags in sharedStrings xml file */
240
279
  const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
241
- // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
280
+ // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
242
281
  responseText.push(
243
282
  Array.from(sheetsXmlCNodesList)
244
- // Filter c nodes than do not have any valid v nodes
245
- .filter(cNode => cNode.getElementsByTagName("v")[0]
246
- && cNode.getElementsByTagName("v")[0].childNodes[0]
247
- && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
283
+ // Filter out invalid c nodes
284
+ .filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
248
285
  .map(cNode => {
249
- /** Flag whether this node's value represents a string index */
250
- const isString = cNode.getAttribute("t") == "s";
251
- /** Find value nodes represented by v tags */
252
- const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
253
- // Validate text
254
- if (isString && value >= sharedStrings.length)
255
- throw ERRORMSG.fileCorrupted(filepath);
256
-
257
- return isString
258
- ? sharedStrings[value]
259
- : value;
286
+ // Processing if this is a valid inline string c node.
287
+ if (isValidInlineStringCNode(cNode))
288
+ return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
289
+
290
+ // Processing if this c node has a valid v node.
291
+ if (hasValidVNodeInCNode(cNode)) {
292
+ /** Flag whether this node's value represents an index in the shared string array */
293
+ const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
294
+ /** Find value nodes represented by v tags */
295
+ const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
296
+ // Validate text
297
+ if (isIndexInSharedStrings && value >= sharedStrings.length)
298
+ throw ERRORMSG.fileCorrupted(filepath);
299
+
300
+ return isIndexInSharedStrings
301
+ ? sharedStrings[value]
302
+ : value;
303
+ }
304
+ // TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
305
+ // Not the case now but it could happen and it is better to be safe.
306
+ return '';
260
307
  })
261
308
  // Join each cell text within a sheet with a space.
262
309
  .join(config.newlineDelimiter ?? "\n")
@@ -634,7 +681,7 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
634
681
 
635
682
 
636
683
  // Run this library on CLI
637
- if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop() == "officeparser")) {
684
+ if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser")) {
638
685
  if (process.argv.length == 2) {
639
686
  // continue
640
687
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.1.1",
3
+ "version": "4.1.2",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [