officeparser 4.0.5 → 4.0.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/officeParser.js +27 -11
  2. package/package.json +1 -1
package/officeParser.js CHANGED
@@ -69,7 +69,8 @@ function parseWord(filepath, callback, config) {
69
69
  { filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
70
70
  )
71
71
  .then(files => {
72
- if (files.length == 0)
72
+ // Verify if atleast the document xml file exists in the extracted files list.
73
+ if (!files.map(file => file.path).includes(mainContentFile))
73
74
  throw ERRORMSG.fileCorrupted(filepath);
74
75
 
75
76
  return [...files.filter(file => file.path == mainContentFile),
@@ -99,7 +100,10 @@ function parseWord(filepath, callback, config) {
99
100
  // Find text nodes with w:t tags
100
101
  const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
101
102
  // Join the texts within this paragraph node without any spaces or delimiters.
102
- return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
103
+ return Array.from(xmlTextNodeList)
104
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
105
+ .map(textNode => textNode.childNodes[0].nodeValue)
106
+ .join("");
103
107
  })
104
108
  // Join each paragraph text with a new line delimiter.
105
109
  .join(config.newlineDelimiter ?? "\n")
@@ -132,8 +136,8 @@ function parsePowerPoint(filepath, callback, config) {
132
136
  { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
133
137
  )
134
138
  .then(files => {
135
- // Check if files is corrupted
136
- if (files.length == 0)
139
+ // Verify if atleast the slides xml files exist in the extracted files list.
140
+ if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
137
141
  throw ERRORMSG.fileCorrupted(filepath);
138
142
 
139
143
  // Check if any sorting is required.
@@ -166,7 +170,10 @@ function parsePowerPoint(filepath, callback, config) {
166
170
  .map(paragraphNode => {
167
171
  /** Find text nodes with a:t tags */
168
172
  const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
169
- return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
173
+ return Array.from(xmlTextNodeList)
174
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
175
+ .map(textNode => textNode.childNodes[0].nodeValue)
176
+ .join("");
170
177
  })
171
178
  .join(config.newlineDelimiter ?? "\n")
172
179
  );
@@ -200,7 +207,8 @@ function parseExcel(filepath, callback, config) {
200
207
  { filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
201
208
  )
202
209
  .then(files => {
203
- if (files.length == 0)
210
+ // Verify if atleast the slides xml files exist in the extracted files list.
211
+ if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
204
212
  throw ERRORMSG.fileCorrupted(filepath);
205
213
 
206
214
  return {
@@ -225,7 +233,8 @@ function parseExcel(filepath, callback, config) {
225
233
  /** Find text nodes with t tags in sharedStrings xml file */
226
234
  const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
227
235
  /** Create shared string array. This will be used as a map to get strings from within sheet files. */
228
- const sharedStrings = Array.from(sharedStringsXmlTNodesList).map(tNode => tNode.childNodes[0].nodeValue);
236
+ const sharedStrings = Array.from(sharedStringsXmlTNodesList)
237
+ .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
229
238
 
230
239
  // Parse Sheet files
231
240
  xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
@@ -234,8 +243,10 @@ function parseExcel(filepath, callback, config) {
234
243
  // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
235
244
  responseText.push(
236
245
  Array.from(sheetsXmlCNodesList)
237
- // Filter c nodes than do not have any v nodes
238
- .filter(cNode => cNode.getElementsByTagName("v").length != 0)
246
+ // Filter c nodes than do not have any valid v nodes
247
+ .filter(cNode => cNode.getElementsByTagName("v")[0]
248
+ && cNode.getElementsByTagName("v")[0].childNodes[0]
249
+ && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
239
250
  .map(cNode => {
240
251
  /** Flag whether this node's value represents a string index */
241
252
  const isString = cNode.getAttribute("t") == "s";
@@ -266,7 +277,10 @@ function parseExcel(filepath, callback, config) {
266
277
  .map(paragraphNode => {
267
278
  /** Find text nodes with a:t tags */
268
279
  const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
269
- return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
280
+ return Array.from(xmlTextNodeList)
281
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
282
+ .map(textNode => textNode.childNodes[0].nodeValue)
283
+ .join("");
270
284
  })
271
285
  .join(config.newlineDelimiter ?? "\n")
272
286
  );
@@ -279,6 +293,7 @@ function parseExcel(filepath, callback, config) {
279
293
  /** Store all the text content to respond */
280
294
  responseText.push(
281
295
  Array.from(chartsXmlCVNodesList)
296
+ .filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
282
297
  .map(cVNode => cVNode.childNodes[0].nodeValue)
283
298
  .join(config.newlineDelimiter ?? "\n")
284
299
  );
@@ -311,7 +326,8 @@ function parseOpenOffice(filepath, callback, config) {
311
326
  { filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
312
327
  )
313
328
  .then(files => {
314
- if (files.length == 0)
329
+ // Verify if atleast the content xml file exists in the extracted files list.
330
+ if (!files.map(file => file.path).includes(mainContentFilePath))
315
331
  throw ERRORMSG.fileCorrupted(filepath);
316
332
 
317
333
  return {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.0.5",
3
+ "version": "4.0.7",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [