officeparser 5.2.0 → 5.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/officeParser.js CHANGED
@@ -8,9 +8,6 @@ const fileType = require('file-type');
8
8
  const fs = require('fs');
9
9
  const yauzl = require('yauzl');
10
10
 
11
- /** Load pdfjs-dist once at module scope. This returns a Promise that resolves to the module. */
12
- const pdfjsPromise = import('pdfjs-dist/legacy/build/pdf.mjs');
13
-
14
11
  /** Header for error messages */
15
12
  const ERRORHEADER = "[OfficeParser]: ";
16
13
  /** Error messages */
@@ -238,12 +235,19 @@ function parseExcel(file, callback, config) {
238
235
  && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
239
236
  }
240
237
 
241
- /** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
242
- const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
243
- : [];
238
+ /** Find text nodes with t tags in sharedStrings.xml file. If the sharedStringsFile is not present, we return an empty array. */
239
+ const sharedStringsXmlSiNodesList = xmlContentFilesObject.sharedStringsFile != undefined
240
+ ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("si")
241
+ : [];
242
+
244
243
  /** Create shared string array. This will be used as a map to get strings from within sheet files. */
245
- const sharedStrings = Array.from(sharedStringsXmlTNodesList)
246
- .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
244
+ const sharedStrings = Array.from(sharedStringsXmlSiNodesList)
245
+ .map(siNode => {
246
+ // Concatenate all <t> nodes within the <si> node
247
+ return Array.from(siNode.getElementsByTagName("t"))
248
+ .map(tNode => tNode.childNodes[0]?.nodeValue ?? '') // Extract text content from each <t> node
249
+ .join(''); // Combine all <t> node text into a single string
250
+ });
247
251
 
248
252
  // Parse Sheet files
249
253
  xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
@@ -264,13 +268,14 @@ function parseExcel(file, callback, config) {
264
268
  /** Flag whether this node's value represents an index in the shared string array */
265
269
  const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
266
270
  /** Find value nodes represented by v tags */
267
- const value = parseInt(cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue, 10);
271
+ const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
272
+ const valueAsIndex = Number(value);
268
273
  // Validate text
269
- if (isIndexInSharedStrings && value >= sharedStrings.length)
274
+ if (isIndexInSharedStrings && (valueAsIndex != parseInt(value, 10) || valueAsIndex >= sharedStrings.length))
270
275
  throw ERRORMSG.fileCorrupted(file);
271
276
 
272
277
  return isIndexInSharedStrings
273
- ? sharedStrings[value]
278
+ ? sharedStrings[valueAsIndex]
274
279
  : value;
275
280
  }
276
281
  // Should not reach here. If we do, it means we are not filtering out items that we are not ready to process.
@@ -375,13 +380,19 @@ function parseOpenOffice(file, callback, config) {
375
380
  function traversal(node, xmlTextArray, isFirstRecursion) {
376
381
  if (!node.childNodes || node.childNodes.length == 0) {
377
382
  if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
383
+ // If the corresponding value is of type float, we take the value from office:value attribute.
384
+ // However, it is not on the parentNode but rather grandparentNode.
385
+ const value = node.parentNode.parentNode?.getAttribute('office:value-type') == 'float'
386
+ ? Number(node.parentNode.parentNode.getAttribute('office:value'))
387
+ : node.nodeValue;
388
+
378
389
  if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
379
- notesText.push(node.nodeValue);
390
+ notesText.push(value);
380
391
  if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
381
392
  notesText.push(config.newlineDelimiter ?? "\n");
382
393
  }
383
394
  else {
384
- xmlTextArray.push(node.nodeValue);
395
+ xmlTextArray.push(value);
385
396
  if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
386
397
  xmlTextArray.push(config.newlineDelimiter ?? "\n");
387
398
  }
@@ -450,7 +461,8 @@ function parseOpenOffice(file, callback, config) {
450
461
  */
451
462
  async function parsePdf(file, callback, config) {
452
463
  // Wait for pdfjs module to be loaded once
453
- const pdfjs = await pdfjsPromise;
464
+ // Lazy import pdfjs to avoid Node startup issues for environments that don't use PDF parsing
465
+ const pdfjs = await import('pdfjs-dist/legacy/build/pdf.mjs');
454
466
 
455
467
  // Get the pdfjs document for the filepath or Uint8Array buffers.
456
468
  // pdfjs does not accept Buffers directly, so we convert them to Uint8Array.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "5.2.0",
3
+ "version": "5.2.2",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [
@@ -18,16 +18,16 @@ export type OfficeParserConfig = {
18
18
  putNotesAtLast?: boolean;
19
19
  };
20
20
  /** Main async function with callback to execute parseOffice for supported files
21
- * @param {string | Buffer} file File path or file buffers
22
- * @param {function} callback Callback function that returns value or error
23
- * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
21
+ * @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
22
+ * @param {function} callback Callback function that returns value or error
23
+ * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
24
24
  * @returns {void}
25
25
  */
26
- export function parseOffice(file: string | Buffer, callback: Function, config?: OfficeParserConfig): void;
26
+ export function parseOffice(srcFile: string | Buffer | ArrayBuffer, callback: Function, config?: OfficeParserConfig): void;
27
27
  /** Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
28
- * @param {string | Buffer} file File path or file buffers
29
- * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
28
+ * @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
29
+ * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
30
30
  * @returns {Promise<string>}
31
31
  */
32
- export function parseOfficeAsync(file: string | Buffer, config?: OfficeParserConfig): Promise<string>;
32
+ export function parseOfficeAsync(srcFile: string | Buffer | ArrayBuffer, config?: OfficeParserConfig): Promise<string>;
33
33
  //# sourceMappingURL=officeParser.d.ts.map