officeparser 4.0.8 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/officeParser.js CHANGED
@@ -4,7 +4,7 @@ const decompress = require('decompress');
4
4
  const fs = require('fs');
5
5
  const rimraf = require('rimraf');
6
6
  const fileType = require('file-type');
7
- const pdfParse = require('pdf-parse');
7
+ const pdfjs = require('./pdfjs-dist-build/pdf.js');
8
8
  const { DOMParser } = require('@xmldom/xmldom');
9
9
 
10
10
  /** Header for error messages */
@@ -431,9 +431,6 @@ function parseOpenOffice(filepath, callback, config) {
431
431
  .catch(e => callback(undefined, e));
432
432
  }
433
433
 
434
- /** Header for error messages */
435
- const PDFPARSEERRORHEADER = "[pdf-parse]: ";
436
-
437
434
  /** Main function for parsing text from pdf files
438
435
  * @param {string} filepath File path
439
436
  * @param {function} callback Callback function that returns value or error
@@ -441,17 +438,35 @@ const PDFPARSEERRORHEADER = "[pdf-parse]: ";
441
438
  * @returns {void}
442
439
  */
443
440
  function parsePdf(filepath, callback, config) {
444
- // Get the data buffer for the given file path.
445
- const dataBuffer = fs.readFileSync(filepath);
446
-
447
- pdfParse(dataBuffer)
448
- .then(data => {
449
- let text = data.text;
450
- if (!!config.newlineDelimiter && config.newlineDelimiter != "\n")
451
- text = text.replaceAll("\n", config.newlineDelimiter)
452
- callback(text, undefined);
453
- })
454
- .catch(e => callback(undefined, PDFPARSEERRORHEADER + e));
441
+ // Get the pdfjs document for the filepath.
442
+ pdfjs.getDocument(filepath).promise
443
+ // We go through each page and build our text content promise array.
444
+ .then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).map(pageNr => document.getPage(pageNr).then(page => page.getTextContent()))))
445
+ // Each textContent item has property 'items' which is an array of objects.
446
+ // Each object element in the array has text stored in their 'str' key.
447
+ // The concatenation of str is what makes our pdf content.
448
+ // str already contains any space that was in the text.
449
+ // So, we only care about when to add the new line.
450
+ // That we determine using transform[5] value which is the y-coordinate of the item object.
451
+ // So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
452
+ .then(textContentArray => {
453
+ /** Store all the text content to respond */
454
+ const responseText = textContentArray
455
+ .map(textContent => textContent.items) // Get all the items
456
+ .flat() // Flatten all the items object
457
+ .reduce((a, v) => (
458
+ {
459
+ text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
460
+ transform5: v.transform[5]
461
+ }),
462
+ {
463
+ text: '',
464
+ transform5: undefined
465
+ }).text;
466
+
467
+ callback(responseText, undefined);
468
+ })
469
+ .catch(e => callback(undefined, e));
455
470
  }
456
471
 
457
472
  /** Main async function with callback to execute parseOffice for supported files
@@ -584,7 +599,7 @@ let globalFileNameIterator = 0;
584
599
  * to allow the files to be sorted in chronological order
585
600
  * @param {string} tempFilesLocation Directory whether this new file needs to be stored
586
601
  * @param {string} ext File extension for this new generated file name
587
- * @returns {string}
602
+ * @returns {string}
588
603
  */
589
604
  function getNewFileName(tempFilesLocation, ext) {
590
605
  // Get the iterator part of the file name
package/package.json CHANGED
@@ -1,11 +1,12 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.0.8",
3
+ "version": "4.1.0",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [
7
7
  "officeParser.js",
8
- "typings/officeParser.d.ts"
8
+ "typings/officeParser.d.ts",
9
+ "pdfjs-dist-build/*"
9
10
  ],
10
11
  "types": "typings/officeParser.d.ts",
11
12
  "scripts": {
@@ -45,7 +46,7 @@
45
46
  "@xmldom/xmldom": "^0.8.10",
46
47
  "decompress": "^4.2.0",
47
48
  "file-type": "^16.5.4",
48
- "pdf-parse": "^1.1.1",
49
+ "node-ensure": "^0.0.0",
49
50
  "rimraf": "^2.6.3"
50
51
  },
51
52
  "devDependencies": {