officeparser 5.1.1 → 5.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/officeParser.js +33 -20
- package/package.json +2 -1
- package/pdfjs-dist-build/pdf.js +0 -17644
- package/pdfjs-dist-build/pdf.js.map +0 -1
- package/pdfjs-dist-build/pdf.worker.js +0 -45163
- package/pdfjs-dist-build/pdf.worker.js.map +0 -1
package/officeParser.js
CHANGED
|
@@ -6,9 +6,11 @@ const concat = require('concat-stream');
|
|
|
6
6
|
const { DOMParser } = require('@xmldom/xmldom');
|
|
7
7
|
const fileType = require('file-type');
|
|
8
8
|
const fs = require('fs');
|
|
9
|
-
const pdfjs = require('./pdfjs-dist-build/pdf.js');
|
|
10
9
|
const yauzl = require('yauzl');
|
|
11
10
|
|
|
11
|
+
/** Load pdfjs-dist once at module scope. This returns a Promise that resolves to the module. */
|
|
12
|
+
const pdfjsPromise = import('pdfjs-dist/legacy/build/pdf.mjs');
|
|
13
|
+
|
|
12
14
|
/** Header for error messages */
|
|
13
15
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
14
16
|
/** Error messages */
|
|
@@ -271,8 +273,10 @@ function parseExcel(file, callback, config) {
|
|
|
271
273
|
? sharedStrings[value]
|
|
272
274
|
: value;
|
|
273
275
|
}
|
|
274
|
-
//
|
|
275
|
-
// Not the case now but it could happen
|
|
276
|
+
// Should not reach here. If we do, it means we are not filtering out items that we are not ready to process.
|
|
277
|
+
// Not the case now but it could happen if we change the filtering logic without updating the processing logic.
|
|
278
|
+
// So, it is better to error out here.
|
|
279
|
+
handleError(`Invalid c node found in sheet xml content: ${cNode}`, callback, config.outputErrorToConsole);
|
|
276
280
|
return '';
|
|
277
281
|
})
|
|
278
282
|
// Join each cell text within a sheet with a space.
|
|
@@ -442,14 +446,17 @@ function parseOpenOffice(file, callback, config) {
|
|
|
442
446
|
* @param {string | Buffer} file File path or Buffers
|
|
443
447
|
* @param {function} callback Callback function that returns value or error
|
|
444
448
|
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
445
|
-
* @returns {void}
|
|
449
|
+
* @returns {Promise<void>}
|
|
446
450
|
*/
|
|
447
|
-
function parsePdf(file, callback, config) {
|
|
448
|
-
//
|
|
449
|
-
|
|
450
|
-
|
|
451
|
+
async function parsePdf(file, callback, config) {
|
|
452
|
+
// Wait for pdfjs module to be loaded once
|
|
453
|
+
const pdfjs = await pdfjsPromise;
|
|
454
|
+
|
|
455
|
+
// Get the pdfjs document for the filepath or Uint8Array buffers.
|
|
456
|
+
// pdfjs does not accept Buffers directly, so we convert them to Uint8Array.
|
|
457
|
+
pdfjs.getDocument(file instanceof Buffer ? new Uint8Array(file) : file).promise
|
|
451
458
|
// We go through each page and build our text content promise array.
|
|
452
|
-
.then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).
|
|
459
|
+
.then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => document.getPage(index + 1).then(page => page.getTextContent()))))
|
|
453
460
|
// Each textContent item has property 'items' which is an array of objects.
|
|
454
461
|
// Each object element in the array has text stored in their 'str' key.
|
|
455
462
|
// The concatenation of str is what makes our pdf content.
|
|
@@ -462,17 +469,23 @@ function parsePdf(file, callback, config) {
|
|
|
462
469
|
const responseText = textContentArray
|
|
463
470
|
.map(textContent => textContent.items) // Get all the items
|
|
464
471
|
.flat() // Flatten all the items object
|
|
465
|
-
.
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
472
|
+
.reduce((a, v) => (
|
|
473
|
+
// the items could be TextItem or a TextMarkedContent.
|
|
474
|
+
// We are only interested in the TextItem which has a str property.
|
|
475
|
+
'str' in v && v.str != ''
|
|
476
|
+
? {
|
|
477
|
+
text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
|
|
478
|
+
transform5: v.transform[5]
|
|
479
|
+
} : {
|
|
480
|
+
text: a.text,
|
|
481
|
+
transform5: a.transform5
|
|
482
|
+
}
|
|
483
|
+
),
|
|
484
|
+
{
|
|
485
|
+
text: '',
|
|
486
|
+
transform5: undefined
|
|
487
|
+
}).text;
|
|
488
|
+
|
|
476
489
|
callback(responseText, undefined);
|
|
477
490
|
})
|
|
478
491
|
.catch(e => callback(undefined, e));
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.2.0",
|
|
4
4
|
"description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
|
|
5
5
|
"main": "officeParser.js",
|
|
6
6
|
"files": [
|
|
@@ -47,6 +47,7 @@
|
|
|
47
47
|
"concat-stream": "^2.0.0",
|
|
48
48
|
"file-type": "^16.5.4",
|
|
49
49
|
"node-ensure": "^0.0.0",
|
|
50
|
+
"pdfjs-dist": "^5.3.31",
|
|
50
51
|
"yauzl": "^3.1.3"
|
|
51
52
|
},
|
|
52
53
|
"devDependencies": {
|