officeparser 3.3.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -214
- package/officeParser.js +445 -539
- package/package.json +7 -3
- package/typings/officeParser.d.ts +29 -67
package/README.md
CHANGED
|
@@ -9,9 +9,11 @@ A Node.js library to parse text out of any office file.
|
|
|
9
9
|
- [`odt`](https://en.wikipedia.org/wiki/OpenDocument)
|
|
10
10
|
- [`odp`](https://en.wikipedia.org/wiki/OpenDocument)
|
|
11
11
|
- [`ods`](https://en.wikipedia.org/wiki/OpenDocument)
|
|
12
|
+
- [`pdf`](https://en.wikipedia.org/wiki/PDF)
|
|
12
13
|
|
|
13
14
|
|
|
14
15
|
#### Update
|
|
16
|
+
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
|
15
17
|
* 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
|
|
16
18
|
* 2023/04/07 - Added typings to methods to help with Typescript projects.
|
|
17
19
|
* 2022/12/28 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
|
|
@@ -47,24 +49,26 @@ Otherwise, you can simply use npx to instantly extract parsed data.
|
|
|
47
49
|
npx officeparser <fileName>
|
|
48
50
|
```
|
|
49
51
|
|
|
50
|
-
----------
|
|
51
52
|
|
|
52
|
-
|
|
53
|
+
## Library Usage
|
|
53
54
|
```js
|
|
54
55
|
const officeParser = require('officeparser');
|
|
55
56
|
|
|
56
57
|
// callback
|
|
57
|
-
officeParser.parseOffice("/path/to/officeFile", function(data, err){
|
|
58
|
+
officeParser.parseOffice("/path/to/officeFile", function(data, err) {
|
|
58
59
|
// "data" string in the callback here is the text parsed from the office file passed in the first argument above
|
|
59
|
-
if (err)
|
|
60
|
-
|
|
60
|
+
if (err) {
|
|
61
|
+
console.log(err);
|
|
62
|
+
return;
|
|
63
|
+
}
|
|
64
|
+
console.log(data);
|
|
61
65
|
})
|
|
62
66
|
|
|
63
67
|
// promise
|
|
64
68
|
officeParser.parseOfficeAsync("/path/to/officeFile");
|
|
65
69
|
// "data" string in the promise here is the text parsed from the office file passed in the argument above
|
|
66
|
-
.then(
|
|
67
|
-
.catch(
|
|
70
|
+
.then(data => console.log(data))
|
|
71
|
+
.catch(err => console.error(err))
|
|
68
72
|
|
|
69
73
|
// async/await
|
|
70
74
|
try {
|
|
@@ -81,252 +85,129 @@ try {
|
|
|
81
85
|
// on parseOffice or parseOfficeAsync functions.
|
|
82
86
|
|
|
83
87
|
// get file buffers
|
|
84
|
-
const fileBuffers = fs.
|
|
88
|
+
const fileBuffers = fs.readFileSync("/path/to/officeFile");
|
|
85
89
|
// get parsed text from officeParser
|
|
86
90
|
// NOTE: Only works with parseOffice. Old functions are not supported.
|
|
87
91
|
officeParser.parseOfficeAsync(fileBuffers);
|
|
88
|
-
.then(
|
|
89
|
-
.catch(
|
|
92
|
+
.then(data => console.log(data))
|
|
93
|
+
.catch(err => console.error(err))
|
|
90
94
|
```
|
|
91
95
|
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
96
|
+
### Configuration Object: OfficeParserConfig
|
|
97
|
+
*Optionally add a config object as 3rd variable to parseOffice for the following configurations*
|
|
98
|
+
| flag | datatype | explanation |
|
|
99
|
+
|----------------------|----------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|
100
|
+
| preserveTempFiles | boolean | Flag to not delete the internal content files and the possible duplicate temp files that it uses after unzipping office files. Default is false. It always deletes all of those files. |
|
|
101
|
+
| outputErrorToConsole | boolean | Flag to show all the logs to console in case of an error. |
|
|
102
|
+
| newlineDelimiter | string | The delimiter used for every new line in places that allow multiline text like word. Default is \n. |
|
|
103
|
+
| ignoreNotes | boolean | Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default. |
|
|
104
|
+
| putNotesAtLast | boolean | Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored. |
|
|
95
105
|
|
|
96
106
|
```js
|
|
97
|
-
const
|
|
98
|
-
|
|
99
|
-
//
|
|
100
|
-
|
|
101
|
-
officeParser.setDecompressionLocation("/tmp"); // New decompression location would be "/tmp/officeDist"
|
|
102
|
-
|
|
103
|
-
// P.S.: Setting location on a Windows environment with '\' hierarchy requires to be entered twice '\\'
|
|
104
|
-
officeParser.setDecompressionLocation("C:\\tmp"); // New decompression location would be "C:\tmp\officeDist"
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
officeParser.parseOffice("/path/to/officeFile", function(data, err){
|
|
108
|
-
// "data" string in the callback here is the text parsed from the office file passed in the first argument above
|
|
109
|
-
if (err) return console.log(err);
|
|
110
|
-
console.log(data)
|
|
111
|
-
})
|
|
112
|
-
```
|
|
113
|
-
|
|
114
|
-
*Optionally add false as 3rd variable to parseOffice to not delete the generated officeDist folder*
|
|
107
|
+
const config = {
|
|
108
|
+
newlineDelimiter: " ", // Separate new lines with a space instead of the default \n.
|
|
109
|
+
ignoreNotes: true // Ignore notes while parsing presentation files like pptx or odp.
|
|
110
|
+
}
|
|
115
111
|
|
|
116
|
-
```js
|
|
117
112
|
// callback
|
|
118
113
|
officeParser.parseOffice("/path/to/officeFile", function(data, err){
|
|
119
|
-
if (err)
|
|
120
|
-
|
|
121
|
-
|
|
114
|
+
if (err) {
|
|
115
|
+
console.log(err);
|
|
116
|
+
return;
|
|
117
|
+
}
|
|
118
|
+
console.log(data);
|
|
119
|
+
}, config)
|
|
122
120
|
|
|
123
121
|
// promise
|
|
124
|
-
officeParser.parseOfficeAsync("/path/to/officeFile",
|
|
125
|
-
.then((data) => console.log(data))
|
|
126
|
-
.catch((err) => console.error(err))
|
|
127
|
-
|
|
128
|
-
// async/await
|
|
129
|
-
try {
|
|
130
|
-
const data = await officeParser.parseOfficeAsync("/path/to/officeFile", false);
|
|
131
|
-
console.log(data);
|
|
132
|
-
} catch (err) {
|
|
133
|
-
// resolve error
|
|
134
|
-
console.log(err);
|
|
135
|
-
}
|
|
122
|
+
officeParser.parseOfficeAsync("/path/to/officeFile", config);
|
|
123
|
+
.then((data) => console.log(data))
|
|
124
|
+
.catch((err) => console.error(err))
|
|
136
125
|
```
|
|
137
126
|
|
|
138
|
-
**Example**
|
|
127
|
+
**Example - JavaScript**
|
|
139
128
|
```js
|
|
140
129
|
const officeParser = require('officeparser');
|
|
141
130
|
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
callSomeOtherFunction(newText);
|
|
147
|
-
})
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
// promise
|
|
151
|
-
officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx");
|
|
152
|
-
.then((data) => {
|
|
153
|
-
var newText = data + "look, I can parse a powerpoint file"
|
|
154
|
-
callSomeOtherFunction(newText);
|
|
155
|
-
})
|
|
156
|
-
.catch((err) => console.error(err))
|
|
157
|
-
|
|
158
|
-
// Using relative path for file is also fine
|
|
159
|
-
officeParser.parseOffice("files/myWorkSheet.ods", function(data, err){
|
|
160
|
-
if (err) return console.log(err);
|
|
161
|
-
var newText = data + "look, I can parse an excel file"
|
|
162
|
-
callSomeOtherFunction(newText);
|
|
163
|
-
})
|
|
131
|
+
const config = {
|
|
132
|
+
newlineDelimiter: " ", // Separate new lines with a space instead of the default \n.
|
|
133
|
+
ignoreNotes: true // Ignore notes while parsing presentation files like pptx or odp.
|
|
134
|
+
}
|
|
164
135
|
|
|
165
|
-
//
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
}
|
|
171
|
-
|
|
172
|
-
|
|
136
|
+
// relative path is also fine => eg: files/myWorkSheet.ods
|
|
137
|
+
officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config);
|
|
138
|
+
.then(data => {
|
|
139
|
+
const newText = data + " look, I can parse a powerpoint file";
|
|
140
|
+
callSomeOtherFunction(newText);
|
|
141
|
+
})
|
|
142
|
+
.catch((err) => console.error(err));
|
|
143
|
+
|
|
144
|
+
// Search for a term in the parsed text.
|
|
145
|
+
function searchForTermInOfficeFile(searchterm, filepath) {
|
|
146
|
+
return officeParser.parseOfficeAsync(filepath)
|
|
147
|
+
.then(data => data.indexOf(searchterm) != -1)
|
|
173
148
|
}
|
|
174
149
|
```
|
|
175
150
|
|
|
176
151
|
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
### Old but functional way of extracting text from word, powerpoint and excel files
|
|
180
|
-
*These were the initial methods of parsing text till parseOffice method came into existence. These still exist and form the skeleton to this module as parseOffice redirects the below functions anyway. These functions will forever remain available to guarantee long-term usage of this module. I will ensure backward-compatibility with all previous versions.*
|
|
181
|
-
|
|
182
|
-
**Usage**
|
|
183
|
-
```js
|
|
152
|
+
**Example - TypeScript**
|
|
153
|
+
```ts
|
|
184
154
|
const officeParser = require('officeparser');
|
|
185
155
|
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
//
|
|
189
|
-
|
|
190
|
-
console.log(data)
|
|
191
|
-
})
|
|
192
|
-
|
|
193
|
-
officeParser.parsePowerPoint("/path/to/powerpoint.pptx", function(data, err){
|
|
194
|
-
// "data" string in the callback here is the text parsed from the powerpoint file passed in the first argument above
|
|
195
|
-
if (err) return console.log(err);
|
|
196
|
-
console.log(data)
|
|
197
|
-
})
|
|
198
|
-
|
|
199
|
-
officeParser.parseExcel("/path/to/excel.xlsx", function(data, err){
|
|
200
|
-
// "data" string in the callback here is the text parsed from the excel file passed in the first argument above
|
|
201
|
-
if (err) return console.log(err);
|
|
202
|
-
console.log(data)
|
|
203
|
-
})
|
|
204
|
-
|
|
205
|
-
officeParser.parseOpenOffice("/path/to/writer.odt", function(data, err){
|
|
206
|
-
// "data" string in the callback here is the text parsed from the writer file passed in the first argument above
|
|
207
|
-
if (err) return console.log(err);
|
|
208
|
-
console.log(data)
|
|
209
|
-
})
|
|
210
|
-
|
|
211
|
-
// promise
|
|
212
|
-
officeParser.parseWordAsync("/path/to/word.docx");
|
|
213
|
-
.then((data) => {
|
|
214
|
-
// data is the parsed text
|
|
215
|
-
})
|
|
216
|
-
officeParser.parsePowerPointAsync("/path/to/powerpoint.pptx");
|
|
217
|
-
.then((data) => {
|
|
218
|
-
// data is the parsed text
|
|
219
|
-
})
|
|
220
|
-
officeParser.parseExcelAsync("/path/to/excel.xlsx");
|
|
221
|
-
.then((data) => {
|
|
222
|
-
// data is the parsed text
|
|
223
|
-
})
|
|
224
|
-
officeParser.parseOpenOfficeAsync("/path/to/writer.odt");
|
|
225
|
-
.then((data) => {
|
|
226
|
-
// data is the parsed text
|
|
227
|
-
})
|
|
228
|
-
|
|
229
|
-
// async/await
|
|
230
|
-
try {
|
|
231
|
-
// "data" string returned from promise here is the text parsed from the office file passed in the first argument
|
|
232
|
-
const data1 = await officeParser.parseWordAsync("/path/to/word.docx");
|
|
233
|
-
|
|
234
|
-
const data2 = await officeParser.parsePowerPointAsync("/path/to/powerpoint.pptx");
|
|
235
|
-
|
|
236
|
-
const data3 = await officeParser.parseExcelAsync("/path/to/excel.xlsx");
|
|
156
|
+
const config: OfficeParserConfig = {
|
|
157
|
+
newlineDelimiter: " ", // Separate new lines with a space instead of the default \n.
|
|
158
|
+
ignoreNotes: true // Ignore notes while parsing presentation files like pptx or odp.
|
|
159
|
+
}
|
|
237
160
|
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
161
|
+
// relative path is also fine => eg: files/myWorkSheet.ods
|
|
162
|
+
officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config);
|
|
163
|
+
.then(data => {
|
|
164
|
+
const newText = data + " look, I can parse a powerpoint file";
|
|
165
|
+
callSomeOtherFunction(newText);
|
|
166
|
+
})
|
|
167
|
+
.catch((err) => console.error(err));
|
|
168
|
+
|
|
169
|
+
// Search for a term in the parsed text.
|
|
170
|
+
function searchForTermInOfficeFile(searchterm, filepath): Promise<boolean> {
|
|
171
|
+
return officeParser.parseOfficeAsync(filepath)
|
|
172
|
+
.then(data => data.indexOf(searchterm) != -1)
|
|
242
173
|
}
|
|
243
174
|
```
|
|
244
175
|
|
|
245
|
-
**Example**
|
|
246
|
-
```js
|
|
247
|
-
const officeParser = require('officeparser');
|
|
248
|
-
|
|
249
|
-
// callback
|
|
250
|
-
officeParser.parseWord("C:\\files\\myText.docx", function(data, err){
|
|
251
|
-
if (err) return console.log(err);
|
|
252
|
-
var newText = data + "look, I can parse a word file"
|
|
253
|
-
callSomeOtherFunction(newText);
|
|
254
|
-
})
|
|
255
176
|
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
callSomeOtherFunction(newText);
|
|
260
|
-
})
|
|
177
|
+
\
|
|
178
|
+
\
|
|
179
|
+
**Please take note: I have breached convention in placing err as second argument in my callback but please understand that I had to do it to not break other people's existing modules.**
|
|
261
180
|
|
|
262
|
-
|
|
263
|
-
officeParser.parseExcel("files/myWorkSheet.xlsx", function(data, err){
|
|
264
|
-
if (err) return console.log(err);
|
|
265
|
-
var newText = data + "look, I can parse an excel file"
|
|
266
|
-
callSomeOtherFunction(newText);
|
|
267
|
-
})
|
|
181
|
+
*Optionally change decompression location for office Files at personalised locations for environments with restricted write access*
|
|
268
182
|
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
var newText = data + "look, I can parse an OpenOffice file"
|
|
272
|
-
callSomeOtherFunction(newText);
|
|
273
|
-
})
|
|
183
|
+
```js
|
|
184
|
+
const officeParser = require('officeparser');
|
|
274
185
|
|
|
275
|
-
//
|
|
276
|
-
|
|
277
|
-
.
|
|
278
|
-
let newText1 = data1 + "look, I can parse a word file";
|
|
279
|
-
callSomeOtherFunction(newText1);
|
|
280
|
-
})
|
|
281
|
-
.catch((err) => console.error(err))
|
|
186
|
+
// Default decompress location for office Files is "officeDist" in the directory where Node is started.
|
|
187
|
+
// Put this file before parseOffice method to take effect.
|
|
188
|
+
officeParser.setDecompressionLocation("/tmp"); // New decompression location would be "/tmp/officeDist"
|
|
282
189
|
|
|
283
|
-
|
|
284
|
-
.
|
|
285
|
-
let newText2 = data2 + "look, I can parse a powerpoint file";
|
|
286
|
-
callSomeOtherFunction(newText2);
|
|
287
|
-
})
|
|
288
|
-
.catch((err) => console.error(err))
|
|
190
|
+
// P.S.: Setting location on a Windows environment with '\' hierarchy requires to be entered twice '\\'
|
|
191
|
+
officeParser.setDecompressionLocation("C:\\tmp"); // New decompression location would be "C:\tmp\officeDist"
|
|
289
192
|
|
|
290
|
-
officeParser.parseExcelAsync("files/myWorkSheet.xlsx");
|
|
291
|
-
.then((data) => {
|
|
292
|
-
let newText3 = data3 + "look, I can parse an excel file";
|
|
293
|
-
callSomeOtherFunction(newText3);
|
|
294
|
-
})
|
|
295
|
-
.catch((err) => console.error(err))
|
|
296
193
|
|
|
297
|
-
officeParser.
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
194
|
+
officeParser.parseOffice("/path/to/officeFile", function(data, err){
|
|
195
|
+
// "data" string in the callback here is the text parsed from the office file passed in the first argument above
|
|
196
|
+
if (err) {
|
|
197
|
+
console.log(err);
|
|
198
|
+
return;
|
|
199
|
+
}
|
|
200
|
+
console.log(data);
|
|
301
201
|
})
|
|
302
|
-
.catch((err) => console.error(err))
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
// async/await
|
|
306
|
-
try {
|
|
307
|
-
const data1 = await officeParser.parseWordAsync("C:\\files\\myText.docx");
|
|
308
|
-
let newText1 = data1 + "look, I can parse a word file";
|
|
309
|
-
await callSomeOtherFunction(newText1);
|
|
310
|
-
|
|
311
|
-
const data2 = await officeParser.parsePowerPointAsync("/Users/harsh/Desktop/files/mySlides.pptx");
|
|
312
|
-
let newText2 = data2 + "look, I can parse a powerpoint file";
|
|
313
|
-
await callSomeOtherFunction(newText2);
|
|
314
|
-
|
|
315
|
-
// Using relative path for file is also fine
|
|
316
|
-
const data3 = await officeParser.parseExcelAsync("files/myWorkSheet.xlsx");
|
|
317
|
-
let newText3 = data3 + "look, I can parse an excel file";
|
|
318
|
-
await callSomeOtherFunction(newText3);
|
|
319
|
-
|
|
320
|
-
const data4 = await officeParser.parseOpenOfficeAsync("files/myDocument.odt");
|
|
321
|
-
let newText4 = data4 + "look, I can parse an OpenOffice file";
|
|
322
|
-
await callSomeOtherFunction(newText4);
|
|
323
|
-
} catch (err) {
|
|
324
|
-
// resolve error
|
|
325
|
-
console.log(err);
|
|
326
|
-
}
|
|
327
202
|
```
|
|
328
203
|
|
|
204
|
+
## Known Bugs
|
|
205
|
+
1. Inconsistency and incorrectness in the positioning of footnotes and endnotes in .docx files where the footnotes and endnotes would end up at the end of the parsed text whereas it would be positioned exactly after the referenced word in .odt files.
|
|
206
|
+
2. The charts and objects information of .odt files are not accurate and may end up showing a few NaN in some cases.
|
|
329
207
|
----------
|
|
330
208
|
|
|
209
|
+
**npm**
|
|
210
|
+
https://npmjs.com/package/officeparser
|
|
211
|
+
|
|
331
212
|
**github**
|
|
332
213
|
https://github.com/harshankur/officeParser
|