officeparser 4.2.0 → 5.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -9
- package/officeParser.js +578 -548
- package/package.json +5 -4
- package/typings/officeParser.d.ts +2 -11
package/README.md
CHANGED
|
@@ -13,6 +13,7 @@ A Node.js library to parse text out of any office file.
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
#### Update
|
|
16
|
+
* 2024/10/21 - Replaced extracting zip files from decompress to yauzl. This means that we now extract files in memory and we no longer need to write them to disk. Removed config flags related to extracted files. Added flags for CLI execution.
|
|
16
17
|
* 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
|
|
17
18
|
* 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
|
|
18
19
|
* 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
|
|
@@ -37,7 +38,6 @@ A Node.js library to parse text out of any office file.
|
|
|
37
38
|
|
|
38
39
|
## Install via npm
|
|
39
40
|
|
|
40
|
-
|
|
41
41
|
```
|
|
42
42
|
npm i officeparser
|
|
43
43
|
```
|
|
@@ -45,14 +45,20 @@ npm i officeparser
|
|
|
45
45
|
## Command Line usage
|
|
46
46
|
If you want to call the installed officeParser.js file, use below command
|
|
47
47
|
```
|
|
48
|
-
node
|
|
48
|
+
node <path/to/officeParser.js> [--configOption=value] [FILE_PATH]
|
|
49
|
+
node officeparser [--configOption=value] [FILE_PATH]
|
|
49
50
|
```
|
|
50
51
|
|
|
51
|
-
Otherwise, you can simply use npx to instantly extract parsed data.
|
|
52
|
+
Otherwise, you can simply use npx without installing the node module to instantly extract parsed data.
|
|
52
53
|
```
|
|
53
|
-
npx officeparser
|
|
54
|
+
npx officeparser [--configOption=value] [FILE_PATH]
|
|
54
55
|
```
|
|
55
56
|
|
|
57
|
+
### Config Options:
|
|
58
|
+
- `--ignoreNotes=[true|false]` Flag to ignore notes from files like PowerPoint. Default is false.
|
|
59
|
+
- `--newlineDelimiter=[delimiter]` The delimiter to use for new lines. Default is `\n`.
|
|
60
|
+
- `--putNotesAtLast=[true|false]` Flag to collect notes at the end of files like PowerPoint. Default is false.
|
|
61
|
+
- `--outputErrorToConsole=[true|false]` Flag to output errors to the console. Default is false.
|
|
56
62
|
|
|
57
63
|
## Library Usage
|
|
58
64
|
```js
|
|
@@ -101,8 +107,6 @@ officeParser.parseOfficeAsync(fileBuffers);
|
|
|
101
107
|
*Optionally add a config object as 3rd variable to parseOffice for the following configurations*
|
|
102
108
|
| Flag | DataType | Default | Explanation |
|
|
103
109
|
|----------------------|----------|------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|
104
|
-
| tempFilesLocation | string | officeParserTemp | The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. **Please ensure that this directory actually exists.** Default is officeParserTemp. |
|
|
105
|
-
| preserveTempFiles | boolean | false | Flag to not delete the internal content files and the possible duplicate temp files that it uses after unzipping office files. Default is false. It always deletes all of those files. |
|
|
106
110
|
| outputErrorToConsole | boolean | false | Flag to show all the logs to console in case of an error. Default is false. |
|
|
107
111
|
| newlineDelimiter | string | \n | The delimiter used for every new line in places that allow multiline text like word. Default is \n. |
|
|
108
112
|
| ignoreNotes | boolean | false | Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default. |
|
|
@@ -157,7 +161,7 @@ function searchForTermInOfficeFile(searchterm, filepath) {
|
|
|
157
161
|
|
|
158
162
|
**Example - TypeScript**
|
|
159
163
|
```ts
|
|
160
|
-
|
|
164
|
+
import { OfficeParserConfig, parseOfficeAsync } from 'officeparser';
|
|
161
165
|
|
|
162
166
|
const config: OfficeParserConfig = {
|
|
163
167
|
newlineDelimiter: " ", // Separate new lines with a space instead of the default \n.
|
|
@@ -165,7 +169,7 @@ const config: OfficeParserConfig = {
|
|
|
165
169
|
}
|
|
166
170
|
|
|
167
171
|
// relative path is also fine => eg: files/myWorkSheet.ods
|
|
168
|
-
|
|
172
|
+
parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config);
|
|
169
173
|
.then(data => {
|
|
170
174
|
const newText = data + " look, I can parse a powerpoint file";
|
|
171
175
|
callSomeOtherFunction(newText);
|
|
@@ -174,7 +178,7 @@ officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config
|
|
|
174
178
|
|
|
175
179
|
// Search for a term in the parsed text.
|
|
176
180
|
function searchForTermInOfficeFile(searchterm: string, filepath: string): Promise<boolean> {
|
|
177
|
-
return
|
|
181
|
+
return parseOfficeAsync(filepath)
|
|
178
182
|
.then(data => data.indexOf(searchterm) != -1)
|
|
179
183
|
}
|
|
180
184
|
```
|