officeparser 6.1.1 → 7.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +301 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +74 -31
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +828 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +783 -5
- package/dist/types.js +73 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.d.ts +8 -3
- package/dist/utils/envUtils.js +117 -34
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +110 -52
- package/dist/utils/moduleLoader.js +19 -11
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +26 -7
package/dist/OfficeParser.js
CHANGED
|
@@ -12,6 +12,9 @@
|
|
|
12
12
|
* - ODT, ODP, ODS (OpenDocument formats)
|
|
13
13
|
* - PDF (Portable Document Format)
|
|
14
14
|
* - RTF (Rich Text Format)
|
|
15
|
+
* - CSV (Comma-Separated Values)
|
|
16
|
+
* - MD (Markdown)
|
|
17
|
+
* - HTML (HyperText Markup Language)
|
|
15
18
|
*
|
|
16
19
|
* **Usage:**
|
|
17
20
|
* ```typescript
|
|
@@ -35,13 +38,18 @@
|
|
|
35
38
|
*/
|
|
36
39
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
37
40
|
exports.OfficeParser = void 0;
|
|
38
|
-
const
|
|
41
|
+
const CsvParser_js_1 = require("./parsers/CsvParser.js");
|
|
39
42
|
const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
|
|
43
|
+
const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
|
|
44
|
+
const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
|
|
40
45
|
const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
|
|
41
46
|
const PdfParser_js_1 = require("./parsers/PdfParser.js");
|
|
42
47
|
const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
|
|
43
48
|
const RtfParser_js_1 = require("./parsers/RtfParser.js");
|
|
44
49
|
const WordParser_js_1 = require("./parsers/WordParser.js");
|
|
50
|
+
const types_js_1 = require("./types.js");
|
|
51
|
+
const configUtils_js_1 = require("./utils/configUtils.js");
|
|
52
|
+
const envUtils_js_1 = require("./utils/envUtils.js");
|
|
45
53
|
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
46
54
|
const moduleLoader_js_1 = require("./utils/moduleLoader.js");
|
|
47
55
|
const ocrUtils_js_1 = require("./utils/ocrUtils.js");
|
|
@@ -72,6 +80,9 @@ class OfficeParser {
|
|
|
72
80
|
* - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
|
|
73
81
|
* - `.pdf` → PdfParser (PDF.js)
|
|
74
82
|
* - `.rtf` → RtfParser (custom RTF parser)
|
|
83
|
+
* - `.csv` → CsvParser
|
|
84
|
+
* - `.md` → MarkdownParser
|
|
85
|
+
* - `.html` → HtmlParser
|
|
75
86
|
*
|
|
76
87
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
77
88
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|
|
@@ -107,28 +118,20 @@ class OfficeParser {
|
|
|
107
118
|
else {
|
|
108
119
|
actualConfig = configOrCallback || {};
|
|
109
120
|
}
|
|
110
|
-
const internalConfig =
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
ocrLanguage: 'eng',
|
|
118
|
-
includeRawContent: false,
|
|
119
|
-
serializeRawContent: true,
|
|
120
|
-
preserveXmlWhitespace: false,
|
|
121
|
-
pdfWorkerSrc: '',
|
|
122
|
-
ocrConfig: {},
|
|
123
|
-
includeBreakNodes: false,
|
|
124
|
-
...actualConfig
|
|
121
|
+
const internalConfig = (0, configUtils_js_1.resolveParserConfig)(actualConfig);
|
|
122
|
+
const parsingWarnings = [];
|
|
123
|
+
const originalOnWarning = internalConfig.onWarning;
|
|
124
|
+
internalConfig.onWarning = (issue) => {
|
|
125
|
+
parsingWarnings.push(issue);
|
|
126
|
+
if (originalOnWarning)
|
|
127
|
+
originalOnWarning(issue);
|
|
125
128
|
};
|
|
126
129
|
let buffer = Buffer.alloc(0);
|
|
127
|
-
let ext = '';
|
|
130
|
+
let ext = internalConfig.fileType ?? '';
|
|
128
131
|
let filePath;
|
|
129
132
|
try {
|
|
130
133
|
if (!file) {
|
|
131
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
134
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
|
|
132
135
|
}
|
|
133
136
|
if (file instanceof ArrayBuffer) {
|
|
134
137
|
buffer = Buffer.from(file);
|
|
@@ -144,29 +147,59 @@ class OfficeParser {
|
|
|
144
147
|
// shim 'fs' so it won't crash at build time.
|
|
145
148
|
const fs = await import('fs');
|
|
146
149
|
if (!fs.existsSync(file)) {
|
|
147
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
150
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
|
|
148
151
|
}
|
|
149
152
|
if (fs.lstatSync(file).isDirectory()) {
|
|
150
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
153
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
|
|
151
154
|
}
|
|
152
155
|
buffer = fs.readFileSync(file);
|
|
153
|
-
ext = file.split('.').pop()
|
|
156
|
+
ext = ext || file.split('.').pop() || '';
|
|
154
157
|
}
|
|
155
158
|
else {
|
|
156
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
159
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
|
|
157
160
|
}
|
|
158
|
-
if
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
161
|
+
// Attempt to detect file type from buffer only if extension is unknown.
|
|
162
|
+
// This matches v6 behavior and prevents crashes in older Node environments
|
|
163
|
+
// where file-type 22.x might be incompatible.
|
|
164
|
+
if (buffer.length > 0 && !ext) {
|
|
165
|
+
try {
|
|
166
|
+
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
167
|
+
const type = await fileTypeFromBuffer(buffer);
|
|
168
|
+
if (type) {
|
|
169
|
+
ext = type.ext;
|
|
170
|
+
}
|
|
171
|
+
else {
|
|
172
|
+
// If no extension could be detected and none was provided,
|
|
173
|
+
// it might be a text-based format (csv, md, html) which
|
|
174
|
+
// lack magic bytes. We'll let the switch default handle it.
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
catch (error) {
|
|
178
|
+
// Log warning but don't crash; the switch below will handle unsupported/missing ext
|
|
179
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
else if (buffer.length > 0 && ext) {
|
|
183
|
+
// If extension is known, we can optionally verify it, but we wrap it
|
|
184
|
+
// in a try-catch to avoid breaking Node 18 if file-type fails to load.
|
|
185
|
+
try {
|
|
186
|
+
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
187
|
+
const type = await fileTypeFromBuffer(buffer);
|
|
188
|
+
if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
|
|
189
|
+
// Mismatch found between authoritative extension and detected content
|
|
190
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
|
|
191
|
+
}
|
|
163
192
|
}
|
|
164
|
-
|
|
165
|
-
|
|
193
|
+
catch (error) {
|
|
194
|
+
// Log warning so user knows verification could not be performed
|
|
195
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
|
|
166
196
|
}
|
|
167
197
|
}
|
|
198
|
+
if (!ext) {
|
|
199
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
|
|
200
|
+
}
|
|
168
201
|
let result;
|
|
169
|
-
switch (ext) {
|
|
202
|
+
switch (ext.toLowerCase()) {
|
|
170
203
|
case 'docx':
|
|
171
204
|
result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
|
|
172
205
|
break;
|
|
@@ -187,9 +220,19 @@ class OfficeParser {
|
|
|
187
220
|
case 'rtf':
|
|
188
221
|
result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
|
|
189
222
|
break;
|
|
223
|
+
case 'csv':
|
|
224
|
+
result = await (0, CsvParser_js_1.parseCsv)(buffer, internalConfig);
|
|
225
|
+
break;
|
|
226
|
+
case 'html':
|
|
227
|
+
result = await (0, HtmlParser_js_1.parseHtml)(buffer, internalConfig);
|
|
228
|
+
break;
|
|
229
|
+
case 'md':
|
|
230
|
+
result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
|
|
231
|
+
break;
|
|
190
232
|
default:
|
|
191
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
233
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
|
|
192
234
|
}
|
|
235
|
+
result.warnings = parsingWarnings;
|
|
193
236
|
if (callback)
|
|
194
237
|
callback(result);
|
|
195
238
|
return result;
|
package/dist/cli.d.ts
CHANGED
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
9
9
|
*
|
|
10
10
|
* Options (--key=value):
|
|
11
|
-
* --
|
|
11
|
+
* --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
|
|
12
|
+
* --output=path Save result to a file
|
|
13
|
+
* --toText=true Legacy flag for plain text output
|
|
12
14
|
* --ocr=true Enable OCR for images
|
|
13
15
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
14
16
|
* --extractAttachments=true Extract embedded attachments
|
package/dist/cli.js
CHANGED
|
@@ -9,7 +9,9 @@
|
|
|
9
9
|
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
10
10
|
*
|
|
11
11
|
* Options (--key=value):
|
|
12
|
-
* --
|
|
12
|
+
* --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
|
|
13
|
+
* --output=path Save result to a file
|
|
14
|
+
* --toText=true Legacy flag for plain text output
|
|
13
15
|
* --ocr=true Enable OCR for images
|
|
14
16
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
15
17
|
* --extractAttachments=true Extract embedded attachments
|
|
@@ -18,12 +20,49 @@
|
|
|
18
20
|
* --includeRawContent=true Include raw content in AST
|
|
19
21
|
* --outputErrorToConsole=true Log errors to console
|
|
20
22
|
*/
|
|
23
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
24
|
+
if (k2 === undefined) k2 = k;
|
|
25
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
26
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
27
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
28
|
+
}
|
|
29
|
+
Object.defineProperty(o, k2, desc);
|
|
30
|
+
}) : (function(o, m, k, k2) {
|
|
31
|
+
if (k2 === undefined) k2 = k;
|
|
32
|
+
o[k2] = m[k];
|
|
33
|
+
}));
|
|
34
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
35
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
36
|
+
}) : function(o, v) {
|
|
37
|
+
o["default"] = v;
|
|
38
|
+
});
|
|
39
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
40
|
+
var ownKeys = function(o) {
|
|
41
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
42
|
+
var ar = [];
|
|
43
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
44
|
+
return ar;
|
|
45
|
+
};
|
|
46
|
+
return ownKeys(o);
|
|
47
|
+
};
|
|
48
|
+
return function (mod) {
|
|
49
|
+
if (mod && mod.__esModule) return mod;
|
|
50
|
+
var result = {};
|
|
51
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
52
|
+
__setModuleDefault(result, mod);
|
|
53
|
+
return result;
|
|
54
|
+
};
|
|
55
|
+
})();
|
|
21
56
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
22
57
|
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
58
|
+
const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
|
|
59
|
+
const fs = __importStar(require("fs"));
|
|
23
60
|
const args = process.argv.slice(2);
|
|
24
61
|
let fileArg;
|
|
25
62
|
let toText = false;
|
|
26
63
|
let verbose = false;
|
|
64
|
+
let outputFormat;
|
|
65
|
+
let outputFile;
|
|
27
66
|
const configArgs = [];
|
|
28
67
|
function isConfigOption(arg) {
|
|
29
68
|
return arg.startsWith('--') && arg.includes('=');
|
|
@@ -43,34 +82,75 @@ if (fileArg) {
|
|
|
43
82
|
const cleanKey = key.replace('--', '');
|
|
44
83
|
const lowerValue = value.toLowerCase();
|
|
45
84
|
const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
85
|
+
const knownBooleans = new Set([
|
|
86
|
+
'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'putNotesAtLast',
|
|
87
|
+
'includeRawContent', 'outputErrorToConsole', 'serializeRawContent',
|
|
88
|
+
'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
|
|
89
|
+
]);
|
|
90
|
+
if (cleanKey === 'format') {
|
|
91
|
+
outputFormat = value;
|
|
51
92
|
}
|
|
52
|
-
else if (cleanKey === '
|
|
53
|
-
|
|
54
|
-
verbose = boolValue;
|
|
55
|
-
else
|
|
56
|
-
console.warn(`Invalid value for verbose: ${value}`);
|
|
93
|
+
else if (cleanKey === 'output') {
|
|
94
|
+
outputFile = value;
|
|
57
95
|
}
|
|
58
96
|
else {
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
97
|
+
if (boolValue !== undefined) {
|
|
98
|
+
if (cleanKey === 'toText')
|
|
99
|
+
toText = boolValue;
|
|
100
|
+
else if (cleanKey === 'verbose') {
|
|
101
|
+
verbose = boolValue;
|
|
102
|
+
if (verbose)
|
|
103
|
+
config.outputErrorToConsole = true;
|
|
104
|
+
}
|
|
105
|
+
else {
|
|
106
|
+
// @ts-ignore
|
|
107
|
+
config[cleanKey] = boolValue;
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
else if (knownBooleans.has(cleanKey)) {
|
|
111
|
+
console.warn(`Invalid boolean value for --${cleanKey}: ${value}. Using default.`);
|
|
112
|
+
}
|
|
113
|
+
else {
|
|
114
|
+
// @ts-ignore
|
|
64
115
|
config[cleanKey] = value;
|
|
116
|
+
}
|
|
65
117
|
}
|
|
66
118
|
});
|
|
67
119
|
OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
|
|
68
120
|
.then(async (ast) => {
|
|
69
|
-
|
|
70
|
-
|
|
121
|
+
let output;
|
|
122
|
+
if (outputFormat) {
|
|
123
|
+
const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat);
|
|
124
|
+
if (Array.isArray(result.value)) {
|
|
125
|
+
output = JSON.stringify(result.value, null, 2);
|
|
126
|
+
}
|
|
127
|
+
else {
|
|
128
|
+
output = result.value;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
else if (toText) {
|
|
132
|
+
output = ast.toText();
|
|
133
|
+
}
|
|
134
|
+
else {
|
|
135
|
+
output = JSON.stringify(ast, null, 2);
|
|
136
|
+
}
|
|
137
|
+
if (outputFile) {
|
|
138
|
+
if (output instanceof Uint8Array) {
|
|
139
|
+
fs.writeFileSync(outputFile, output);
|
|
140
|
+
}
|
|
141
|
+
else {
|
|
142
|
+
fs.writeFileSync(outputFile, output, 'utf8');
|
|
143
|
+
}
|
|
144
|
+
if (verbose)
|
|
145
|
+
console.log(`Output written to ${outputFile}`);
|
|
71
146
|
}
|
|
72
147
|
else {
|
|
73
|
-
|
|
148
|
+
if (output instanceof Uint8Array) {
|
|
149
|
+
process.stdout.write(output);
|
|
150
|
+
}
|
|
151
|
+
else {
|
|
152
|
+
process.stdout.write(output + '\n');
|
|
153
|
+
}
|
|
74
154
|
}
|
|
75
155
|
// Ensure OCR workers are terminated for clean CLI exit
|
|
76
156
|
if (config.ocr) {
|
|
@@ -97,7 +177,9 @@ else {
|
|
|
97
177
|
console.log('Usage: officeparser <file> [--option=value]');
|
|
98
178
|
console.log('');
|
|
99
179
|
console.log('Options:');
|
|
100
|
-
console.log(' --
|
|
180
|
+
console.log(' --format=json|md|html|rtf|csv|text|pdf|chunks Convert to specified format');
|
|
181
|
+
console.log(' --output=file.ext Save output to file instead of stdout');
|
|
182
|
+
console.log(' --toText=true Output plain text instead of JSON AST (legacy)');
|
|
101
183
|
console.log(' --ocr=true Enable OCR for images');
|
|
102
184
|
console.log(' --ocrLanguage=eng OCR language (default: eng)');
|
|
103
185
|
console.log(' --extractAttachments=true Extract embedded attachments');
|
|
@@ -111,7 +193,9 @@ else {
|
|
|
111
193
|
console.log('');
|
|
112
194
|
console.log('Examples:');
|
|
113
195
|
console.log(' officeparser document.docx');
|
|
114
|
-
console.log(' officeparser document.docx --
|
|
115
|
-
console.log(' officeparser
|
|
196
|
+
console.log(' officeparser document.docx --format=html --output=doc.html');
|
|
197
|
+
console.log(' officeparser document.docx --format=md');
|
|
198
|
+
console.log(' officeparser report.pdf --ocr=true --format=text');
|
|
199
|
+
console.log(' officeparser data.xlsx --format=csv --output=data.csv');
|
|
116
200
|
console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
|
|
117
201
|
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { DeepRequired, DocumentStructureChunkingConfig, FixedSizeChunkingConfig, FullGeneratorConfig, OfficeParserConfig, SemanticChunkingConfig } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* The default regex used for identifying sentence boundaries.
|
|
4
|
+
* When this default is used, the generator employs a high-fidelity "robust"
|
|
5
|
+
* segmenter that accounts for common abbreviations (Mr., Dr., etc.).
|
|
6
|
+
*/
|
|
7
|
+
export declare const DEFAULT_SENTENCE_BOUNDARY_REGEX: RegExp;
|
|
8
|
+
/**
|
|
9
|
+
* Common abbreviations that should not trigger a sentence split when followed by a period.
|
|
10
|
+
*/
|
|
11
|
+
export declare const DEFAULT_ABBREVIATIONS: string[];
|
|
12
|
+
/**
|
|
13
|
+
* Default configuration for the OfficeParser.
|
|
14
|
+
*/
|
|
15
|
+
export declare const DEFAULT_OFFICE_PARSER_CONFIG: DeepRequired<OfficeParserConfig>;
|
|
16
|
+
/**
|
|
17
|
+
* Default configuration for Fixed-Size chunking.
|
|
18
|
+
*/
|
|
19
|
+
export declare const DEFAULT_FIXED_SIZE_CHUNKING_CONFIG: Required<Omit<FixedSizeChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
|
|
20
|
+
sentenceBoundaryRegex: string | RegExp;
|
|
21
|
+
abbreviations: string[];
|
|
22
|
+
};
|
|
23
|
+
/**
|
|
24
|
+
* Default configuration for Document-Structure chunking.
|
|
25
|
+
*/
|
|
26
|
+
export declare const DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG: Required<Omit<DocumentStructureChunkingConfig, 'sentenceBoundaryRegex' | 'abbreviations'>> & {
|
|
27
|
+
sentenceBoundaryRegex: string | RegExp;
|
|
28
|
+
abbreviations: string[];
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* Default configuration for Semantic chunking.
|
|
32
|
+
* Note: `embeddingFunction` has no meaningful default and must be provided by the user.
|
|
33
|
+
*/
|
|
34
|
+
export declare const DEFAULT_SEMANTIC_CHUNKING_CONFIG: Required<Omit<SemanticChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
|
|
35
|
+
sentenceBoundaryRegex: string | RegExp;
|
|
36
|
+
abbreviations: string[];
|
|
37
|
+
};
|
|
38
|
+
/**
|
|
39
|
+
* Default configuration for the OfficeGenerator.
|
|
40
|
+
*/
|
|
41
|
+
export declare const DEFAULT_GENERATOR_CONFIG: FullGeneratorConfig;
|
package/dist/defaults.js
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.DEFAULT_GENERATOR_CONFIG = exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = exports.DEFAULT_OFFICE_PARSER_CONFIG = exports.DEFAULT_ABBREVIATIONS = exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = void 0;
|
|
4
|
+
const PDFJS_VERSION = '5.6.205';
|
|
5
|
+
const DEFAULT_PDF_WORKER_SRC = `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
|
|
6
|
+
/**
|
|
7
|
+
* The default regex used for identifying sentence boundaries.
|
|
8
|
+
* When this default is used, the generator employs a high-fidelity "robust"
|
|
9
|
+
* segmenter that accounts for common abbreviations (Mr., Dr., etc.).
|
|
10
|
+
*/
|
|
11
|
+
exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = /[.!?。!?]/;
|
|
12
|
+
/**
|
|
13
|
+
* Common abbreviations that should not trigger a sentence split when followed by a period.
|
|
14
|
+
*/
|
|
15
|
+
exports.DEFAULT_ABBREVIATIONS = ['Mr', 'Dr', 'Ms', 'Inc', 'Ltd', 'Prof', 'Sr', 'Jr', 'vs', 'etc'];
|
|
16
|
+
/**
|
|
17
|
+
* Default configuration for OCR.
|
|
18
|
+
*/
|
|
19
|
+
const DEFAULT_OCR_CONFIG = {
|
|
20
|
+
language: 'eng',
|
|
21
|
+
workerPath: '',
|
|
22
|
+
corePath: '',
|
|
23
|
+
langPath: '',
|
|
24
|
+
autoTerminateTimeout: 10000,
|
|
25
|
+
};
|
|
26
|
+
/**
|
|
27
|
+
* Default configuration for the OfficeParser.
|
|
28
|
+
*/
|
|
29
|
+
exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
30
|
+
outputErrorToConsole: false,
|
|
31
|
+
onWarning: () => { },
|
|
32
|
+
newlineDelimiter: '\n',
|
|
33
|
+
ignoreNotes: false,
|
|
34
|
+
putNotesAtLast: false,
|
|
35
|
+
extractAttachments: false,
|
|
36
|
+
includeRawContent: false,
|
|
37
|
+
ocr: false,
|
|
38
|
+
ocrLanguage: 'eng',
|
|
39
|
+
ocrConfig: DEFAULT_OCR_CONFIG,
|
|
40
|
+
serializeRawContent: true,
|
|
41
|
+
preserveXmlWhitespace: false,
|
|
42
|
+
pdfWorkerSrc: DEFAULT_PDF_WORKER_SRC,
|
|
43
|
+
includeBreakNodes: false,
|
|
44
|
+
ignoreInternalLinks: false,
|
|
45
|
+
fileType: null,
|
|
46
|
+
csvDelimiter: ',',
|
|
47
|
+
};
|
|
48
|
+
/**
|
|
49
|
+
* Default configuration for HTML generation.
|
|
50
|
+
*/
|
|
51
|
+
const DEFAULT_HTML_GENERATOR_CONFIG = {
|
|
52
|
+
standalone: true,
|
|
53
|
+
chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
|
|
54
|
+
};
|
|
55
|
+
/**
|
|
56
|
+
* Default configuration for PDF generation.
|
|
57
|
+
*/
|
|
58
|
+
const DEFAULT_PDF_GENERATOR_CONFIG = {
|
|
59
|
+
format: 'A4',
|
|
60
|
+
width: '',
|
|
61
|
+
height: '',
|
|
62
|
+
landscape: false,
|
|
63
|
+
printBackground: true,
|
|
64
|
+
scale: 1,
|
|
65
|
+
margin: {
|
|
66
|
+
top: 0,
|
|
67
|
+
right: 0,
|
|
68
|
+
bottom: 0,
|
|
69
|
+
left: 0
|
|
70
|
+
},
|
|
71
|
+
displayHeaderFooter: false,
|
|
72
|
+
headerTemplate: '',
|
|
73
|
+
footerTemplate: '',
|
|
74
|
+
launchOptions: {
|
|
75
|
+
headless: true,
|
|
76
|
+
args: ['--no-sandbox', '--disable-setuid-sandbox']
|
|
77
|
+
},
|
|
78
|
+
};
|
|
79
|
+
/**
|
|
80
|
+
* Default configuration for CSV generation.
|
|
81
|
+
*/
|
|
82
|
+
const DEFAULT_CSV_GENERATOR_CONFIG = {
|
|
83
|
+
sheets: '',
|
|
84
|
+
mergeSheets: true,
|
|
85
|
+
columnDelimiter: ',',
|
|
86
|
+
};
|
|
87
|
+
/**
|
|
88
|
+
* Default configuration for Markdown generation.
|
|
89
|
+
*/
|
|
90
|
+
const DEFAULT_MD_GENERATOR_CONFIG = {
|
|
91
|
+
fallbackToHtml: true,
|
|
92
|
+
};
|
|
93
|
+
/**
|
|
94
|
+
* Default configuration for plain text generation.
|
|
95
|
+
*/
|
|
96
|
+
const DEFAULT_TEXT_GENERATOR_CONFIG = {
|
|
97
|
+
newlineDelimiter: '\n',
|
|
98
|
+
preserveLayout: false,
|
|
99
|
+
};
|
|
100
|
+
/**
|
|
101
|
+
* Default configuration for Fixed-Size chunking.
|
|
102
|
+
*/
|
|
103
|
+
exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = {
|
|
104
|
+
strategy: 'fixed-size',
|
|
105
|
+
chunkSize: 1000,
|
|
106
|
+
chunkOverlap: 200,
|
|
107
|
+
separators: ['\n\n', '\n', ' ', ''],
|
|
108
|
+
stripWhitespace: true,
|
|
109
|
+
includeMetadata: true,
|
|
110
|
+
addStartIndex: false,
|
|
111
|
+
lengthFunction: (text) => text.length,
|
|
112
|
+
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
113
|
+
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
114
|
+
};
|
|
115
|
+
/**
|
|
116
|
+
* Default configuration for Document-Structure chunking.
|
|
117
|
+
*/
|
|
118
|
+
exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = {
|
|
119
|
+
strategy: 'document-structure',
|
|
120
|
+
splitBy: 'paragraph',
|
|
121
|
+
maxChunkSize: 1000,
|
|
122
|
+
tableSplitStrategy: 'row',
|
|
123
|
+
stripWhitespace: true,
|
|
124
|
+
includeMetadata: true,
|
|
125
|
+
addStartIndex: false,
|
|
126
|
+
lengthFunction: (text) => text.length,
|
|
127
|
+
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
128
|
+
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
129
|
+
};
|
|
130
|
+
/**
|
|
131
|
+
* Default configuration for Semantic chunking.
|
|
132
|
+
* Note: `embeddingFunction` has no meaningful default and must be provided by the user.
|
|
133
|
+
*/
|
|
134
|
+
exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = {
|
|
135
|
+
strategy: 'semantic',
|
|
136
|
+
similarityThreshold: 0.8,
|
|
137
|
+
maxChunkSize: 2000,
|
|
138
|
+
bufferSize: 1,
|
|
139
|
+
embeddingBatchSize: 50,
|
|
140
|
+
stripWhitespace: true,
|
|
141
|
+
includeMetadata: true,
|
|
142
|
+
addStartIndex: false,
|
|
143
|
+
lengthFunction: (text) => text.length,
|
|
144
|
+
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
145
|
+
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
146
|
+
};
|
|
147
|
+
/**
|
|
148
|
+
* The resolved default chunking config (uses document-structure as default strategy).
|
|
149
|
+
*/
|
|
150
|
+
const DEFAULT_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG;
|
|
151
|
+
/**
|
|
152
|
+
* Default configuration for the OfficeGenerator.
|
|
153
|
+
*/
|
|
154
|
+
exports.DEFAULT_GENERATOR_CONFIG = {
|
|
155
|
+
onNode: () => { },
|
|
156
|
+
onWarning: () => { },
|
|
157
|
+
styleMap: [],
|
|
158
|
+
includeFormatting: true,
|
|
159
|
+
generateIds: true,
|
|
160
|
+
renderMetadata: false,
|
|
161
|
+
ignoreDefaultStyleMap: false,
|
|
162
|
+
includeImages: true,
|
|
163
|
+
includeCharts: true,
|
|
164
|
+
ignoreInternalLinks: false,
|
|
165
|
+
htmlConfig: DEFAULT_HTML_GENERATOR_CONFIG,
|
|
166
|
+
mdConfig: DEFAULT_MD_GENERATOR_CONFIG,
|
|
167
|
+
pdfConfig: DEFAULT_PDF_GENERATOR_CONFIG,
|
|
168
|
+
csvConfig: DEFAULT_CSV_GENERATOR_CONFIG,
|
|
169
|
+
textConfig: DEFAULT_TEXT_GENERATOR_CONFIG,
|
|
170
|
+
rtfConfig: {},
|
|
171
|
+
chunksConfig: DEFAULT_CHUNKING_CONFIG,
|
|
172
|
+
};
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType } from '../types.js';
|
|
2
|
+
import { StyleMapper } from '../utils/styleMapper.js';
|
|
3
|
+
/**
|
|
4
|
+
* Base class for all document generators.
|
|
5
|
+
* Provides common traversal logic and configuration handling.
|
|
6
|
+
*/
|
|
7
|
+
export declare abstract class BaseGenerator<D extends string = string> {
|
|
8
|
+
protected destination: D;
|
|
9
|
+
protected config: FullGeneratorConfig;
|
|
10
|
+
protected ast: OfficeParserAST;
|
|
11
|
+
protected messages: OfficeIssue[];
|
|
12
|
+
protected styleMapper: StyleMapper;
|
|
13
|
+
constructor(destination: D, ast: OfficeParserAST, config?: GeneratorConfig<D> | FullGeneratorConfig);
|
|
14
|
+
/**
|
|
15
|
+
* Retrieves the semantic mapping for a node, respecting the includeFormatting flag.
|
|
16
|
+
* Per design requirements: Style mapping is bypassed if formatting is disabled.
|
|
17
|
+
*/
|
|
18
|
+
protected getSemanticMapping(node: OfficeContentNode): {
|
|
19
|
+
tag: string;
|
|
20
|
+
classes: string[];
|
|
21
|
+
attributes: Record<string, string>;
|
|
22
|
+
fresh: boolean;
|
|
23
|
+
} | undefined;
|
|
24
|
+
/**
|
|
25
|
+
* Entry point for generation.
|
|
26
|
+
*/
|
|
27
|
+
abstract generate(): Promise<ConversionResult>;
|
|
28
|
+
/**
|
|
29
|
+
* Centralized logic for handling the onNode callback.
|
|
30
|
+
* Evaluates the callback and returns a result that tells the generator how to proceed.
|
|
31
|
+
*
|
|
32
|
+
* @returns
|
|
33
|
+
* - `string`: Use this as the node's output, skip default processing.
|
|
34
|
+
* - `false`: Skip this node and its subtree.
|
|
35
|
+
* - `void`: Proceed with default processing.
|
|
36
|
+
*/
|
|
37
|
+
protected handleOnNode(node: OfficeContentNode): Promise<string | false | void>;
|
|
38
|
+
/**
|
|
39
|
+
* Recursively processes nodes and builds output.
|
|
40
|
+
*
|
|
41
|
+
* @param node - The current node being processed
|
|
42
|
+
* @param processor - A function that takes a node and its children's output and returns the node's output string.
|
|
43
|
+
* @returns The generated string for this node and its subtree.
|
|
44
|
+
*/
|
|
45
|
+
protected processNodeRecursive(node: OfficeContentNode, processor: (node: OfficeContentNode, childrenOutput: string) => string | Promise<string>): Promise<string>;
|
|
46
|
+
/**
|
|
47
|
+
* Helper to generate a unique ID (slug) from text.
|
|
48
|
+
*/
|
|
49
|
+
protected slugify(text: string): string;
|
|
50
|
+
/**
|
|
51
|
+
* Recursively extracts plain text from a node and its children.
|
|
52
|
+
*/
|
|
53
|
+
protected getNodeText(node: OfficeContentNode): string;
|
|
54
|
+
/**
|
|
55
|
+
* Reports a warning to the user and collects it for the final result.
|
|
56
|
+
*/
|
|
57
|
+
protected warn(type: OfficeWarningType, info?: any, node?: OfficeContentNode): void;
|
|
58
|
+
}
|