officeparser 6.1.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +284 -86
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -28
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +107 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +878 -5
- package/dist/officeparser.browser.iife.js +703 -49
- package/dist/officeparser.browser.mjs +703 -49
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +237 -128
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +132 -123
- package/dist/parsers/RtfParser.d.ts +22 -2
- package/dist/parsers/RtfParser.js +1398 -1282
- package/dist/parsers/WordParser.d.ts +3 -2
- package/dist/parsers/WordParser.js +333 -115
- package/dist/sbom.cdx.json +103 -103
- package/dist/types.d.ts +833 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +28 -9
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Main generator class providing document conversion functionality.
|
|
4
|
+
*/
|
|
5
|
+
export declare class OfficeGenerator {
|
|
6
|
+
/**
|
|
7
|
+
* Generates a file of the specified type from an AST.
|
|
8
|
+
* This is the single source of truth for generation logic.
|
|
9
|
+
*
|
|
10
|
+
* @param ast - The OfficeParserAST to generate from
|
|
11
|
+
* @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
|
|
12
|
+
* @param config - Optional configuration for the generator
|
|
13
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages
|
|
14
|
+
* @throws {Error} If the destination format is unsupported
|
|
15
|
+
*/
|
|
16
|
+
static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
|
|
17
|
+
type: T;
|
|
18
|
+
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
|
|
19
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.OfficeGenerator = void 0;
|
|
4
|
+
const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
|
|
5
|
+
const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
|
|
6
|
+
const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
|
|
7
|
+
const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
|
|
8
|
+
const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
|
|
9
|
+
const RtfGenerator_js_1 = require("./generators/RtfGenerator.js");
|
|
10
|
+
const TextGenerator_js_1 = require("./generators/TextGenerator.js");
|
|
11
|
+
const types_js_1 = require("./types.js");
|
|
12
|
+
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
13
|
+
/**
|
|
14
|
+
* Main generator class providing document conversion functionality.
|
|
15
|
+
*/
|
|
16
|
+
class OfficeGenerator {
|
|
17
|
+
/**
|
|
18
|
+
* Generates a file of the specified type from an AST.
|
|
19
|
+
* This is the single source of truth for generation logic.
|
|
20
|
+
*
|
|
21
|
+
* @param ast - The OfficeParserAST to generate from
|
|
22
|
+
* @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
|
|
23
|
+
* @param config - Optional configuration for the generator
|
|
24
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages
|
|
25
|
+
* @throws {Error} If the destination format is unsupported
|
|
26
|
+
*/
|
|
27
|
+
static async generate(ast, destination, config) {
|
|
28
|
+
switch (destination.toLowerCase()) {
|
|
29
|
+
case 'text':
|
|
30
|
+
return new TextGenerator_js_1.TextGenerator(ast, config).generate();
|
|
31
|
+
case 'md':
|
|
32
|
+
return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
|
|
33
|
+
case 'html':
|
|
34
|
+
return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
|
|
35
|
+
case 'pdf':
|
|
36
|
+
return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
|
|
37
|
+
case 'csv':
|
|
38
|
+
return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
|
|
39
|
+
case 'rtf':
|
|
40
|
+
return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
|
|
41
|
+
case 'chunks':
|
|
42
|
+
return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
|
|
43
|
+
default:
|
|
44
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
exports.OfficeGenerator = OfficeGenerator;
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -11,6 +11,9 @@
|
|
|
11
11
|
* - ODT, ODP, ODS (OpenDocument formats)
|
|
12
12
|
* - PDF (Portable Document Format)
|
|
13
13
|
* - RTF (Rich Text Format)
|
|
14
|
+
* - CSV (Comma-Separated Values)
|
|
15
|
+
* - MD (Markdown)
|
|
16
|
+
* - HTML (HyperText Markup Language)
|
|
14
17
|
*
|
|
15
18
|
* **Usage:**
|
|
16
19
|
* ```typescript
|
|
@@ -60,6 +63,9 @@ export declare class OfficeParser {
|
|
|
60
63
|
* - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
|
|
61
64
|
* - `.pdf` → PdfParser (PDF.js)
|
|
62
65
|
* - `.rtf` → RtfParser (custom RTF parser)
|
|
66
|
+
* - `.csv` → CsvParser
|
|
67
|
+
* - `.md` → MarkdownParser
|
|
68
|
+
* - `.html` → HtmlParser
|
|
63
69
|
*
|
|
64
70
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
65
71
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|
package/dist/OfficeParser.js
CHANGED
|
@@ -12,6 +12,9 @@
|
|
|
12
12
|
* - ODT, ODP, ODS (OpenDocument formats)
|
|
13
13
|
* - PDF (Portable Document Format)
|
|
14
14
|
* - RTF (Rich Text Format)
|
|
15
|
+
* - CSV (Comma-Separated Values)
|
|
16
|
+
* - MD (Markdown)
|
|
17
|
+
* - HTML (HyperText Markup Language)
|
|
15
18
|
*
|
|
16
19
|
* **Usage:**
|
|
17
20
|
* ```typescript
|
|
@@ -35,13 +38,18 @@
|
|
|
35
38
|
*/
|
|
36
39
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
37
40
|
exports.OfficeParser = void 0;
|
|
38
|
-
const
|
|
41
|
+
const CsvParser_js_1 = require("./parsers/CsvParser.js");
|
|
39
42
|
const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
|
|
43
|
+
const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
|
|
44
|
+
const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
|
|
40
45
|
const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
|
|
41
46
|
const PdfParser_js_1 = require("./parsers/PdfParser.js");
|
|
42
47
|
const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
|
|
43
48
|
const RtfParser_js_1 = require("./parsers/RtfParser.js");
|
|
44
49
|
const WordParser_js_1 = require("./parsers/WordParser.js");
|
|
50
|
+
const types_js_1 = require("./types.js");
|
|
51
|
+
const configUtils_js_1 = require("./utils/configUtils.js");
|
|
52
|
+
const envUtils_js_1 = require("./utils/envUtils.js");
|
|
45
53
|
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
46
54
|
const moduleLoader_js_1 = require("./utils/moduleLoader.js");
|
|
47
55
|
const ocrUtils_js_1 = require("./utils/ocrUtils.js");
|
|
@@ -72,6 +80,9 @@ class OfficeParser {
|
|
|
72
80
|
* - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
|
|
73
81
|
* - `.pdf` → PdfParser (PDF.js)
|
|
74
82
|
* - `.rtf` → RtfParser (custom RTF parser)
|
|
83
|
+
* - `.csv` → CsvParser
|
|
84
|
+
* - `.md` → MarkdownParser
|
|
85
|
+
* - `.html` → HtmlParser
|
|
75
86
|
*
|
|
76
87
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
77
88
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|
|
@@ -107,27 +118,20 @@ class OfficeParser {
|
|
|
107
118
|
else {
|
|
108
119
|
actualConfig = configOrCallback || {};
|
|
109
120
|
}
|
|
110
|
-
const internalConfig =
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
ocrLanguage: 'eng',
|
|
118
|
-
includeRawContent: false,
|
|
119
|
-
serializeRawContent: true,
|
|
120
|
-
preserveXmlWhitespace: false,
|
|
121
|
-
pdfWorkerSrc: '',
|
|
122
|
-
ocrConfig: {},
|
|
123
|
-
...actualConfig
|
|
121
|
+
const internalConfig = (0, configUtils_js_1.resolveParserConfig)(actualConfig);
|
|
122
|
+
const parsingWarnings = [];
|
|
123
|
+
const originalOnWarning = internalConfig.onWarning;
|
|
124
|
+
internalConfig.onWarning = (issue) => {
|
|
125
|
+
parsingWarnings.push(issue);
|
|
126
|
+
if (originalOnWarning)
|
|
127
|
+
originalOnWarning(issue);
|
|
124
128
|
};
|
|
125
129
|
let buffer = Buffer.alloc(0);
|
|
126
|
-
let ext = '';
|
|
130
|
+
let ext = internalConfig.fileType ?? '';
|
|
127
131
|
let filePath;
|
|
128
132
|
try {
|
|
129
133
|
if (!file) {
|
|
130
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
134
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
|
|
131
135
|
}
|
|
132
136
|
if (file instanceof ArrayBuffer) {
|
|
133
137
|
buffer = Buffer.from(file);
|
|
@@ -143,29 +147,42 @@ class OfficeParser {
|
|
|
143
147
|
// shim 'fs' so it won't crash at build time.
|
|
144
148
|
const fs = await import('fs');
|
|
145
149
|
if (!fs.existsSync(file)) {
|
|
146
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
150
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
|
|
147
151
|
}
|
|
148
152
|
if (fs.lstatSync(file).isDirectory()) {
|
|
149
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
153
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
|
|
150
154
|
}
|
|
151
155
|
buffer = fs.readFileSync(file);
|
|
152
|
-
ext = file.split('.').pop()
|
|
156
|
+
ext = ext || file.split('.').pop() || '';
|
|
153
157
|
}
|
|
154
158
|
else {
|
|
155
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
159
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
|
|
156
160
|
}
|
|
157
|
-
if
|
|
161
|
+
// Always attempt to detect file type from buffer if it exists,
|
|
162
|
+
// but respect the authoritative 'ext' if it was already set.
|
|
163
|
+
if (buffer.length > 0) {
|
|
158
164
|
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
159
165
|
const type = await fileTypeFromBuffer(buffer);
|
|
160
|
-
if (
|
|
161
|
-
|
|
166
|
+
if (!ext) {
|
|
167
|
+
if (type) {
|
|
168
|
+
ext = type.ext;
|
|
169
|
+
}
|
|
170
|
+
else {
|
|
171
|
+
// If no extension could be detected and none was provided,
|
|
172
|
+
// it might be a text-based format (csv, md, html) which
|
|
173
|
+
// lack magic bytes. We'll let the switch default handle it.
|
|
174
|
+
}
|
|
162
175
|
}
|
|
163
|
-
else {
|
|
164
|
-
|
|
176
|
+
else if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
|
|
177
|
+
// Mismatch found between authoritative extension and detected content
|
|
178
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
|
|
165
179
|
}
|
|
166
180
|
}
|
|
181
|
+
if (!ext) {
|
|
182
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
|
|
183
|
+
}
|
|
167
184
|
let result;
|
|
168
|
-
switch (ext) {
|
|
185
|
+
switch (ext.toLowerCase()) {
|
|
169
186
|
case 'docx':
|
|
170
187
|
result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
|
|
171
188
|
break;
|
|
@@ -186,9 +203,19 @@ class OfficeParser {
|
|
|
186
203
|
case 'rtf':
|
|
187
204
|
result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
|
|
188
205
|
break;
|
|
206
|
+
case 'csv':
|
|
207
|
+
result = await (0, CsvParser_js_1.parseCsv)(buffer, internalConfig);
|
|
208
|
+
break;
|
|
209
|
+
case 'html':
|
|
210
|
+
result = await (0, HtmlParser_js_1.parseHtml)(buffer, internalConfig);
|
|
211
|
+
break;
|
|
212
|
+
case 'md':
|
|
213
|
+
result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
|
|
214
|
+
break;
|
|
189
215
|
default:
|
|
190
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
216
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
|
|
191
217
|
}
|
|
218
|
+
result.warnings = parsingWarnings;
|
|
192
219
|
if (callback)
|
|
193
220
|
callback(result);
|
|
194
221
|
return result;
|
package/dist/cli.d.ts
CHANGED
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
9
9
|
*
|
|
10
10
|
* Options (--key=value):
|
|
11
|
-
* --
|
|
11
|
+
* --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
|
|
12
|
+
* --output=path Save result to a file
|
|
13
|
+
* --toText=true Legacy flag for plain text output
|
|
12
14
|
* --ocr=true Enable OCR for images
|
|
13
15
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
14
16
|
* --extractAttachments=true Extract embedded attachments
|
package/dist/cli.js
CHANGED
|
@@ -9,7 +9,9 @@
|
|
|
9
9
|
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
10
10
|
*
|
|
11
11
|
* Options (--key=value):
|
|
12
|
-
* --
|
|
12
|
+
* --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
|
|
13
|
+
* --output=path Save result to a file
|
|
14
|
+
* --toText=true Legacy flag for plain text output
|
|
13
15
|
* --ocr=true Enable OCR for images
|
|
14
16
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
15
17
|
* --extractAttachments=true Extract embedded attachments
|
|
@@ -18,12 +20,49 @@
|
|
|
18
20
|
* --includeRawContent=true Include raw content in AST
|
|
19
21
|
* --outputErrorToConsole=true Log errors to console
|
|
20
22
|
*/
|
|
23
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
24
|
+
if (k2 === undefined) k2 = k;
|
|
25
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
26
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
27
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
28
|
+
}
|
|
29
|
+
Object.defineProperty(o, k2, desc);
|
|
30
|
+
}) : (function(o, m, k, k2) {
|
|
31
|
+
if (k2 === undefined) k2 = k;
|
|
32
|
+
o[k2] = m[k];
|
|
33
|
+
}));
|
|
34
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
35
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
36
|
+
}) : function(o, v) {
|
|
37
|
+
o["default"] = v;
|
|
38
|
+
});
|
|
39
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
40
|
+
var ownKeys = function(o) {
|
|
41
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
42
|
+
var ar = [];
|
|
43
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
44
|
+
return ar;
|
|
45
|
+
};
|
|
46
|
+
return ownKeys(o);
|
|
47
|
+
};
|
|
48
|
+
return function (mod) {
|
|
49
|
+
if (mod && mod.__esModule) return mod;
|
|
50
|
+
var result = {};
|
|
51
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
52
|
+
__setModuleDefault(result, mod);
|
|
53
|
+
return result;
|
|
54
|
+
};
|
|
55
|
+
})();
|
|
21
56
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
22
57
|
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
58
|
+
const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
|
|
59
|
+
const fs = __importStar(require("fs"));
|
|
23
60
|
const args = process.argv.slice(2);
|
|
24
61
|
let fileArg;
|
|
25
62
|
let toText = false;
|
|
26
63
|
let verbose = false;
|
|
64
|
+
let outputFormat;
|
|
65
|
+
let outputFile;
|
|
27
66
|
const configArgs = [];
|
|
28
67
|
function isConfigOption(arg) {
|
|
29
68
|
return arg.startsWith('--') && arg.includes('=');
|
|
@@ -43,34 +82,75 @@ if (fileArg) {
|
|
|
43
82
|
const cleanKey = key.replace('--', '');
|
|
44
83
|
const lowerValue = value.toLowerCase();
|
|
45
84
|
const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
85
|
+
const knownBooleans = new Set([
|
|
86
|
+
'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'putNotesAtLast',
|
|
87
|
+
'includeRawContent', 'outputErrorToConsole', 'serializeRawContent',
|
|
88
|
+
'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
|
|
89
|
+
]);
|
|
90
|
+
if (cleanKey === 'format') {
|
|
91
|
+
outputFormat = value;
|
|
51
92
|
}
|
|
52
|
-
else if (cleanKey === '
|
|
53
|
-
|
|
54
|
-
verbose = boolValue;
|
|
55
|
-
else
|
|
56
|
-
console.warn(`Invalid value for verbose: ${value}`);
|
|
93
|
+
else if (cleanKey === 'output') {
|
|
94
|
+
outputFile = value;
|
|
57
95
|
}
|
|
58
96
|
else {
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
97
|
+
if (boolValue !== undefined) {
|
|
98
|
+
if (cleanKey === 'toText')
|
|
99
|
+
toText = boolValue;
|
|
100
|
+
else if (cleanKey === 'verbose') {
|
|
101
|
+
verbose = boolValue;
|
|
102
|
+
if (verbose)
|
|
103
|
+
config.outputErrorToConsole = true;
|
|
104
|
+
}
|
|
105
|
+
else {
|
|
106
|
+
// @ts-ignore
|
|
107
|
+
config[cleanKey] = boolValue;
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
else if (knownBooleans.has(cleanKey)) {
|
|
111
|
+
console.warn(`Invalid boolean value for --${cleanKey}: ${value}. Using default.`);
|
|
112
|
+
}
|
|
113
|
+
else {
|
|
114
|
+
// @ts-ignore
|
|
64
115
|
config[cleanKey] = value;
|
|
116
|
+
}
|
|
65
117
|
}
|
|
66
118
|
});
|
|
67
119
|
OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
|
|
68
120
|
.then(async (ast) => {
|
|
69
|
-
|
|
70
|
-
|
|
121
|
+
let output;
|
|
122
|
+
if (outputFormat) {
|
|
123
|
+
const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat);
|
|
124
|
+
if (Array.isArray(result.value)) {
|
|
125
|
+
output = JSON.stringify(result.value, null, 2);
|
|
126
|
+
}
|
|
127
|
+
else {
|
|
128
|
+
output = result.value;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
else if (toText) {
|
|
132
|
+
output = ast.toText();
|
|
133
|
+
}
|
|
134
|
+
else {
|
|
135
|
+
output = JSON.stringify(ast, null, 2);
|
|
136
|
+
}
|
|
137
|
+
if (outputFile) {
|
|
138
|
+
if (output instanceof Uint8Array) {
|
|
139
|
+
fs.writeFileSync(outputFile, output);
|
|
140
|
+
}
|
|
141
|
+
else {
|
|
142
|
+
fs.writeFileSync(outputFile, output, 'utf8');
|
|
143
|
+
}
|
|
144
|
+
if (verbose)
|
|
145
|
+
console.log(`Output written to ${outputFile}`);
|
|
71
146
|
}
|
|
72
147
|
else {
|
|
73
|
-
|
|
148
|
+
if (output instanceof Uint8Array) {
|
|
149
|
+
process.stdout.write(output);
|
|
150
|
+
}
|
|
151
|
+
else {
|
|
152
|
+
process.stdout.write(output + '\n');
|
|
153
|
+
}
|
|
74
154
|
}
|
|
75
155
|
// Ensure OCR workers are terminated for clean CLI exit
|
|
76
156
|
if (config.ocr) {
|
|
@@ -97,7 +177,9 @@ else {
|
|
|
97
177
|
console.log('Usage: officeparser <file> [--option=value]');
|
|
98
178
|
console.log('');
|
|
99
179
|
console.log('Options:');
|
|
100
|
-
console.log(' --
|
|
180
|
+
console.log(' --format=json|md|html|rtf|csv|text|pdf|chunks Convert to specified format');
|
|
181
|
+
console.log(' --output=file.ext Save output to file instead of stdout');
|
|
182
|
+
console.log(' --toText=true Output plain text instead of JSON AST (legacy)');
|
|
101
183
|
console.log(' --ocr=true Enable OCR for images');
|
|
102
184
|
console.log(' --ocrLanguage=eng OCR language (default: eng)');
|
|
103
185
|
console.log(' --extractAttachments=true Extract embedded attachments');
|
|
@@ -106,11 +188,14 @@ else {
|
|
|
106
188
|
console.log(' --includeRawContent=true Include raw content in AST');
|
|
107
189
|
console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
|
|
108
190
|
console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
|
|
191
|
+
console.log(' --includeBreakNodes=false Include break nodes (DOCX only, default: false)');
|
|
109
192
|
console.log(' --verbose=true Show full error stack traces');
|
|
110
193
|
console.log('');
|
|
111
194
|
console.log('Examples:');
|
|
112
195
|
console.log(' officeparser document.docx');
|
|
113
|
-
console.log(' officeparser document.docx --
|
|
114
|
-
console.log(' officeparser
|
|
196
|
+
console.log(' officeparser document.docx --format=html --output=doc.html');
|
|
197
|
+
console.log(' officeparser document.docx --format=md');
|
|
198
|
+
console.log(' officeparser report.pdf --ocr=true --format=text');
|
|
199
|
+
console.log(' officeparser data.xlsx --format=csv --output=data.csv');
|
|
115
200
|
console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
|
|
116
201
|
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { DeepRequired, DocumentStructureChunkingConfig, FixedSizeChunkingConfig, FullGeneratorConfig, OfficeParserConfig, SemanticChunkingConfig } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* The default regex used for identifying sentence boundaries.
|
|
4
|
+
* When this default is used, the generator employs a high-fidelity "robust"
|
|
5
|
+
* segmenter that accounts for common abbreviations (Mr., Dr., etc.).
|
|
6
|
+
*/
|
|
7
|
+
export declare const DEFAULT_SENTENCE_BOUNDARY_REGEX: RegExp;
|
|
8
|
+
/**
|
|
9
|
+
* Common abbreviations that should not trigger a sentence split when followed by a period.
|
|
10
|
+
*/
|
|
11
|
+
export declare const DEFAULT_ABBREVIATIONS: string[];
|
|
12
|
+
/**
|
|
13
|
+
* Default configuration for the OfficeParser.
|
|
14
|
+
*/
|
|
15
|
+
export declare const DEFAULT_OFFICE_PARSER_CONFIG: DeepRequired<OfficeParserConfig>;
|
|
16
|
+
/**
|
|
17
|
+
* Default configuration for Fixed-Size chunking.
|
|
18
|
+
*/
|
|
19
|
+
export declare const DEFAULT_FIXED_SIZE_CHUNKING_CONFIG: Required<Omit<FixedSizeChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
|
|
20
|
+
sentenceBoundaryRegex: string | RegExp;
|
|
21
|
+
abbreviations: string[];
|
|
22
|
+
};
|
|
23
|
+
/**
|
|
24
|
+
* Default configuration for Document-Structure chunking.
|
|
25
|
+
*/
|
|
26
|
+
export declare const DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG: Required<Omit<DocumentStructureChunkingConfig, 'sentenceBoundaryRegex' | 'abbreviations'>> & {
|
|
27
|
+
sentenceBoundaryRegex: string | RegExp;
|
|
28
|
+
abbreviations: string[];
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* Default configuration for Semantic chunking.
|
|
32
|
+
* Note: `embeddingFunction` has no meaningful default and must be provided by the user.
|
|
33
|
+
*/
|
|
34
|
+
export declare const DEFAULT_SEMANTIC_CHUNKING_CONFIG: Required<Omit<SemanticChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
|
|
35
|
+
sentenceBoundaryRegex: string | RegExp;
|
|
36
|
+
abbreviations: string[];
|
|
37
|
+
};
|
|
38
|
+
/**
|
|
39
|
+
* Default configuration for the OfficeGenerator.
|
|
40
|
+
*/
|
|
41
|
+
export declare const DEFAULT_GENERATOR_CONFIG: FullGeneratorConfig;
|
package/dist/defaults.js
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.DEFAULT_GENERATOR_CONFIG = exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = exports.DEFAULT_OFFICE_PARSER_CONFIG = exports.DEFAULT_ABBREVIATIONS = exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = void 0;
|
|
4
|
+
const PDFJS_VERSION = '5.6.205';
|
|
5
|
+
const DEFAULT_PDF_WORKER_SRC = `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
|
|
6
|
+
/**
|
|
7
|
+
* The default regex used for identifying sentence boundaries.
|
|
8
|
+
* When this default is used, the generator employs a high-fidelity "robust"
|
|
9
|
+
* segmenter that accounts for common abbreviations (Mr., Dr., etc.).
|
|
10
|
+
*/
|
|
11
|
+
exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = /[.!?。!?]/;
|
|
12
|
+
/**
|
|
13
|
+
* Common abbreviations that should not trigger a sentence split when followed by a period.
|
|
14
|
+
*/
|
|
15
|
+
exports.DEFAULT_ABBREVIATIONS = ['Mr', 'Dr', 'Ms', 'Inc', 'Ltd', 'Prof', 'Sr', 'Jr', 'vs', 'etc'];
|
|
16
|
+
/**
|
|
17
|
+
* Default configuration for OCR.
|
|
18
|
+
*/
|
|
19
|
+
const DEFAULT_OCR_CONFIG = {
|
|
20
|
+
language: 'eng',
|
|
21
|
+
workerPath: '',
|
|
22
|
+
corePath: '',
|
|
23
|
+
langPath: '',
|
|
24
|
+
autoTerminateTimeout: 10000,
|
|
25
|
+
};
|
|
26
|
+
/**
|
|
27
|
+
* Default configuration for the OfficeParser.
|
|
28
|
+
*/
|
|
29
|
+
exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
30
|
+
outputErrorToConsole: false,
|
|
31
|
+
onWarning: () => { },
|
|
32
|
+
newlineDelimiter: '\n',
|
|
33
|
+
ignoreNotes: false,
|
|
34
|
+
putNotesAtLast: false,
|
|
35
|
+
extractAttachments: false,
|
|
36
|
+
includeRawContent: false,
|
|
37
|
+
ocr: false,
|
|
38
|
+
ocrLanguage: 'eng',
|
|
39
|
+
ocrConfig: DEFAULT_OCR_CONFIG,
|
|
40
|
+
serializeRawContent: true,
|
|
41
|
+
preserveXmlWhitespace: false,
|
|
42
|
+
pdfWorkerSrc: DEFAULT_PDF_WORKER_SRC,
|
|
43
|
+
includeBreakNodes: false,
|
|
44
|
+
ignoreInternalLinks: false,
|
|
45
|
+
fileType: null,
|
|
46
|
+
csvDelimiter: ',',
|
|
47
|
+
};
|
|
48
|
+
/**
|
|
49
|
+
* Default configuration for HTML generation.
|
|
50
|
+
*/
|
|
51
|
+
const DEFAULT_HTML_GENERATOR_CONFIG = {
|
|
52
|
+
standalone: true,
|
|
53
|
+
chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
|
|
54
|
+
};
|
|
55
|
+
/**
|
|
56
|
+
* Default configuration for PDF generation.
|
|
57
|
+
*/
|
|
58
|
+
const DEFAULT_PDF_GENERATOR_CONFIG = {
|
|
59
|
+
format: 'A4',
|
|
60
|
+
width: '',
|
|
61
|
+
height: '',
|
|
62
|
+
landscape: false,
|
|
63
|
+
printBackground: true,
|
|
64
|
+
scale: 1,
|
|
65
|
+
margin: {
|
|
66
|
+
top: 0,
|
|
67
|
+
right: 0,
|
|
68
|
+
bottom: 0,
|
|
69
|
+
left: 0
|
|
70
|
+
},
|
|
71
|
+
displayHeaderFooter: false,
|
|
72
|
+
headerTemplate: '',
|
|
73
|
+
footerTemplate: '',
|
|
74
|
+
launchOptions: {
|
|
75
|
+
headless: true,
|
|
76
|
+
args: ['--no-sandbox', '--disable-setuid-sandbox']
|
|
77
|
+
},
|
|
78
|
+
};
|
|
79
|
+
/**
|
|
80
|
+
* Default configuration for CSV generation.
|
|
81
|
+
*/
|
|
82
|
+
const DEFAULT_CSV_GENERATOR_CONFIG = {
|
|
83
|
+
sheets: '',
|
|
84
|
+
mergeSheets: true,
|
|
85
|
+
columnDelimiter: ',',
|
|
86
|
+
};
|
|
87
|
+
/**
|
|
88
|
+
* Default configuration for Markdown generation.
|
|
89
|
+
*/
|
|
90
|
+
const DEFAULT_MD_GENERATOR_CONFIG = {
|
|
91
|
+
fallbackToHtml: true,
|
|
92
|
+
};
|
|
93
|
+
/**
|
|
94
|
+
* Default configuration for plain text generation.
|
|
95
|
+
*/
|
|
96
|
+
const DEFAULT_TEXT_GENERATOR_CONFIG = {
|
|
97
|
+
newlineDelimiter: '\n',
|
|
98
|
+
preserveLayout: false,
|
|
99
|
+
};
|
|
100
|
+
/**
|
|
101
|
+
* Default configuration for Fixed-Size chunking.
|
|
102
|
+
*/
|
|
103
|
+
exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = {
|
|
104
|
+
strategy: 'fixed-size',
|
|
105
|
+
chunkSize: 1000,
|
|
106
|
+
chunkOverlap: 200,
|
|
107
|
+
separators: ['\n\n', '\n', ' ', ''],
|
|
108
|
+
stripWhitespace: true,
|
|
109
|
+
includeMetadata: true,
|
|
110
|
+
addStartIndex: false,
|
|
111
|
+
lengthFunction: (text) => text.length,
|
|
112
|
+
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
113
|
+
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
114
|
+
};
|
|
115
|
+
/**
|
|
116
|
+
* Default configuration for Document-Structure chunking.
|
|
117
|
+
*/
|
|
118
|
+
exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = {
|
|
119
|
+
strategy: 'document-structure',
|
|
120
|
+
splitBy: 'paragraph',
|
|
121
|
+
maxChunkSize: 1000,
|
|
122
|
+
tableSplitStrategy: 'row',
|
|
123
|
+
stripWhitespace: true,
|
|
124
|
+
includeMetadata: true,
|
|
125
|
+
addStartIndex: false,
|
|
126
|
+
lengthFunction: (text) => text.length,
|
|
127
|
+
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
128
|
+
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
129
|
+
};
|
|
130
|
+
/**
|
|
131
|
+
* Default configuration for Semantic chunking.
|
|
132
|
+
* Note: `embeddingFunction` has no meaningful default and must be provided by the user.
|
|
133
|
+
*/
|
|
134
|
+
exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = {
|
|
135
|
+
strategy: 'semantic',
|
|
136
|
+
similarityThreshold: 0.8,
|
|
137
|
+
maxChunkSize: 2000,
|
|
138
|
+
bufferSize: 1,
|
|
139
|
+
embeddingBatchSize: 50,
|
|
140
|
+
stripWhitespace: true,
|
|
141
|
+
includeMetadata: true,
|
|
142
|
+
addStartIndex: false,
|
|
143
|
+
lengthFunction: (text) => text.length,
|
|
144
|
+
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
145
|
+
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
146
|
+
};
|
|
147
|
+
/**
|
|
148
|
+
* The resolved default chunking config (uses document-structure as default strategy).
|
|
149
|
+
*/
|
|
150
|
+
const DEFAULT_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG;
|
|
151
|
+
/**
|
|
152
|
+
* Default configuration for the OfficeGenerator.
|
|
153
|
+
*/
|
|
154
|
+
exports.DEFAULT_GENERATOR_CONFIG = {
|
|
155
|
+
onNode: () => { },
|
|
156
|
+
onWarning: () => { },
|
|
157
|
+
styleMap: [],
|
|
158
|
+
includeFormatting: true,
|
|
159
|
+
generateIds: true,
|
|
160
|
+
renderMetadata: false,
|
|
161
|
+
ignoreDefaultStyleMap: false,
|
|
162
|
+
includeImages: true,
|
|
163
|
+
includeCharts: true,
|
|
164
|
+
ignoreInternalLinks: false,
|
|
165
|
+
htmlConfig: DEFAULT_HTML_GENERATOR_CONFIG,
|
|
166
|
+
mdConfig: DEFAULT_MD_GENERATOR_CONFIG,
|
|
167
|
+
pdfConfig: DEFAULT_PDF_GENERATOR_CONFIG,
|
|
168
|
+
csvConfig: DEFAULT_CSV_GENERATOR_CONFIG,
|
|
169
|
+
textConfig: DEFAULT_TEXT_GENERATOR_CONFIG,
|
|
170
|
+
rtfConfig: {},
|
|
171
|
+
chunksConfig: DEFAULT_CHUNKING_CONFIG,
|
|
172
|
+
};
|