officeparser 5.2.2 → 6.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +152 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +850 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -790
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Office Parser - Main Entry Point
|
|
3
|
+
*
|
|
4
|
+
* This module provides the main `OfficeParser` class with a single static method
|
|
5
|
+
* that automatically detects file types and routes to the appropriate parser.
|
|
6
|
+
*
|
|
7
|
+
* **Supported Formats:**
|
|
8
|
+
* - DOCX (Word documents)
|
|
9
|
+
* - XLSX (Excel spreadsheets)
|
|
10
|
+
* - PPTX (PowerPoint presentations)
|
|
11
|
+
* - ODT, ODP, ODS (OpenDocument formats)
|
|
12
|
+
* - PDF (Portable Document Format)
|
|
13
|
+
* - RTF (Rich Text Format)
|
|
14
|
+
*
|
|
15
|
+
* **Usage:**
|
|
16
|
+
* ```typescript
|
|
17
|
+
* import { OfficeParser } from 'officeparser';
|
|
18
|
+
*
|
|
19
|
+
* // Parse from file path
|
|
20
|
+
* const ast = await OfficeParser.parseOffice('document.docx', {
|
|
21
|
+
* extractAttachments: true,
|
|
22
|
+
* ocr: true
|
|
23
|
+
* });
|
|
24
|
+
*
|
|
25
|
+
* // Parse from Buffer
|
|
26
|
+
* const buffer = fs.readFileSync('document.pdf');
|
|
27
|
+
* const ast = await OfficeParser.parseOffice(buffer);
|
|
28
|
+
*
|
|
29
|
+
* // Get plain text
|
|
30
|
+
* console.log(ast.toText());
|
|
31
|
+
* ```
|
|
32
|
+
*
|
|
33
|
+
* @module OfficeParser
|
|
34
|
+
*/
|
|
35
|
+
/// <reference types="node" />
|
|
36
|
+
import { OfficeParserAST, OfficeParserConfig } from './types';
|
|
37
|
+
/**
|
|
38
|
+
* Main parser class providing office document parsing functionality.
|
|
39
|
+
*
|
|
40
|
+
* This class contains a single static method `parseOffice` that serves as the
|
|
41
|
+
* universal entry point for parsing any supported office document format.
|
|
42
|
+
*/
|
|
43
|
+
export declare class OfficeParser {
|
|
44
|
+
/**
|
|
45
|
+
* Parses an office document and returns a structured AST.
|
|
46
|
+
*
|
|
47
|
+
* This method:
|
|
48
|
+
* 1. Accepts a file path, Buffer, or ArrayBuffer
|
|
49
|
+
* 2. Detects the file type (from extension or content)
|
|
50
|
+
* 3. Routes to the appropriate format-specific parser
|
|
51
|
+
* 4. Returns a unified AST structure
|
|
52
|
+
*
|
|
53
|
+
* **File Type Detection:**
|
|
54
|
+
* - If a file path is provided, uses the file extension
|
|
55
|
+
* - If a Buffer is provided, uses magic bytes detection (file-type library)
|
|
56
|
+
*
|
|
57
|
+
* **Supported Formats and Routes:**
|
|
58
|
+
* - `.docx` → WordParser (OOXML)
|
|
59
|
+
* - `.xlsx` → ExcelParser (OOXML)
|
|
60
|
+
* - `.pptx` → PowerPointParser (OOXML)
|
|
61
|
+
* - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
|
|
62
|
+
* - `.pdf` → PdfParser (PDF.js)
|
|
63
|
+
* - `.rtf` → RtfParser (custom RTF parser)
|
|
64
|
+
*
|
|
65
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
66
|
+
* @param config - Optional configuration object (defaults applied for all omitted options)
|
|
67
|
+
* @returns A promise resolving to the parsed OfficeParserAST
|
|
68
|
+
* @throws {Error} If file doesn't exist, format is unsupported, or parsing fails
|
|
69
|
+
*
|
|
70
|
+
* @example
|
|
71
|
+
* ```typescript
|
|
72
|
+
* // Parse a DOCX file
|
|
73
|
+
* const ast = await OfficeParser.parseOffice('report.docx', {
|
|
74
|
+
* extractAttachments: true,
|
|
75
|
+
* includeRawContent: false
|
|
76
|
+
* });
|
|
77
|
+
*
|
|
78
|
+
* // Parse a Buffer with OCR enabled
|
|
79
|
+
* const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
|
|
80
|
+
* const ast = await OfficeParser.parseOffice(buffer, {
|
|
81
|
+
* ocr: true,
|
|
82
|
+
* ocrLanguage: 'eng+fra'
|
|
83
|
+
* });
|
|
84
|
+
*
|
|
85
|
+
* // Extract text
|
|
86
|
+
* const text = ast.toText();
|
|
87
|
+
* ```
|
|
88
|
+
*/
|
|
89
|
+
static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
90
|
+
}
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Office Parser - Main Entry Point
|
|
4
|
+
*
|
|
5
|
+
* This module provides the main `OfficeParser` class with a single static method
|
|
6
|
+
* that automatically detects file types and routes to the appropriate parser.
|
|
7
|
+
*
|
|
8
|
+
* **Supported Formats:**
|
|
9
|
+
* - DOCX (Word documents)
|
|
10
|
+
* - XLSX (Excel spreadsheets)
|
|
11
|
+
* - PPTX (PowerPoint presentations)
|
|
12
|
+
* - ODT, ODP, ODS (OpenDocument formats)
|
|
13
|
+
* - PDF (Portable Document Format)
|
|
14
|
+
* - RTF (Rich Text Format)
|
|
15
|
+
*
|
|
16
|
+
* **Usage:**
|
|
17
|
+
* ```typescript
|
|
18
|
+
* import { OfficeParser } from 'officeparser';
|
|
19
|
+
*
|
|
20
|
+
* // Parse from file path
|
|
21
|
+
* const ast = await OfficeParser.parseOffice('document.docx', {
|
|
22
|
+
* extractAttachments: true,
|
|
23
|
+
* ocr: true
|
|
24
|
+
* });
|
|
25
|
+
*
|
|
26
|
+
* // Parse from Buffer
|
|
27
|
+
* const buffer = fs.readFileSync('document.pdf');
|
|
28
|
+
* const ast = await OfficeParser.parseOffice(buffer);
|
|
29
|
+
*
|
|
30
|
+
* // Get plain text
|
|
31
|
+
* console.log(ast.toText());
|
|
32
|
+
* ```
|
|
33
|
+
*
|
|
34
|
+
* @module OfficeParser
|
|
35
|
+
*/
|
|
36
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
37
|
+
if (k2 === undefined) k2 = k;
|
|
38
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
39
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
40
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
41
|
+
}
|
|
42
|
+
Object.defineProperty(o, k2, desc);
|
|
43
|
+
}) : (function(o, m, k, k2) {
|
|
44
|
+
if (k2 === undefined) k2 = k;
|
|
45
|
+
o[k2] = m[k];
|
|
46
|
+
}));
|
|
47
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
48
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
49
|
+
}) : function(o, v) {
|
|
50
|
+
o["default"] = v;
|
|
51
|
+
});
|
|
52
|
+
var __importStar = (this && this.__importStar) || function (mod) {
|
|
53
|
+
if (mod && mod.__esModule) return mod;
|
|
54
|
+
var result = {};
|
|
55
|
+
if (mod != null) for (var k in mod) if (k !== "default" && Object.prototype.hasOwnProperty.call(mod, k)) __createBinding(result, mod, k);
|
|
56
|
+
__setModuleDefault(result, mod);
|
|
57
|
+
return result;
|
|
58
|
+
};
|
|
59
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
60
|
+
exports.OfficeParser = void 0;
|
|
61
|
+
const fileType = __importStar(require("file-type"));
|
|
62
|
+
const fs = __importStar(require("fs"));
|
|
63
|
+
const ExcelParser_1 = require("./parsers/ExcelParser");
|
|
64
|
+
const OpenOfficeParser_1 = require("./parsers/OpenOfficeParser");
|
|
65
|
+
const PdfParser_1 = require("./parsers/PdfParser");
|
|
66
|
+
const PowerPointParser_1 = require("./parsers/PowerPointParser");
|
|
67
|
+
const RtfParser_1 = require("./parsers/RtfParser");
|
|
68
|
+
const WordParser_1 = require("./parsers/WordParser");
|
|
69
|
+
const errorUtils_1 = require("./utils/errorUtils");
|
|
70
|
+
/**
|
|
71
|
+
* Main parser class providing office document parsing functionality.
|
|
72
|
+
*
|
|
73
|
+
* This class contains a single static method `parseOffice` that serves as the
|
|
74
|
+
* universal entry point for parsing any supported office document format.
|
|
75
|
+
*/
|
|
76
|
+
class OfficeParser {
|
|
77
|
+
/**
|
|
78
|
+
* Parses an office document and returns a structured AST.
|
|
79
|
+
*
|
|
80
|
+
* This method:
|
|
81
|
+
* 1. Accepts a file path, Buffer, or ArrayBuffer
|
|
82
|
+
* 2. Detects the file type (from extension or content)
|
|
83
|
+
* 3. Routes to the appropriate format-specific parser
|
|
84
|
+
* 4. Returns a unified AST structure
|
|
85
|
+
*
|
|
86
|
+
* **File Type Detection:**
|
|
87
|
+
* - If a file path is provided, uses the file extension
|
|
88
|
+
* - If a Buffer is provided, uses magic bytes detection (file-type library)
|
|
89
|
+
*
|
|
90
|
+
* **Supported Formats and Routes:**
|
|
91
|
+
* - `.docx` → WordParser (OOXML)
|
|
92
|
+
* - `.xlsx` → ExcelParser (OOXML)
|
|
93
|
+
* - `.pptx` → PowerPointParser (OOXML)
|
|
94
|
+
* - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
|
|
95
|
+
* - `.pdf` → PdfParser (PDF.js)
|
|
96
|
+
* - `.rtf` → RtfParser (custom RTF parser)
|
|
97
|
+
*
|
|
98
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
99
|
+
* @param config - Optional configuration object (defaults applied for all omitted options)
|
|
100
|
+
* @returns A promise resolving to the parsed OfficeParserAST
|
|
101
|
+
* @throws {Error} If file doesn't exist, format is unsupported, or parsing fails
|
|
102
|
+
*
|
|
103
|
+
* @example
|
|
104
|
+
* ```typescript
|
|
105
|
+
* // Parse a DOCX file
|
|
106
|
+
* const ast = await OfficeParser.parseOffice('report.docx', {
|
|
107
|
+
* extractAttachments: true,
|
|
108
|
+
* includeRawContent: false
|
|
109
|
+
* });
|
|
110
|
+
*
|
|
111
|
+
* // Parse a Buffer with OCR enabled
|
|
112
|
+
* const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
|
|
113
|
+
* const ast = await OfficeParser.parseOffice(buffer, {
|
|
114
|
+
* ocr: true,
|
|
115
|
+
* ocrLanguage: 'eng+fra'
|
|
116
|
+
* });
|
|
117
|
+
*
|
|
118
|
+
* // Extract text
|
|
119
|
+
* const text = ast.toText();
|
|
120
|
+
* ```
|
|
121
|
+
*/
|
|
122
|
+
static async parseOffice(file, configOrCallback, config) {
|
|
123
|
+
let callback;
|
|
124
|
+
let actualConfig = {};
|
|
125
|
+
if (typeof configOrCallback === 'function') {
|
|
126
|
+
callback = configOrCallback;
|
|
127
|
+
actualConfig = config || {};
|
|
128
|
+
}
|
|
129
|
+
else {
|
|
130
|
+
actualConfig = configOrCallback || {};
|
|
131
|
+
}
|
|
132
|
+
const internalConfig = {
|
|
133
|
+
ignoreNotes: false,
|
|
134
|
+
newlineDelimiter: '\n',
|
|
135
|
+
putNotesAtLast: false,
|
|
136
|
+
outputErrorToConsole: false,
|
|
137
|
+
extractAttachments: false,
|
|
138
|
+
ocr: false,
|
|
139
|
+
ocrLanguage: 'eng',
|
|
140
|
+
includeRawContent: false,
|
|
141
|
+
pdfWorkerSrc: '',
|
|
142
|
+
...actualConfig
|
|
143
|
+
};
|
|
144
|
+
let buffer = Buffer.alloc(0);
|
|
145
|
+
let ext = '';
|
|
146
|
+
let filePath;
|
|
147
|
+
try {
|
|
148
|
+
if (!file) {
|
|
149
|
+
throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
|
|
150
|
+
}
|
|
151
|
+
if (file instanceof ArrayBuffer) {
|
|
152
|
+
buffer = Buffer.from(file);
|
|
153
|
+
}
|
|
154
|
+
else if (Buffer.isBuffer(file)) {
|
|
155
|
+
buffer = file;
|
|
156
|
+
}
|
|
157
|
+
else if (typeof file === 'string') {
|
|
158
|
+
filePath = file;
|
|
159
|
+
if (!fs.existsSync(file)) {
|
|
160
|
+
throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
|
|
161
|
+
}
|
|
162
|
+
if (fs.lstatSync(file).isDirectory()) {
|
|
163
|
+
throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
|
|
164
|
+
}
|
|
165
|
+
buffer = fs.readFileSync(file);
|
|
166
|
+
ext = file.split('.').pop()?.toLowerCase() || '';
|
|
167
|
+
}
|
|
168
|
+
else {
|
|
169
|
+
throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.INVALID_INPUT, internalConfig);
|
|
170
|
+
}
|
|
171
|
+
if (!ext) {
|
|
172
|
+
const type = await fileType.fromBuffer(buffer);
|
|
173
|
+
if (type) {
|
|
174
|
+
ext = type.ext.toLowerCase();
|
|
175
|
+
}
|
|
176
|
+
else {
|
|
177
|
+
throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
let result;
|
|
181
|
+
switch (ext) {
|
|
182
|
+
case 'docx':
|
|
183
|
+
result = await (0, WordParser_1.parseWord)(buffer, internalConfig);
|
|
184
|
+
break;
|
|
185
|
+
case 'pptx':
|
|
186
|
+
result = await (0, PowerPointParser_1.parsePowerPoint)(buffer, internalConfig);
|
|
187
|
+
break;
|
|
188
|
+
case 'xlsx':
|
|
189
|
+
result = await (0, ExcelParser_1.parseExcel)(buffer, internalConfig);
|
|
190
|
+
break;
|
|
191
|
+
case 'odt':
|
|
192
|
+
case 'odp':
|
|
193
|
+
case 'ods':
|
|
194
|
+
result = await (0, OpenOfficeParser_1.parseOpenOffice)(buffer, internalConfig);
|
|
195
|
+
break;
|
|
196
|
+
case 'pdf':
|
|
197
|
+
result = await (0, PdfParser_1.parsePdf)(buffer, internalConfig);
|
|
198
|
+
break;
|
|
199
|
+
case 'rtf':
|
|
200
|
+
result = await (0, RtfParser_1.parseRtf)(buffer, internalConfig);
|
|
201
|
+
break;
|
|
202
|
+
default:
|
|
203
|
+
throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
|
|
204
|
+
}
|
|
205
|
+
if (callback)
|
|
206
|
+
callback(result);
|
|
207
|
+
return result;
|
|
208
|
+
}
|
|
209
|
+
catch (error) {
|
|
210
|
+
const wrappedError = (0, errorUtils_1.getWrappedError)(error, internalConfig, filePath);
|
|
211
|
+
if (callback)
|
|
212
|
+
callback(undefined, wrappedError);
|
|
213
|
+
throw wrappedError;
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
exports.OfficeParser = OfficeParser;
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* officeparser - Universal Office Document Parser
|
|
4
|
+
*
|
|
5
|
+
* A comprehensive Node.js library for parsing Microsoft Office and OpenDocument files
|
|
6
|
+
* into structured Abstract Syntax Trees (AST) with full formatting information.
|
|
7
|
+
*
|
|
8
|
+
* **Supported Formats:**
|
|
9
|
+
* - Microsoft Office: DOCX, XLSX, PPTX (Office Open XML)
|
|
10
|
+
* - OpenDocument: ODT, ODP, ODS (ODF)
|
|
11
|
+
* - Legacy: RTF (Rich Text Format)
|
|
12
|
+
* - Portable: PDF
|
|
13
|
+
*
|
|
14
|
+
* **Key Features:**
|
|
15
|
+
* - Unified AST output across all formats
|
|
16
|
+
* - Rich text formatting (bold, italic, colors, fonts, etc.)
|
|
17
|
+
* - Document structure (headings, lists, tables)
|
|
18
|
+
* - Image extraction with optional OCR
|
|
19
|
+
* - Metadata extraction
|
|
20
|
+
* - TypeScript support with full type definitions
|
|
21
|
+
*
|
|
22
|
+
* **Quick Start:**
|
|
23
|
+
* ```typescript
|
|
24
|
+
* import { OfficeParser } from 'officeparser';
|
|
25
|
+
*
|
|
26
|
+
* const ast = await OfficeParser.parseOffice('document.docx', {
|
|
27
|
+
* extractAttachments: true,
|
|
28
|
+
* ocr: true,
|
|
29
|
+
* includeRawContent: false
|
|
30
|
+
* });
|
|
31
|
+
*
|
|
32
|
+
* console.log(ast.toText()); // Plain text output
|
|
33
|
+
* console.log(ast.content); // Structured content tree
|
|
34
|
+
* console.log(ast.metadata); // Document metadata
|
|
35
|
+
* ```
|
|
36
|
+
*
|
|
37
|
+
* **Main Exports:**
|
|
38
|
+
* - `OfficeParser` - Main parser class
|
|
39
|
+
* - `OfficeParserConfig` - Configuration interface
|
|
40
|
+
* - `OfficeParserAST` - AST result interface
|
|
41
|
+
* - `OfficeContentNode` - Content tree node interface
|
|
42
|
+
* - All type definitions
|
|
43
|
+
*
|
|
44
|
+
* @packageDocumentation
|
|
45
|
+
* @module officeparser
|
|
46
|
+
*/
|
|
47
|
+
import { OfficeParser } from './OfficeParser';
|
|
48
|
+
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata } from './types';
|
|
49
|
+
declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
50
|
+
export { OfficeParser, parseOffice, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata };
|
|
51
|
+
export default OfficeParser;
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
"use strict";
|
|
3
|
+
/**
|
|
4
|
+
* officeparser - Universal Office Document Parser
|
|
5
|
+
*
|
|
6
|
+
* A comprehensive Node.js library for parsing Microsoft Office and OpenDocument files
|
|
7
|
+
* into structured Abstract Syntax Trees (AST) with full formatting information.
|
|
8
|
+
*
|
|
9
|
+
* **Supported Formats:**
|
|
10
|
+
* - Microsoft Office: DOCX, XLSX, PPTX (Office Open XML)
|
|
11
|
+
* - OpenDocument: ODT, ODP, ODS (ODF)
|
|
12
|
+
* - Legacy: RTF (Rich Text Format)
|
|
13
|
+
* - Portable: PDF
|
|
14
|
+
*
|
|
15
|
+
* **Key Features:**
|
|
16
|
+
* - Unified AST output across all formats
|
|
17
|
+
* - Rich text formatting (bold, italic, colors, fonts, etc.)
|
|
18
|
+
* - Document structure (headings, lists, tables)
|
|
19
|
+
* - Image extraction with optional OCR
|
|
20
|
+
* - Metadata extraction
|
|
21
|
+
* - TypeScript support with full type definitions
|
|
22
|
+
*
|
|
23
|
+
* **Quick Start:**
|
|
24
|
+
* ```typescript
|
|
25
|
+
* import { OfficeParser } from 'officeparser';
|
|
26
|
+
*
|
|
27
|
+
* const ast = await OfficeParser.parseOffice('document.docx', {
|
|
28
|
+
* extractAttachments: true,
|
|
29
|
+
* ocr: true,
|
|
30
|
+
* includeRawContent: false
|
|
31
|
+
* });
|
|
32
|
+
*
|
|
33
|
+
* console.log(ast.toText()); // Plain text output
|
|
34
|
+
* console.log(ast.content); // Structured content tree
|
|
35
|
+
* console.log(ast.metadata); // Document metadata
|
|
36
|
+
* ```
|
|
37
|
+
*
|
|
38
|
+
* **Main Exports:**
|
|
39
|
+
* - `OfficeParser` - Main parser class
|
|
40
|
+
* - `OfficeParserConfig` - Configuration interface
|
|
41
|
+
* - `OfficeParserAST` - AST result interface
|
|
42
|
+
* - `OfficeContentNode` - Content tree node interface
|
|
43
|
+
* - All type definitions
|
|
44
|
+
*
|
|
45
|
+
* @packageDocumentation
|
|
46
|
+
* @module officeparser
|
|
47
|
+
*/
|
|
48
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
49
|
+
exports.parseOffice = exports.OfficeParser = void 0;
|
|
50
|
+
const OfficeParser_1 = require("./OfficeParser");
|
|
51
|
+
Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_1.OfficeParser; } });
|
|
52
|
+
const parseOffice = OfficeParser_1.OfficeParser.parseOffice;
|
|
53
|
+
exports.parseOffice = parseOffice;
|
|
54
|
+
// Default export for backward compatibility
|
|
55
|
+
exports.default = OfficeParser_1.OfficeParser;
|
|
56
|
+
// CLI handling - allows running as: node index.js file.docx
|
|
57
|
+
if (typeof require !== 'undefined' && typeof module !== 'undefined' && require.main === module) {
|
|
58
|
+
const args = process.argv.slice(2);
|
|
59
|
+
let fileArg;
|
|
60
|
+
let toText = false;
|
|
61
|
+
const configArgs = [];
|
|
62
|
+
function isConfigOption(arg) {
|
|
63
|
+
return arg.startsWith('--') && arg.includes('=');
|
|
64
|
+
}
|
|
65
|
+
args.forEach(arg => {
|
|
66
|
+
if (isConfigOption(arg)) {
|
|
67
|
+
configArgs.push(arg);
|
|
68
|
+
}
|
|
69
|
+
else if (!fileArg) {
|
|
70
|
+
fileArg = arg;
|
|
71
|
+
}
|
|
72
|
+
});
|
|
73
|
+
if (fileArg) {
|
|
74
|
+
const config = {};
|
|
75
|
+
configArgs.forEach(arg => {
|
|
76
|
+
const [key, value] = arg.split('=');
|
|
77
|
+
const cleanKey = key.replace('--', '');
|
|
78
|
+
if (cleanKey === 'toText') {
|
|
79
|
+
if (value.toLowerCase() === 'true')
|
|
80
|
+
toText = true;
|
|
81
|
+
else if (value.toLowerCase() === 'false')
|
|
82
|
+
toText = false;
|
|
83
|
+
else
|
|
84
|
+
console.log(`Invalid value for toText: ${value}`);
|
|
85
|
+
}
|
|
86
|
+
// @ts-ignore
|
|
87
|
+
else if (value.toLowerCase() === 'true')
|
|
88
|
+
config[cleanKey] = true;
|
|
89
|
+
// @ts-ignore
|
|
90
|
+
else if (value.toLowerCase() === 'false')
|
|
91
|
+
config[cleanKey] = false;
|
|
92
|
+
// @ts-ignore
|
|
93
|
+
else
|
|
94
|
+
config[cleanKey] = value;
|
|
95
|
+
});
|
|
96
|
+
OfficeParser_1.OfficeParser.parseOffice(fileArg, config)
|
|
97
|
+
.then((ast) => {
|
|
98
|
+
if (toText)
|
|
99
|
+
console.log(ast.toText());
|
|
100
|
+
else
|
|
101
|
+
console.log(JSON.stringify(ast, null, 2));
|
|
102
|
+
})
|
|
103
|
+
.catch(console.error);
|
|
104
|
+
}
|
|
105
|
+
else {
|
|
106
|
+
console.log("Usage: node officeparser [file] [--option=value]");
|
|
107
|
+
}
|
|
108
|
+
}
|