officeparser 5.2.2 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,90 @@
1
+ /**
2
+ * Office Parser - Main Entry Point
3
+ *
4
+ * This module provides the main `OfficeParser` class with a single static method
5
+ * that automatically detects file types and routes to the appropriate parser.
6
+ *
7
+ * **Supported Formats:**
8
+ * - DOCX (Word documents)
9
+ * - XLSX (Excel spreadsheets)
10
+ * - PPTX (PowerPoint presentations)
11
+ * - ODT, ODP, ODS (OpenDocument formats)
12
+ * - PDF (Portable Document Format)
13
+ * - RTF (Rich Text Format)
14
+ *
15
+ * **Usage:**
16
+ * ```typescript
17
+ * import { OfficeParser } from 'officeparser';
18
+ *
19
+ * // Parse from file path
20
+ * const ast = await OfficeParser.parseOffice('document.docx', {
21
+ * extractAttachments: true,
22
+ * ocr: true
23
+ * });
24
+ *
25
+ * // Parse from Buffer
26
+ * const buffer = fs.readFileSync('document.pdf');
27
+ * const ast = await OfficeParser.parseOffice(buffer);
28
+ *
29
+ * // Get plain text
30
+ * console.log(ast.toText());
31
+ * ```
32
+ *
33
+ * @module OfficeParser
34
+ */
35
+ /// <reference types="node" />
36
+ import { OfficeParserAST, OfficeParserConfig } from './types';
37
+ /**
38
+ * Main parser class providing office document parsing functionality.
39
+ *
40
+ * This class contains a single static method `parseOffice` that serves as the
41
+ * universal entry point for parsing any supported office document format.
42
+ */
43
+ export declare class OfficeParser {
44
+ /**
45
+ * Parses an office document and returns a structured AST.
46
+ *
47
+ * This method:
48
+ * 1. Accepts a file path, Buffer, or ArrayBuffer
49
+ * 2. Detects the file type (from extension or content)
50
+ * 3. Routes to the appropriate format-specific parser
51
+ * 4. Returns a unified AST structure
52
+ *
53
+ * **File Type Detection:**
54
+ * - If a file path is provided, uses the file extension
55
+ * - If a Buffer is provided, uses magic bytes detection (file-type library)
56
+ *
57
+ * **Supported Formats and Routes:**
58
+ * - `.docx` → WordParser (OOXML)
59
+ * - `.xlsx` → ExcelParser (OOXML)
60
+ * - `.pptx` → PowerPointParser (OOXML)
61
+ * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
62
+ * - `.pdf` → PdfParser (PDF.js)
63
+ * - `.rtf` → RtfParser (custom RTF parser)
64
+ *
65
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the document
66
+ * @param config - Optional configuration object (defaults applied for all omitted options)
67
+ * @returns A promise resolving to the parsed OfficeParserAST
68
+ * @throws {Error} If file doesn't exist, format is unsupported, or parsing fails
69
+ *
70
+ * @example
71
+ * ```typescript
72
+ * // Parse a DOCX file
73
+ * const ast = await OfficeParser.parseOffice('report.docx', {
74
+ * extractAttachments: true,
75
+ * includeRawContent: false
76
+ * });
77
+ *
78
+ * // Parse a Buffer with OCR enabled
79
+ * const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
80
+ * const ast = await OfficeParser.parseOffice(buffer, {
81
+ * ocr: true,
82
+ * ocrLanguage: 'eng+fra'
83
+ * });
84
+ *
85
+ * // Extract text
86
+ * const text = ast.toText();
87
+ * ```
88
+ */
89
+ static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
90
+ }
@@ -0,0 +1,217 @@
1
+ "use strict";
2
+ /**
3
+ * Office Parser - Main Entry Point
4
+ *
5
+ * This module provides the main `OfficeParser` class with a single static method
6
+ * that automatically detects file types and routes to the appropriate parser.
7
+ *
8
+ * **Supported Formats:**
9
+ * - DOCX (Word documents)
10
+ * - XLSX (Excel spreadsheets)
11
+ * - PPTX (PowerPoint presentations)
12
+ * - ODT, ODP, ODS (OpenDocument formats)
13
+ * - PDF (Portable Document Format)
14
+ * - RTF (Rich Text Format)
15
+ *
16
+ * **Usage:**
17
+ * ```typescript
18
+ * import { OfficeParser } from 'officeparser';
19
+ *
20
+ * // Parse from file path
21
+ * const ast = await OfficeParser.parseOffice('document.docx', {
22
+ * extractAttachments: true,
23
+ * ocr: true
24
+ * });
25
+ *
26
+ * // Parse from Buffer
27
+ * const buffer = fs.readFileSync('document.pdf');
28
+ * const ast = await OfficeParser.parseOffice(buffer);
29
+ *
30
+ * // Get plain text
31
+ * console.log(ast.toText());
32
+ * ```
33
+ *
34
+ * @module OfficeParser
35
+ */
36
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
37
+ if (k2 === undefined) k2 = k;
38
+ var desc = Object.getOwnPropertyDescriptor(m, k);
39
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
40
+ desc = { enumerable: true, get: function() { return m[k]; } };
41
+ }
42
+ Object.defineProperty(o, k2, desc);
43
+ }) : (function(o, m, k, k2) {
44
+ if (k2 === undefined) k2 = k;
45
+ o[k2] = m[k];
46
+ }));
47
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
48
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
49
+ }) : function(o, v) {
50
+ o["default"] = v;
51
+ });
52
+ var __importStar = (this && this.__importStar) || function (mod) {
53
+ if (mod && mod.__esModule) return mod;
54
+ var result = {};
55
+ if (mod != null) for (var k in mod) if (k !== "default" && Object.prototype.hasOwnProperty.call(mod, k)) __createBinding(result, mod, k);
56
+ __setModuleDefault(result, mod);
57
+ return result;
58
+ };
59
+ Object.defineProperty(exports, "__esModule", { value: true });
60
+ exports.OfficeParser = void 0;
61
+ const fileType = __importStar(require("file-type"));
62
+ const fs = __importStar(require("fs"));
63
+ const ExcelParser_1 = require("./parsers/ExcelParser");
64
+ const OpenOfficeParser_1 = require("./parsers/OpenOfficeParser");
65
+ const PdfParser_1 = require("./parsers/PdfParser");
66
+ const PowerPointParser_1 = require("./parsers/PowerPointParser");
67
+ const RtfParser_1 = require("./parsers/RtfParser");
68
+ const WordParser_1 = require("./parsers/WordParser");
69
+ const errorUtils_1 = require("./utils/errorUtils");
70
+ /**
71
+ * Main parser class providing office document parsing functionality.
72
+ *
73
+ * This class contains a single static method `parseOffice` that serves as the
74
+ * universal entry point for parsing any supported office document format.
75
+ */
76
+ class OfficeParser {
77
+ /**
78
+ * Parses an office document and returns a structured AST.
79
+ *
80
+ * This method:
81
+ * 1. Accepts a file path, Buffer, or ArrayBuffer
82
+ * 2. Detects the file type (from extension or content)
83
+ * 3. Routes to the appropriate format-specific parser
84
+ * 4. Returns a unified AST structure
85
+ *
86
+ * **File Type Detection:**
87
+ * - If a file path is provided, uses the file extension
88
+ * - If a Buffer is provided, uses magic bytes detection (file-type library)
89
+ *
90
+ * **Supported Formats and Routes:**
91
+ * - `.docx` → WordParser (OOXML)
92
+ * - `.xlsx` → ExcelParser (OOXML)
93
+ * - `.pptx` → PowerPointParser (OOXML)
94
+ * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
95
+ * - `.pdf` → PdfParser (PDF.js)
96
+ * - `.rtf` → RtfParser (custom RTF parser)
97
+ *
98
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the document
99
+ * @param config - Optional configuration object (defaults applied for all omitted options)
100
+ * @returns A promise resolving to the parsed OfficeParserAST
101
+ * @throws {Error} If file doesn't exist, format is unsupported, or parsing fails
102
+ *
103
+ * @example
104
+ * ```typescript
105
+ * // Parse a DOCX file
106
+ * const ast = await OfficeParser.parseOffice('report.docx', {
107
+ * extractAttachments: true,
108
+ * includeRawContent: false
109
+ * });
110
+ *
111
+ * // Parse a Buffer with OCR enabled
112
+ * const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
113
+ * const ast = await OfficeParser.parseOffice(buffer, {
114
+ * ocr: true,
115
+ * ocrLanguage: 'eng+fra'
116
+ * });
117
+ *
118
+ * // Extract text
119
+ * const text = ast.toText();
120
+ * ```
121
+ */
122
+ static async parseOffice(file, configOrCallback, config) {
123
+ let callback;
124
+ let actualConfig = {};
125
+ if (typeof configOrCallback === 'function') {
126
+ callback = configOrCallback;
127
+ actualConfig = config || {};
128
+ }
129
+ else {
130
+ actualConfig = configOrCallback || {};
131
+ }
132
+ const internalConfig = {
133
+ ignoreNotes: false,
134
+ newlineDelimiter: '\n',
135
+ putNotesAtLast: false,
136
+ outputErrorToConsole: false,
137
+ extractAttachments: false,
138
+ ocr: false,
139
+ ocrLanguage: 'eng',
140
+ includeRawContent: false,
141
+ pdfWorkerSrc: '',
142
+ ...actualConfig
143
+ };
144
+ let buffer = Buffer.alloc(0);
145
+ let ext = '';
146
+ let filePath;
147
+ try {
148
+ if (!file) {
149
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
150
+ }
151
+ if (file instanceof ArrayBuffer) {
152
+ buffer = Buffer.from(file);
153
+ }
154
+ else if (Buffer.isBuffer(file)) {
155
+ buffer = file;
156
+ }
157
+ else if (typeof file === 'string') {
158
+ filePath = file;
159
+ if (!fs.existsSync(file)) {
160
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
161
+ }
162
+ if (fs.lstatSync(file).isDirectory()) {
163
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
164
+ }
165
+ buffer = fs.readFileSync(file);
166
+ ext = file.split('.').pop()?.toLowerCase() || '';
167
+ }
168
+ else {
169
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.INVALID_INPUT, internalConfig);
170
+ }
171
+ if (!ext) {
172
+ const type = await fileType.fromBuffer(buffer);
173
+ if (type) {
174
+ ext = type.ext.toLowerCase();
175
+ }
176
+ else {
177
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
178
+ }
179
+ }
180
+ let result;
181
+ switch (ext) {
182
+ case 'docx':
183
+ result = await (0, WordParser_1.parseWord)(buffer, internalConfig);
184
+ break;
185
+ case 'pptx':
186
+ result = await (0, PowerPointParser_1.parsePowerPoint)(buffer, internalConfig);
187
+ break;
188
+ case 'xlsx':
189
+ result = await (0, ExcelParser_1.parseExcel)(buffer, internalConfig);
190
+ break;
191
+ case 'odt':
192
+ case 'odp':
193
+ case 'ods':
194
+ result = await (0, OpenOfficeParser_1.parseOpenOffice)(buffer, internalConfig);
195
+ break;
196
+ case 'pdf':
197
+ result = await (0, PdfParser_1.parsePdf)(buffer, internalConfig);
198
+ break;
199
+ case 'rtf':
200
+ result = await (0, RtfParser_1.parseRtf)(buffer, internalConfig);
201
+ break;
202
+ default:
203
+ throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
204
+ }
205
+ if (callback)
206
+ callback(result);
207
+ return result;
208
+ }
209
+ catch (error) {
210
+ const wrappedError = (0, errorUtils_1.getWrappedError)(error, internalConfig, filePath);
211
+ if (callback)
212
+ callback(undefined, wrappedError);
213
+ throw wrappedError;
214
+ }
215
+ }
216
+ }
217
+ exports.OfficeParser = OfficeParser;
@@ -0,0 +1,51 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * officeparser - Universal Office Document Parser
4
+ *
5
+ * A comprehensive Node.js library for parsing Microsoft Office and OpenDocument files
6
+ * into structured Abstract Syntax Trees (AST) with full formatting information.
7
+ *
8
+ * **Supported Formats:**
9
+ * - Microsoft Office: DOCX, XLSX, PPTX (Office Open XML)
10
+ * - OpenDocument: ODT, ODP, ODS (ODF)
11
+ * - Legacy: RTF (Rich Text Format)
12
+ * - Portable: PDF
13
+ *
14
+ * **Key Features:**
15
+ * - Unified AST output across all formats
16
+ * - Rich text formatting (bold, italic, colors, fonts, etc.)
17
+ * - Document structure (headings, lists, tables)
18
+ * - Image extraction with optional OCR
19
+ * - Metadata extraction
20
+ * - TypeScript support with full type definitions
21
+ *
22
+ * **Quick Start:**
23
+ * ```typescript
24
+ * import { OfficeParser } from 'officeparser';
25
+ *
26
+ * const ast = await OfficeParser.parseOffice('document.docx', {
27
+ * extractAttachments: true,
28
+ * ocr: true,
29
+ * includeRawContent: false
30
+ * });
31
+ *
32
+ * console.log(ast.toText()); // Plain text output
33
+ * console.log(ast.content); // Structured content tree
34
+ * console.log(ast.metadata); // Document metadata
35
+ * ```
36
+ *
37
+ * **Main Exports:**
38
+ * - `OfficeParser` - Main parser class
39
+ * - `OfficeParserConfig` - Configuration interface
40
+ * - `OfficeParserAST` - AST result interface
41
+ * - `OfficeContentNode` - Content tree node interface
42
+ * - All type definitions
43
+ *
44
+ * @packageDocumentation
45
+ * @module officeparser
46
+ */
47
+ import { OfficeParser } from './OfficeParser';
48
+ import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata } from './types';
49
+ declare const parseOffice: typeof OfficeParser.parseOffice;
50
+ export { OfficeParser, parseOffice, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata };
51
+ export default OfficeParser;
package/dist/index.js ADDED
@@ -0,0 +1,108 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * officeparser - Universal Office Document Parser
5
+ *
6
+ * A comprehensive Node.js library for parsing Microsoft Office and OpenDocument files
7
+ * into structured Abstract Syntax Trees (AST) with full formatting information.
8
+ *
9
+ * **Supported Formats:**
10
+ * - Microsoft Office: DOCX, XLSX, PPTX (Office Open XML)
11
+ * - OpenDocument: ODT, ODP, ODS (ODF)
12
+ * - Legacy: RTF (Rich Text Format)
13
+ * - Portable: PDF
14
+ *
15
+ * **Key Features:**
16
+ * - Unified AST output across all formats
17
+ * - Rich text formatting (bold, italic, colors, fonts, etc.)
18
+ * - Document structure (headings, lists, tables)
19
+ * - Image extraction with optional OCR
20
+ * - Metadata extraction
21
+ * - TypeScript support with full type definitions
22
+ *
23
+ * **Quick Start:**
24
+ * ```typescript
25
+ * import { OfficeParser } from 'officeparser';
26
+ *
27
+ * const ast = await OfficeParser.parseOffice('document.docx', {
28
+ * extractAttachments: true,
29
+ * ocr: true,
30
+ * includeRawContent: false
31
+ * });
32
+ *
33
+ * console.log(ast.toText()); // Plain text output
34
+ * console.log(ast.content); // Structured content tree
35
+ * console.log(ast.metadata); // Document metadata
36
+ * ```
37
+ *
38
+ * **Main Exports:**
39
+ * - `OfficeParser` - Main parser class
40
+ * - `OfficeParserConfig` - Configuration interface
41
+ * - `OfficeParserAST` - AST result interface
42
+ * - `OfficeContentNode` - Content tree node interface
43
+ * - All type definitions
44
+ *
45
+ * @packageDocumentation
46
+ * @module officeparser
47
+ */
48
+ Object.defineProperty(exports, "__esModule", { value: true });
49
+ exports.parseOffice = exports.OfficeParser = void 0;
50
+ const OfficeParser_1 = require("./OfficeParser");
51
+ Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_1.OfficeParser; } });
52
+ const parseOffice = OfficeParser_1.OfficeParser.parseOffice;
53
+ exports.parseOffice = parseOffice;
54
+ // Default export for backward compatibility
55
+ exports.default = OfficeParser_1.OfficeParser;
56
+ // CLI handling - allows running as: node index.js file.docx
57
+ if (typeof require !== 'undefined' && typeof module !== 'undefined' && require.main === module) {
58
+ const args = process.argv.slice(2);
59
+ let fileArg;
60
+ let toText = false;
61
+ const configArgs = [];
62
+ function isConfigOption(arg) {
63
+ return arg.startsWith('--') && arg.includes('=');
64
+ }
65
+ args.forEach(arg => {
66
+ if (isConfigOption(arg)) {
67
+ configArgs.push(arg);
68
+ }
69
+ else if (!fileArg) {
70
+ fileArg = arg;
71
+ }
72
+ });
73
+ if (fileArg) {
74
+ const config = {};
75
+ configArgs.forEach(arg => {
76
+ const [key, value] = arg.split('=');
77
+ const cleanKey = key.replace('--', '');
78
+ if (cleanKey === 'toText') {
79
+ if (value.toLowerCase() === 'true')
80
+ toText = true;
81
+ else if (value.toLowerCase() === 'false')
82
+ toText = false;
83
+ else
84
+ console.log(`Invalid value for toText: ${value}`);
85
+ }
86
+ // @ts-ignore
87
+ else if (value.toLowerCase() === 'true')
88
+ config[cleanKey] = true;
89
+ // @ts-ignore
90
+ else if (value.toLowerCase() === 'false')
91
+ config[cleanKey] = false;
92
+ // @ts-ignore
93
+ else
94
+ config[cleanKey] = value;
95
+ });
96
+ OfficeParser_1.OfficeParser.parseOffice(fileArg, config)
97
+ .then((ast) => {
98
+ if (toText)
99
+ console.log(ast.toText());
100
+ else
101
+ console.log(JSON.stringify(ast, null, 2));
102
+ })
103
+ .catch(console.error);
104
+ }
105
+ else {
106
+ console.log("Usage: node officeparser [file] [--option=value]");
107
+ }
108
+ }