officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/dist/index.d.ts
CHANGED
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
1
|
/**
|
|
3
2
|
* officeparser - Universal Office Document Parser
|
|
4
3
|
*
|
|
@@ -44,8 +43,9 @@
|
|
|
44
43
|
* @packageDocumentation
|
|
45
44
|
* @module officeparser
|
|
46
45
|
*/
|
|
47
|
-
import { OfficeParser } from './OfficeParser';
|
|
48
|
-
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata } from './types';
|
|
46
|
+
import { OfficeParser } from './OfficeParser.js';
|
|
47
|
+
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata } from './types';
|
|
49
48
|
declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
50
|
-
|
|
49
|
+
declare const terminateOcr: typeof OfficeParser.terminateOcr;
|
|
50
|
+
export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, };
|
|
51
51
|
export default OfficeParser;
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
1
|
"use strict";
|
|
3
2
|
/**
|
|
4
3
|
* officeparser - Universal Office Document Parser
|
|
@@ -46,63 +45,12 @@
|
|
|
46
45
|
* @module officeparser
|
|
47
46
|
*/
|
|
48
47
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
49
|
-
exports.parseOffice = exports.OfficeParser = void 0;
|
|
50
|
-
const
|
|
51
|
-
Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return
|
|
52
|
-
const parseOffice =
|
|
48
|
+
exports.terminateOcr = exports.parseOffice = exports.OfficeParser = void 0;
|
|
49
|
+
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
50
|
+
Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_js_1.OfficeParser; } });
|
|
51
|
+
const parseOffice = OfficeParser_js_1.OfficeParser.parseOffice;
|
|
53
52
|
exports.parseOffice = parseOffice;
|
|
53
|
+
const terminateOcr = OfficeParser_js_1.OfficeParser.terminateOcr;
|
|
54
|
+
exports.terminateOcr = terminateOcr;
|
|
54
55
|
// Default export for backward compatibility
|
|
55
|
-
exports.default =
|
|
56
|
-
// CLI handling - allows running as: node index.js file.docx
|
|
57
|
-
if (typeof require !== 'undefined' && typeof module !== 'undefined' && require.main === module) {
|
|
58
|
-
const args = process.argv.slice(2);
|
|
59
|
-
let fileArg;
|
|
60
|
-
let toText = false;
|
|
61
|
-
const configArgs = [];
|
|
62
|
-
function isConfigOption(arg) {
|
|
63
|
-
return arg.startsWith('--') && arg.includes('=');
|
|
64
|
-
}
|
|
65
|
-
args.forEach(arg => {
|
|
66
|
-
if (isConfigOption(arg)) {
|
|
67
|
-
configArgs.push(arg);
|
|
68
|
-
}
|
|
69
|
-
else if (!fileArg) {
|
|
70
|
-
fileArg = arg;
|
|
71
|
-
}
|
|
72
|
-
});
|
|
73
|
-
if (fileArg) {
|
|
74
|
-
const config = {};
|
|
75
|
-
configArgs.forEach(arg => {
|
|
76
|
-
const [key, value] = arg.split('=');
|
|
77
|
-
const cleanKey = key.replace('--', '');
|
|
78
|
-
if (cleanKey === 'toText') {
|
|
79
|
-
if (value.toLowerCase() === 'true')
|
|
80
|
-
toText = true;
|
|
81
|
-
else if (value.toLowerCase() === 'false')
|
|
82
|
-
toText = false;
|
|
83
|
-
else
|
|
84
|
-
console.log(`Invalid value for toText: ${value}`);
|
|
85
|
-
}
|
|
86
|
-
// @ts-ignore
|
|
87
|
-
else if (value.toLowerCase() === 'true')
|
|
88
|
-
config[cleanKey] = true;
|
|
89
|
-
// @ts-ignore
|
|
90
|
-
else if (value.toLowerCase() === 'false')
|
|
91
|
-
config[cleanKey] = false;
|
|
92
|
-
// @ts-ignore
|
|
93
|
-
else
|
|
94
|
-
config[cleanKey] = value;
|
|
95
|
-
});
|
|
96
|
-
OfficeParser_1.OfficeParser.parseOffice(fileArg, config)
|
|
97
|
-
.then((ast) => {
|
|
98
|
-
if (toText)
|
|
99
|
-
console.log(ast.toText());
|
|
100
|
-
else
|
|
101
|
-
console.log(JSON.stringify(ast, null, 2));
|
|
102
|
-
})
|
|
103
|
-
.catch(console.error);
|
|
104
|
-
}
|
|
105
|
-
else {
|
|
106
|
-
console.log("Usage: node officeparser [file] [--option=value]");
|
|
107
|
-
}
|
|
108
|
-
}
|
|
56
|
+
exports.default = OfficeParser_js_1.OfficeParser;
|
package/dist/index.mjs
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ESM wrapper for officeparser
|
|
3
|
+
*
|
|
4
|
+
* AUTO-GENERATED — do not edit manually.
|
|
5
|
+
* Generated by scripts/generate-esm-wrapper.js during build.
|
|
6
|
+
*
|
|
7
|
+
* This file re-exports from the CJS build (dist/index.js) to provide
|
|
8
|
+
* proper ESM named exports without duplicating the source code.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import _module from './index.js';
|
|
12
|
+
|
|
13
|
+
// Named exports
|
|
14
|
+
const { OfficeParser, parseOffice, terminateOcr } = _module;
|
|
15
|
+
export { OfficeParser, parseOffice, terminateOcr };
|
|
16
|
+
|
|
17
|
+
// Default export
|
|
18
|
+
export default _module.default ?? _module;
|
|
@@ -1,5 +1,42 @@
|
|
|
1
1
|
// Generated by dts-bundle-generator v9.5.1
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* Configuration options for OCR.
|
|
5
|
+
*/
|
|
6
|
+
export interface OcrConfig {
|
|
7
|
+
/**
|
|
8
|
+
* Language for OCR.
|
|
9
|
+
* Default is 'eng'.
|
|
10
|
+
*
|
|
11
|
+
* You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
|
|
12
|
+
* The OCR engine will then attempt to recognize text in any of the specified languages.
|
|
13
|
+
*
|
|
14
|
+
* See the list of supported languages and their codes here:
|
|
15
|
+
* https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
|
|
16
|
+
*/
|
|
17
|
+
language?: string;
|
|
18
|
+
/**
|
|
19
|
+
* Path to the Tesseract worker script.
|
|
20
|
+
* Primarily used for offline/air-gapped environments.
|
|
21
|
+
*/
|
|
22
|
+
workerPath?: string;
|
|
23
|
+
/**
|
|
24
|
+
* Path to the Tesseract core script.
|
|
25
|
+
* Primarily used for offline/air-gapped environments.
|
|
26
|
+
*/
|
|
27
|
+
corePath?: string;
|
|
28
|
+
/**
|
|
29
|
+
* Path for Tesseract language files (traineddata).
|
|
30
|
+
* Primarily used for offline/air-gapped environments.
|
|
31
|
+
*/
|
|
32
|
+
langPath?: string;
|
|
33
|
+
/**
|
|
34
|
+
* Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
|
|
35
|
+
* Set to 0 to disable auto-termination.
|
|
36
|
+
* Default is 10,000 (10 seconds).
|
|
37
|
+
*/
|
|
38
|
+
autoTerminateTimeout?: number;
|
|
39
|
+
}
|
|
3
40
|
/**
|
|
4
41
|
* Configuration options for the OfficeParser.
|
|
5
42
|
*/
|
|
@@ -42,6 +79,7 @@ export interface OfficeParserConfig {
|
|
|
42
79
|
*/
|
|
43
80
|
ocr?: boolean;
|
|
44
81
|
/**
|
|
82
|
+
* @deprecated Use `ocrConfig.language` instead.
|
|
45
83
|
* Language for OCR.
|
|
46
84
|
* Default is 'eng'.
|
|
47
85
|
*
|
|
@@ -52,14 +90,41 @@ export interface OfficeParserConfig {
|
|
|
52
90
|
* https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
|
|
53
91
|
*/
|
|
54
92
|
ocrLanguage?: string;
|
|
93
|
+
/**
|
|
94
|
+
* Shared OCR configuration for worker pooling and offline support.
|
|
95
|
+
* If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
|
|
96
|
+
*/
|
|
97
|
+
ocrConfig?: OcrConfig;
|
|
98
|
+
/**
|
|
99
|
+
* Flag to serialize raw content (XML) as clean, formatted strings.
|
|
100
|
+
* Only relevant when `includeRawContent` is true.
|
|
101
|
+
* Default is true.
|
|
102
|
+
*
|
|
103
|
+
* If false, the parser will attempt to extract the original raw substring from the
|
|
104
|
+
* source document instead of re-serializing the DOM node.
|
|
105
|
+
*/
|
|
106
|
+
serializeRawContent?: boolean;
|
|
107
|
+
/**
|
|
108
|
+
* Flag to preserve original XML whitespace and line endings when serializing.
|
|
109
|
+
* Only relevant when `includeRawContent` is true and `serializeRawContent` is true.
|
|
110
|
+
* Default is false.
|
|
111
|
+
*/
|
|
112
|
+
preserveXmlWhitespace?: boolean;
|
|
55
113
|
/**
|
|
56
114
|
* The URL/path to the PDF.js worker script.
|
|
57
115
|
*
|
|
58
116
|
* **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
|
|
59
|
-
* If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.
|
|
117
|
+
* If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
|
|
60
118
|
* You can override this with your own local path or a different CDN link.
|
|
61
119
|
*/
|
|
62
120
|
pdfWorkerSrc?: string;
|
|
121
|
+
/**
|
|
122
|
+
* Flag to include break nodes in the AST.
|
|
123
|
+
* This is currently only supported for Word documents. (w:br nodes)
|
|
124
|
+
*
|
|
125
|
+
* Default is false
|
|
126
|
+
*/
|
|
127
|
+
includeBreakNodes?: boolean;
|
|
63
128
|
}
|
|
64
129
|
/**
|
|
65
130
|
* Supported file types for parsing.
|
|
@@ -68,7 +133,7 @@ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods"
|
|
|
68
133
|
/**
|
|
69
134
|
* Types of content nodes in the AST.
|
|
70
135
|
*/
|
|
71
|
-
export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page";
|
|
136
|
+
export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break";
|
|
72
137
|
/**
|
|
73
138
|
* Supported MIME types for attachments.
|
|
74
139
|
*/
|
|
@@ -171,6 +236,20 @@ export interface SheetMetadata {
|
|
|
171
236
|
/** The style of the sheet. */
|
|
172
237
|
style?: string;
|
|
173
238
|
}
|
|
239
|
+
/**
|
|
240
|
+
* Detailed indentation information for paragraphs and headings.
|
|
241
|
+
* Values are typically in twentieths of a point (twips) in OOXML.
|
|
242
|
+
*/
|
|
243
|
+
export interface IndentationMetadata {
|
|
244
|
+
/** Left indentation. */
|
|
245
|
+
left?: number;
|
|
246
|
+
/** Right indentation. */
|
|
247
|
+
right?: number;
|
|
248
|
+
/** First line indentation. */
|
|
249
|
+
firstLine?: number;
|
|
250
|
+
/** Hanging indentation. */
|
|
251
|
+
hanging?: number;
|
|
252
|
+
}
|
|
174
253
|
/**
|
|
175
254
|
* Metadata for a heading.
|
|
176
255
|
*/
|
|
@@ -181,6 +260,8 @@ export interface HeadingMetadata {
|
|
|
181
260
|
alignment?: "left" | "center" | "right" | "justify";
|
|
182
261
|
/** The style of the heading. */
|
|
183
262
|
style?: string;
|
|
263
|
+
/** Detailed indentation information. */
|
|
264
|
+
paragraphIndentation?: IndentationMetadata;
|
|
184
265
|
}
|
|
185
266
|
/**
|
|
186
267
|
* Metadata for a paragraph.
|
|
@@ -190,6 +271,8 @@ export interface ParagraphMetadata {
|
|
|
190
271
|
alignment?: "left" | "center" | "right" | "justify";
|
|
191
272
|
/** The style of the paragraph. */
|
|
192
273
|
style?: string;
|
|
274
|
+
/** Detailed indentation information. */
|
|
275
|
+
paragraphIndentation?: IndentationMetadata;
|
|
193
276
|
}
|
|
194
277
|
/**
|
|
195
278
|
* Metadata for a list item.
|
|
@@ -205,6 +288,8 @@ export interface ListMetadata {
|
|
|
205
288
|
* @example 0 for top-level items, 1 for first nested level
|
|
206
289
|
*/
|
|
207
290
|
indentation: number;
|
|
291
|
+
/** Detailed indentation information. */
|
|
292
|
+
paragraphIndentation?: IndentationMetadata;
|
|
208
293
|
/**
|
|
209
294
|
* Text alignment of the list item.
|
|
210
295
|
* @example 'left', 'center', 'right', 'justify'
|
|
@@ -331,10 +416,35 @@ export interface NoteMetadata {
|
|
|
331
416
|
*/
|
|
332
417
|
noteId?: string;
|
|
333
418
|
}
|
|
419
|
+
/**
|
|
420
|
+
* Metadata for break nodes.
|
|
421
|
+
* Used in DOCX files to track line and page breaks.
|
|
422
|
+
*/
|
|
423
|
+
export interface BreakMetadata {
|
|
424
|
+
/**
|
|
425
|
+
* Type of break. The break type determines the next location where
|
|
426
|
+
* text shall be placed.
|
|
427
|
+
* - 'column': The next text will be placed in the next column.
|
|
428
|
+
* - 'page': The next text will be placed on the next page.
|
|
429
|
+
* - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
|
|
430
|
+
* - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
|
|
431
|
+
* - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
|
|
432
|
+
*/
|
|
433
|
+
breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn";
|
|
434
|
+
/**
|
|
435
|
+
* Specifies the location which shall be used as the next available line when breakType
|
|
436
|
+
* has a value of 'textWrapping'. Should be ignored for other break types.
|
|
437
|
+
* - 'all': text wrapping break shall advance the text to the next line which spans the full width of the line
|
|
438
|
+
* - 'left': text wrapping break shall restart in next text region unblocked on the left
|
|
439
|
+
* - 'none': text wrapping break shall advance the text to the next line regardless of any floating objects
|
|
440
|
+
* - 'right': text wrapping break shall restart in next text region unblocked on the right
|
|
441
|
+
*/
|
|
442
|
+
clear?: "all" | "left" | "none" | "right";
|
|
443
|
+
}
|
|
334
444
|
/**
|
|
335
445
|
* Union type for content metadata.
|
|
336
446
|
*/
|
|
337
|
-
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | undefined;
|
|
447
|
+
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | undefined;
|
|
338
448
|
/**
|
|
339
449
|
* Represents a node in the document content tree.
|
|
340
450
|
* This is the core building block of the parsed document structure.
|
|
@@ -542,6 +652,16 @@ export interface OfficeMetadata {
|
|
|
542
652
|
formatting?: Partial<TextFormatting>;
|
|
543
653
|
/** Style map for styles in the document. */
|
|
544
654
|
styleMap?: Record<string, Partial<TextFormatting>>;
|
|
655
|
+
/**
|
|
656
|
+
* User-defined custom properties embedded in the document.
|
|
657
|
+
* Sources by format:
|
|
658
|
+
* - DOCX/XLSX/PPTX: `docProps/custom.xml` (Office custom document properties)
|
|
659
|
+
* - ODT/ODP/ODS: `meta:user-defined` elements in `meta.xml`
|
|
660
|
+
* - PDF: non-standard entries in the PDF Info dictionary
|
|
661
|
+
* RTF does not support custom properties; the `\info` group is not extracted.
|
|
662
|
+
* Values are typed as string, number, boolean, or Date where the source format provides type information.
|
|
663
|
+
*/
|
|
664
|
+
customProperties?: Record<string, string | number | boolean | Date>;
|
|
545
665
|
}
|
|
546
666
|
/**
|
|
547
667
|
* The Abstract Syntax Tree (AST) returned by the parser.
|
|
@@ -668,8 +788,18 @@ export declare class OfficeParser {
|
|
|
668
788
|
* ```
|
|
669
789
|
*/
|
|
670
790
|
static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
791
|
+
/**
|
|
792
|
+
* Terminates all active OCR workers and cleans up resources.
|
|
793
|
+
*
|
|
794
|
+
* This should be called when the application is shutting down or when OCR
|
|
795
|
+
* is no longer needed to prevent memory leaks and orphaned worker processes.
|
|
796
|
+
*
|
|
797
|
+
* @returns A promise that resolves when all workers have been terminated
|
|
798
|
+
*/
|
|
799
|
+
static terminateOcr(): Promise<void>;
|
|
671
800
|
}
|
|
672
801
|
export declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
802
|
+
export declare const terminateOcr: typeof OfficeParser.terminateOcr;
|
|
673
803
|
|
|
674
804
|
export {
|
|
675
805
|
OfficeParser as default,
|