officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
package/dist/index.d.ts CHANGED
@@ -1,4 +1,3 @@
1
- #!/usr/bin/env node
2
1
  /**
3
2
  * officeparser - Universal Office Document Parser
4
3
  *
@@ -44,8 +43,9 @@
44
43
  * @packageDocumentation
45
44
  * @module officeparser
46
45
  */
47
- import { OfficeParser } from './OfficeParser';
48
- import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata } from './types';
46
+ import { OfficeParser } from './OfficeParser.js';
47
+ import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata } from './types';
49
48
  declare const parseOffice: typeof OfficeParser.parseOffice;
50
- export { OfficeParser, parseOffice, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata };
49
+ declare const terminateOcr: typeof OfficeParser.terminateOcr;
50
+ export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, };
51
51
  export default OfficeParser;
package/dist/index.js CHANGED
@@ -1,4 +1,3 @@
1
- #!/usr/bin/env node
2
1
  "use strict";
3
2
  /**
4
3
  * officeparser - Universal Office Document Parser
@@ -46,63 +45,12 @@
46
45
  * @module officeparser
47
46
  */
48
47
  Object.defineProperty(exports, "__esModule", { value: true });
49
- exports.parseOffice = exports.OfficeParser = void 0;
50
- const OfficeParser_1 = require("./OfficeParser");
51
- Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_1.OfficeParser; } });
52
- const parseOffice = OfficeParser_1.OfficeParser.parseOffice;
48
+ exports.terminateOcr = exports.parseOffice = exports.OfficeParser = void 0;
49
+ const OfficeParser_js_1 = require("./OfficeParser.js");
50
+ Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_js_1.OfficeParser; } });
51
+ const parseOffice = OfficeParser_js_1.OfficeParser.parseOffice;
53
52
  exports.parseOffice = parseOffice;
53
+ const terminateOcr = OfficeParser_js_1.OfficeParser.terminateOcr;
54
+ exports.terminateOcr = terminateOcr;
54
55
  // Default export for backward compatibility
55
- exports.default = OfficeParser_1.OfficeParser;
56
- // CLI handling - allows running as: node index.js file.docx
57
- if (typeof require !== 'undefined' && typeof module !== 'undefined' && require.main === module) {
58
- const args = process.argv.slice(2);
59
- let fileArg;
60
- let toText = false;
61
- const configArgs = [];
62
- function isConfigOption(arg) {
63
- return arg.startsWith('--') && arg.includes('=');
64
- }
65
- args.forEach(arg => {
66
- if (isConfigOption(arg)) {
67
- configArgs.push(arg);
68
- }
69
- else if (!fileArg) {
70
- fileArg = arg;
71
- }
72
- });
73
- if (fileArg) {
74
- const config = {};
75
- configArgs.forEach(arg => {
76
- const [key, value] = arg.split('=');
77
- const cleanKey = key.replace('--', '');
78
- if (cleanKey === 'toText') {
79
- if (value.toLowerCase() === 'true')
80
- toText = true;
81
- else if (value.toLowerCase() === 'false')
82
- toText = false;
83
- else
84
- console.log(`Invalid value for toText: ${value}`);
85
- }
86
- // @ts-ignore
87
- else if (value.toLowerCase() === 'true')
88
- config[cleanKey] = true;
89
- // @ts-ignore
90
- else if (value.toLowerCase() === 'false')
91
- config[cleanKey] = false;
92
- // @ts-ignore
93
- else
94
- config[cleanKey] = value;
95
- });
96
- OfficeParser_1.OfficeParser.parseOffice(fileArg, config)
97
- .then((ast) => {
98
- if (toText)
99
- console.log(ast.toText());
100
- else
101
- console.log(JSON.stringify(ast, null, 2));
102
- })
103
- .catch(console.error);
104
- }
105
- else {
106
- console.log("Usage: node officeparser [file] [--option=value]");
107
- }
108
- }
56
+ exports.default = OfficeParser_js_1.OfficeParser;
package/dist/index.mjs ADDED
@@ -0,0 +1,18 @@
1
+ /**
2
+ * ESM wrapper for officeparser
3
+ *
4
+ * AUTO-GENERATED — do not edit manually.
5
+ * Generated by scripts/generate-esm-wrapper.js during build.
6
+ *
7
+ * This file re-exports from the CJS build (dist/index.js) to provide
8
+ * proper ESM named exports without duplicating the source code.
9
+ */
10
+
11
+ import _module from './index.js';
12
+
13
+ // Named exports
14
+ const { OfficeParser, parseOffice, terminateOcr } = _module;
15
+ export { OfficeParser, parseOffice, terminateOcr };
16
+
17
+ // Default export
18
+ export default _module.default ?? _module;
@@ -1,5 +1,42 @@
1
1
  // Generated by dts-bundle-generator v9.5.1
2
2
 
3
+ /**
4
+ * Configuration options for OCR.
5
+ */
6
+ export interface OcrConfig {
7
+ /**
8
+ * Language for OCR.
9
+ * Default is 'eng'.
10
+ *
11
+ * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
12
+ * The OCR engine will then attempt to recognize text in any of the specified languages.
13
+ *
14
+ * See the list of supported languages and their codes here:
15
+ * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
16
+ */
17
+ language?: string;
18
+ /**
19
+ * Path to the Tesseract worker script.
20
+ * Primarily used for offline/air-gapped environments.
21
+ */
22
+ workerPath?: string;
23
+ /**
24
+ * Path to the Tesseract core script.
25
+ * Primarily used for offline/air-gapped environments.
26
+ */
27
+ corePath?: string;
28
+ /**
29
+ * Path for Tesseract language files (traineddata).
30
+ * Primarily used for offline/air-gapped environments.
31
+ */
32
+ langPath?: string;
33
+ /**
34
+ * Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
35
+ * Set to 0 to disable auto-termination.
36
+ * Default is 10,000 (10 seconds).
37
+ */
38
+ autoTerminateTimeout?: number;
39
+ }
3
40
  /**
4
41
  * Configuration options for the OfficeParser.
5
42
  */
@@ -42,6 +79,7 @@ export interface OfficeParserConfig {
42
79
  */
43
80
  ocr?: boolean;
44
81
  /**
82
+ * @deprecated Use `ocrConfig.language` instead.
45
83
  * Language for OCR.
46
84
  * Default is 'eng'.
47
85
  *
@@ -52,14 +90,41 @@ export interface OfficeParserConfig {
52
90
  * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
53
91
  */
54
92
  ocrLanguage?: string;
93
+ /**
94
+ * Shared OCR configuration for worker pooling and offline support.
95
+ * If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
96
+ */
97
+ ocrConfig?: OcrConfig;
98
+ /**
99
+ * Flag to serialize raw content (XML) as clean, formatted strings.
100
+ * Only relevant when `includeRawContent` is true.
101
+ * Default is true.
102
+ *
103
+ * If false, the parser will attempt to extract the original raw substring from the
104
+ * source document instead of re-serializing the DOM node.
105
+ */
106
+ serializeRawContent?: boolean;
107
+ /**
108
+ * Flag to preserve original XML whitespace and line endings when serializing.
109
+ * Only relevant when `includeRawContent` is true and `serializeRawContent` is true.
110
+ * Default is false.
111
+ */
112
+ preserveXmlWhitespace?: boolean;
55
113
  /**
56
114
  * The URL/path to the PDF.js worker script.
57
115
  *
58
116
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
59
- * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.5.207/build/pdf.worker.min.mjs`.
117
+ * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
60
118
  * You can override this with your own local path or a different CDN link.
61
119
  */
62
120
  pdfWorkerSrc?: string;
121
+ /**
122
+ * Flag to include break nodes in the AST.
123
+ * This is currently only supported for Word documents. (w:br nodes)
124
+ *
125
+ * Default is false
126
+ */
127
+ includeBreakNodes?: boolean;
63
128
  }
64
129
  /**
65
130
  * Supported file types for parsing.
@@ -68,7 +133,7 @@ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods"
68
133
  /**
69
134
  * Types of content nodes in the AST.
70
135
  */
71
- export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page";
136
+ export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break";
72
137
  /**
73
138
  * Supported MIME types for attachments.
74
139
  */
@@ -171,6 +236,20 @@ export interface SheetMetadata {
171
236
  /** The style of the sheet. */
172
237
  style?: string;
173
238
  }
239
+ /**
240
+ * Detailed indentation information for paragraphs and headings.
241
+ * Values are typically in twentieths of a point (twips) in OOXML.
242
+ */
243
+ export interface IndentationMetadata {
244
+ /** Left indentation. */
245
+ left?: number;
246
+ /** Right indentation. */
247
+ right?: number;
248
+ /** First line indentation. */
249
+ firstLine?: number;
250
+ /** Hanging indentation. */
251
+ hanging?: number;
252
+ }
174
253
  /**
175
254
  * Metadata for a heading.
176
255
  */
@@ -181,6 +260,8 @@ export interface HeadingMetadata {
181
260
  alignment?: "left" | "center" | "right" | "justify";
182
261
  /** The style of the heading. */
183
262
  style?: string;
263
+ /** Detailed indentation information. */
264
+ paragraphIndentation?: IndentationMetadata;
184
265
  }
185
266
  /**
186
267
  * Metadata for a paragraph.
@@ -190,6 +271,8 @@ export interface ParagraphMetadata {
190
271
  alignment?: "left" | "center" | "right" | "justify";
191
272
  /** The style of the paragraph. */
192
273
  style?: string;
274
+ /** Detailed indentation information. */
275
+ paragraphIndentation?: IndentationMetadata;
193
276
  }
194
277
  /**
195
278
  * Metadata for a list item.
@@ -205,6 +288,8 @@ export interface ListMetadata {
205
288
  * @example 0 for top-level items, 1 for first nested level
206
289
  */
207
290
  indentation: number;
291
+ /** Detailed indentation information. */
292
+ paragraphIndentation?: IndentationMetadata;
208
293
  /**
209
294
  * Text alignment of the list item.
210
295
  * @example 'left', 'center', 'right', 'justify'
@@ -331,10 +416,35 @@ export interface NoteMetadata {
331
416
  */
332
417
  noteId?: string;
333
418
  }
419
+ /**
420
+ * Metadata for break nodes.
421
+ * Used in DOCX files to track line and page breaks.
422
+ */
423
+ export interface BreakMetadata {
424
+ /**
425
+ * Type of break. The break type determines the next location where
426
+ * text shall be placed.
427
+ * - 'column': The next text will be placed in the next column.
428
+ * - 'page': The next text will be placed on the next page.
429
+ * - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
430
+ * - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
431
+ * - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
432
+ */
433
+ breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn";
434
+ /**
435
+ * Specifies the location which shall be used as the next available line when breakType
436
+ * has a value of 'textWrapping'. Should be ignored for other break types.
437
+ * - 'all': text wrapping break shall advance the text to the next line which spans the full width of the line
438
+ * - 'left': text wrapping break shall restart in next text region unblocked on the left
439
+ * - 'none': text wrapping break shall advance the text to the next line regardless of any floating objects
440
+ * - 'right': text wrapping break shall restart in next text region unblocked on the right
441
+ */
442
+ clear?: "all" | "left" | "none" | "right";
443
+ }
334
444
  /**
335
445
  * Union type for content metadata.
336
446
  */
337
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | undefined;
447
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | undefined;
338
448
  /**
339
449
  * Represents a node in the document content tree.
340
450
  * This is the core building block of the parsed document structure.
@@ -542,6 +652,16 @@ export interface OfficeMetadata {
542
652
  formatting?: Partial<TextFormatting>;
543
653
  /** Style map for styles in the document. */
544
654
  styleMap?: Record<string, Partial<TextFormatting>>;
655
+ /**
656
+ * User-defined custom properties embedded in the document.
657
+ * Sources by format:
658
+ * - DOCX/XLSX/PPTX: `docProps/custom.xml` (Office custom document properties)
659
+ * - ODT/ODP/ODS: `meta:user-defined` elements in `meta.xml`
660
+ * - PDF: non-standard entries in the PDF Info dictionary
661
+ * RTF does not support custom properties; the `\info` group is not extracted.
662
+ * Values are typed as string, number, boolean, or Date where the source format provides type information.
663
+ */
664
+ customProperties?: Record<string, string | number | boolean | Date>;
545
665
  }
546
666
  /**
547
667
  * The Abstract Syntax Tree (AST) returned by the parser.
@@ -668,8 +788,18 @@ export declare class OfficeParser {
668
788
  * ```
669
789
  */
670
790
  static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
791
+ /**
792
+ * Terminates all active OCR workers and cleans up resources.
793
+ *
794
+ * This should be called when the application is shutting down or when OCR
795
+ * is no longer needed to prevent memory leaks and orphaned worker processes.
796
+ *
797
+ * @returns A promise that resolves when all workers have been terminated
798
+ */
799
+ static terminateOcr(): Promise<void>;
671
800
  }
672
801
  export declare const parseOffice: typeof OfficeParser.parseOffice;
802
+ export declare const terminateOcr: typeof OfficeParser.terminateOcr;
673
803
 
674
804
  export {
675
805
  OfficeParser as default,