officeparser 6.0.6 → 6.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +92 -13
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +43 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +116 -0
  6. package/dist/index.d.ts +3 -3
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +756 -0
  10. package/dist/officeparser.browser.iife.js +112 -0
  11. package/dist/officeparser.browser.mjs +111 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +71 -63
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +131 -114
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +85 -88
  20. package/dist/parsers/RtfParser.d.ts +1 -1
  21. package/dist/parsers/RtfParser.js +10 -6
  22. package/dist/parsers/WordParser.d.ts +1 -1
  23. package/dist/parsers/WordParser.js +109 -101
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +69 -1
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +39 -18
  39. package/dist/officeparser.browser.js +0 -153
  40. package/dist/officeparser.browser.js.map +0 -7
@@ -0,0 +1,756 @@
1
+ // Generated by dts-bundle-generator v9.5.1
2
+
3
+ /**
4
+ * Configuration options for OCR.
5
+ */
6
+ export interface OcrConfig {
7
+ /**
8
+ * Language for OCR.
9
+ * Default is 'eng'.
10
+ *
11
+ * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
12
+ * The OCR engine will then attempt to recognize text in any of the specified languages.
13
+ *
14
+ * See the list of supported languages and their codes here:
15
+ * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
16
+ */
17
+ language?: string;
18
+ /**
19
+ * Path to the Tesseract worker script.
20
+ * Primarily used for offline/air-gapped environments.
21
+ */
22
+ workerPath?: string;
23
+ /**
24
+ * Path to the Tesseract core script.
25
+ * Primarily used for offline/air-gapped environments.
26
+ */
27
+ corePath?: string;
28
+ /**
29
+ * Path for Tesseract language files (traineddata).
30
+ * Primarily used for offline/air-gapped environments.
31
+ */
32
+ langPath?: string;
33
+ /**
34
+ * Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
35
+ * Set to 0 to disable auto-termination.
36
+ * Default is 10,000 (10 seconds).
37
+ */
38
+ autoTerminateTimeout?: number;
39
+ }
40
+ /**
41
+ * Configuration options for the OfficeParser.
42
+ */
43
+ export interface OfficeParserConfig {
44
+ /**
45
+ * Flag to show all the logs to console in case of an error irrespective of your own handling.
46
+ * Default is false.
47
+ */
48
+ outputErrorToConsole?: boolean;
49
+ /**
50
+ * The delimiter used for every new line in places that allow multiline text like word.
51
+ * Default is \n.
52
+ */
53
+ newlineDelimiter?: string;
54
+ /**
55
+ * Flag to ignore notes from parsing in files like powerpoint.
56
+ * Default is false. It includes notes in the parsed text by default.
57
+ */
58
+ ignoreNotes?: boolean;
59
+ /**
60
+ * Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint.
61
+ * Default is false. It puts each notes right after its main slide content.
62
+ * If ignoreNotes is set to true, this flag is also ignored.
63
+ * @note This flag currently does not affect RTF files; RTF footnotes/endnotes are always collected and appended at the end of the content.
64
+ */
65
+ putNotesAtLast?: boolean;
66
+ /**
67
+ * Flag to extract attachments like images, charts, etc.
68
+ * Default is false.
69
+ */
70
+ extractAttachments?: boolean;
71
+ /**
72
+ * Flag to include raw content (XML for XML-based formats, RTF for RTF) in the AST.
73
+ * Default is false.
74
+ */
75
+ includeRawContent?: boolean;
76
+ /**
77
+ * Flag to enable OCR for images.
78
+ * Default is false.
79
+ */
80
+ ocr?: boolean;
81
+ /**
82
+ * @deprecated Use `ocrConfig.language` instead.
83
+ * Language for OCR.
84
+ * Default is 'eng'.
85
+ *
86
+ * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
87
+ * The OCR engine will then attempt to recognize text in any of the specified languages.
88
+ *
89
+ * See the list of supported languages and their codes here:
90
+ * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
91
+ */
92
+ ocrLanguage?: string;
93
+ /**
94
+ * Shared OCR configuration for worker pooling and offline support.
95
+ * If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
96
+ */
97
+ ocrConfig?: OcrConfig;
98
+ /**
99
+ * Flag to serialize raw content (XML) as clean, formatted strings.
100
+ * Only relevant when `includeRawContent` is true.
101
+ * Default is true.
102
+ *
103
+ * If false, the parser will attempt to extract the original raw substring from the
104
+ * source document instead of re-serializing the DOM node.
105
+ */
106
+ serializeRawContent?: boolean;
107
+ /**
108
+ * Flag to preserve original XML whitespace and line endings when serializing.
109
+ * Only relevant when `includeRawContent` is true and `serializeRawContent` is true.
110
+ * Default is false.
111
+ */
112
+ preserveXmlWhitespace?: boolean;
113
+ /**
114
+ * The URL/path to the PDF.js worker script.
115
+ *
116
+ * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
117
+ * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
118
+ * You can override this with your own local path or a different CDN link.
119
+ */
120
+ pdfWorkerSrc?: string;
121
+ }
122
+ /**
123
+ * Supported file types for parsing.
124
+ */
125
+ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf";
126
+ /**
127
+ * Types of content nodes in the AST.
128
+ */
129
+ export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page";
130
+ /**
131
+ * Supported MIME types for attachments.
132
+ */
133
+ export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation";
134
+ /**
135
+ * Text formatting options available for text content.
136
+ * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
137
+ * All properties are optional and only present when the formatting is explicitly applied.
138
+ */
139
+ export interface TextFormatting {
140
+ /**
141
+ * Whether the text is bold.
142
+ * Corresponds to `<w:b/>` in OOXML, `\b` in RTF.
143
+ * @example true for **bold text**, false or undefined for normal weight
144
+ */
145
+ bold?: boolean;
146
+ /**
147
+ * Whether the text is italic.
148
+ * Corresponds to `<w:i/>` in OOXML, `\i` in RTF.
149
+ * @example true for *italic text*, false or undefined for normal style
150
+ */
151
+ italic?: boolean;
152
+ /**
153
+ * Whether the text is underlined.
154
+ * Corresponds to `<w:u/>` in OOXML, `\ul` in RTF.
155
+ * @example true for underlined text, false or undefined for no underline
156
+ */
157
+ underline?: boolean;
158
+ /**
159
+ * Whether the text has a strikethrough.
160
+ * Corresponds to `<w:strike/>` in OOXML, `\strike` in RTF.
161
+ * @example true for ~~struck through~~ text
162
+ */
163
+ strikethrough?: boolean;
164
+ /**
165
+ * Text color in hex format (#RRGGBB).
166
+ * Extracted from color tables in RTF or XML color attributes in OOXML.
167
+ * @example "#ff0000" for red, "#00ff00" for green, "#0000ff" for blue
168
+ */
169
+ color?: string;
170
+ /**
171
+ * Background/highlight color in hex format (#RRGGBB).
172
+ * Represents the background color or text highlighting.
173
+ * @example "#ffff00" for yellow highlight, "#d3d3d3" for light gray
174
+ */
175
+ backgroundColor?: string;
176
+ /**
177
+ * Font size with units.
178
+ * Most parsers append 'pt' (points), but ODF may use other units like 'in' (inches) or 'cm'.
179
+ * @example "12pt" for 12pt, "14pt" for 14pt, "0.5in" for 0.5 inches
180
+ */
181
+ size?: string;
182
+ /**
183
+ * Font family/typeface name.
184
+ * Extracted from font tables in RTF or font definitions in OOXML.
185
+ * @example "Arial", "Times New Roman", "Calibri", "Ubuntu Mono"
186
+ */
187
+ font?: string;
188
+ /**
189
+ * Whether the text is subscript (e.g., H₂O).
190
+ * Corresponds to `\sub` in RTF, `<w:vertAlign w:val="subscript"/>` in OOXML.
191
+ * Mutually exclusive with superscript.
192
+ * @example true for subscript text like H₂O
193
+ */
194
+ subscript?: boolean;
195
+ /**
196
+ * Whether the text is superscript (e.g., E=mc²).
197
+ * Corresponds to `\super` in RTF, `<w:vertAlign w:val="superscript"/>` in OOXML.
198
+ * Mutually exclusive with subscript.
199
+ * @example true for superscript text like x²
200
+ */
201
+ superscript?: boolean;
202
+ /**
203
+ * The alignment of the text.
204
+ * Common in spreadsheet cells or paragraph styles.
205
+ * @example "center", "right"
206
+ */
207
+ alignment?: "left" | "center" | "right" | "justify";
208
+ }
209
+ /**
210
+ * Metadata for a slide in PowerPoint.
211
+ */
212
+ export interface SlideMetadata {
213
+ /** The slide number (1-based). */
214
+ slideNumber: number;
215
+ /**
216
+ * The unique ID of the note associated with this slide (if any).
217
+ * @example "slide-note-1"
218
+ */
219
+ noteId?: string;
220
+ /** The style of the slide. */
221
+ style?: string;
222
+ }
223
+ /**
224
+ * Metadata for a sheet in Excel.
225
+ */
226
+ export interface SheetMetadata {
227
+ /** The name of the sheet. */
228
+ sheetName: string;
229
+ /** The style of the sheet. */
230
+ style?: string;
231
+ }
232
+ /**
233
+ * Metadata for a heading.
234
+ */
235
+ export interface HeadingMetadata {
236
+ /** The heading level (e.g., 1 for H1). */
237
+ level: number;
238
+ /** The alignment of the heading. */
239
+ alignment?: "left" | "center" | "right" | "justify";
240
+ /** The style of the heading. */
241
+ style?: string;
242
+ }
243
+ /**
244
+ * Metadata for a paragraph.
245
+ */
246
+ export interface ParagraphMetadata {
247
+ /** The alignment of the paragraph. */
248
+ alignment?: "left" | "center" | "right" | "justify";
249
+ /** The style of the paragraph. */
250
+ style?: string;
251
+ }
252
+ /**
253
+ * Metadata for a list item.
254
+ */
255
+ export interface ListMetadata {
256
+ /**
257
+ * The type of list: 'ordered' (numbered) or 'unordered' (bulleted).
258
+ * @example 'ordered' for numbered lists, 'unordered' for bulleted lists
259
+ */
260
+ listType: "ordered" | "unordered";
261
+ /**
262
+ * The nesting level (indent level) of the list item, starting from 0.
263
+ * @example 0 for top-level items, 1 for first nested level
264
+ */
265
+ indentation: number;
266
+ /**
267
+ * Text alignment of the list item.
268
+ * @example 'left', 'center', 'right', 'justify'
269
+ */
270
+ alignment: "left" | "center" | "right" | "justify";
271
+ /**
272
+ * The list ID from the Word document's numbering definition.
273
+ * Used to identify which list definition this item belongs to.
274
+ * @example '1', '2' for different list definitions
275
+ */
276
+ listId: string;
277
+ /**
278
+ * The zero-based index of this item within its list.
279
+ * Continues incrementing even across paragraph interruptions for the same listId.
280
+ * @example 0, 1, 2, 3 for sequential list items
281
+ */
282
+ itemIndex: number;
283
+ /**
284
+ * The style name of the list item.
285
+ * @example "ListParagraph"
286
+ */
287
+ style?: string;
288
+ }
289
+ /**
290
+ * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
291
+ * Contains positional information about where the cell appears in the table.
292
+ */
293
+ export interface CellMetadata {
294
+ /**
295
+ * The row index of the cell (0-based).
296
+ * @example 0 for the first row, 1 for the second row, etc.
297
+ */
298
+ row: number;
299
+ /**
300
+ * The column index of the cell (0-based).
301
+ * @example 0 for column A, 1 for column B, etc.
302
+ */
303
+ col: number;
304
+ /**
305
+ * The number of rows this cell spans (merges).
306
+ * @example 2 if the cell is merged with the one below it.
307
+ */
308
+ rowSpan?: number;
309
+ /**
310
+ * The number of columns this cell spans (merges).
311
+ * @example 2 if the cell is merged with the one to its right.
312
+ */
313
+ colSpan?: number;
314
+ /** The style of the cell. */
315
+ style?: string;
316
+ }
317
+ /**
318
+ * Metadata for a chart node in the document.
319
+ * Links the chart node to its corresponding attachment in the attachments array.
320
+ */
321
+ export interface ChartMetadata {
322
+ /**
323
+ * The name of the attachment that contains the actual chart data.
324
+ * Use this to look up the full chart data from the attachments array.
325
+ * @example "chart1.xml"
326
+ */
327
+ attachmentName: string;
328
+ }
329
+ /**
330
+ * Metadata for an image node in the document.
331
+ * Links the image node to its corresponding attachment in the attachments array.
332
+ */
333
+ export interface ImageMetadata {
334
+ /**
335
+ * The name of the attachment that contains the actual image data.
336
+ * Use this to look up the full image data from the attachments array.
337
+ * @example "image1.png"
338
+ */
339
+ attachmentName: string;
340
+ /**
341
+ * Alt text (alternative text) describing the image.
342
+ * Extracted from image properties in the document.
343
+ * @example "Company logo"
344
+ */
345
+ altText?: string;
346
+ }
347
+ /**
348
+ * Metadata for PDF page nodes.
349
+ * Indicates which page of the PDF this content came from.
350
+ */
351
+ export interface PageMetadata {
352
+ /**
353
+ * The page number (1-based) from the PDF document.
354
+ * @example 1 for the first page, 2 for the second page, etc.
355
+ */
356
+ pageNumber: number;
357
+ }
358
+ /**
359
+ * Metadata for text nodes that contain hyperlinks.
360
+ * Used to track hyperlinks in text runs.
361
+ */
362
+ export interface TextMetadata {
363
+ /** Style name of the text */
364
+ style?: string;
365
+ /**
366
+ * The hyperlink URL (for external links) or anchor reference (for internal links).
367
+ * @example "https://example.com" or "#_Toc123456"
368
+ */
369
+ link?: string;
370
+ /**
371
+ * Type of hyperlink.
372
+ * - 'internal': Link to a bookmark/anchor within the same document
373
+ * - 'external': Link to an external URL
374
+ */
375
+ linkType?: "internal" | "external";
376
+ }
377
+ /**
378
+ * Metadata for note nodes (footnotes/endnotes).
379
+ * Used in ODT and DOCX files to track notes.
380
+ */
381
+ export interface NoteMetadata {
382
+ /**
383
+ * Type of note: 'footnote' or 'endnote'.
384
+ */
385
+ noteType?: "footnote" | "endnote";
386
+ /**
387
+ * The unique ID of the note from the source document.
388
+ * @example "1", "2"
389
+ */
390
+ noteId?: string;
391
+ }
392
+ /**
393
+ * Union type for content metadata.
394
+ */
395
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | undefined;
396
+ /**
397
+ * Represents a node in the document content tree.
398
+ * This is the core building block of the parsed document structure.
399
+ * Content nodes can be nested to represent hierarchical document structures
400
+ * (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
401
+ *
402
+ * @example
403
+ * // A simple paragraph with formatted text
404
+ * {
405
+ * type: 'paragraph',
406
+ * text: 'Hello world',
407
+ * children: [
408
+ * { type: 'text', text: 'Hello ', formatting: { bold: true } },
409
+ * { type: 'text', text: 'world', formatting: { italic: true } }
410
+ * ]
411
+ * }
412
+ *
413
+ * @example
414
+ * // A heading with metadata
415
+ * {
416
+ * type: 'heading',
417
+ * text: 'Chapter 1',
418
+ * metadata: { level: 1 },
419
+ * children: [...]
420
+ * }
421
+ */
422
+ export interface OfficeContentNode {
423
+ /**
424
+ * The type of the node.
425
+ * Determines how the node should be interpreted and rendered.
426
+ * Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
427
+ */
428
+ type: OfficeContentNodeType;
429
+ /**
430
+ * The complete text content of the node and all its children combined.
431
+ * For container nodes (paragraph, heading), this is the concatenation of all child text.
432
+ * For leaf nodes (text), this is the actual text content.
433
+ * @example "Hello world" for a paragraph containing "Hello " and "world"
434
+ */
435
+ text?: string;
436
+ /**
437
+ * Child nodes that make up this node's content.
438
+ * Used for hierarchical structures:
439
+ * - Paragraphs contain text runs with different formatting
440
+ * - Tables contain rows
441
+ * - Rows contain cells
442
+ * - Cells contain paragraphs
443
+ * @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
444
+ */
445
+ children?: OfficeContentNode[];
446
+ /**
447
+ * Text formatting applied to this node.
448
+ * Only applicable to text-containing nodes.
449
+ * For container nodes like paragraphs, formatting typically appears on child text nodes.
450
+ * @example { bold: true, size: "12", font: "Arial" }
451
+ */
452
+ formatting?: TextFormatting;
453
+ /**
454
+ * Type-specific metadata providing additional context about the node.
455
+ * The metadata structure depends on the node type:
456
+ * - Headings: { level: 1 }
457
+ * - Lists: { listType: 'ordered', indentation: 0 }
458
+ * - Cells: { row: 0, col: 0 }
459
+ * - Slides: { slideNumber: 1 }
460
+ * @example { level: 1 } for a heading
461
+ */
462
+ metadata?: ContentMetadata;
463
+ /**
464
+ * The raw source content for this node.
465
+ * - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
466
+ * - For RTF: contains the raw RTF markup
467
+ * - For PDF: typically not available
468
+ * Only populated when `config.includeRawContent` is true.
469
+ * Useful for debugging or when you need access to format-specific features.
470
+ * @example "<w:p><w:r><w:t>Hello</w:t></w:r></w:p>" for DOCX
471
+ */
472
+ rawContent?: string;
473
+ }
474
+ /**
475
+ * Structured information extracted from a chart.
476
+ */
477
+ export interface ChartData {
478
+ /** Chart title (if any) */
479
+ title?: string;
480
+ /** X-axis title (for continuous or categorical axes) */
481
+ xAxisTitle?: string;
482
+ /** Y-axis title (for value or continuous axes) */
483
+ yAxisTitle?: string;
484
+ /**
485
+ * Collections of data points.
486
+ * For bar/line charts, each dataset is one 'line' or group of bars.
487
+ * For pie charts, there is typically only one dataset.
488
+ */
489
+ dataSets: {
490
+ /** Name of this data group (e.g., 'Sales 2023') */
491
+ name?: string;
492
+ /** Actual numeric or string values for this group */
493
+ values: string[];
494
+ /** Specific labels for each point in this dataset (if defined per point) */
495
+ pointLabels: string[];
496
+ }[];
497
+ /**
498
+ * Labels for the chart facets (e.g., 'Jan', 'Feb', 'Mar' on X-axis).
499
+ * These typically correspond to the data points in each dataSet.
500
+ */
501
+ labels: string[];
502
+ /** Every text node discovered in the chart XML (for keyword search/raw extraction) */
503
+ rawTexts: string[];
504
+ }
505
+ /**
506
+ * Represents an attachment extracted from the document (image, chart, etc.).
507
+ * Attachments are binary resources embedded in the document.
508
+ * Only populated when `config.extractAttachments` is true.
509
+ *
510
+ * @example
511
+ * ```typescript
512
+ * {
513
+ * type: 'image',
514
+ * mimeType: 'image/png',
515
+ * data: 'iVBORw0KGgoAAAANSUhEUgAA...', // Base64
516
+ * name: 'chart1.png',
517
+ * extension: 'png',
518
+ * ocrText: 'Sales Chart Q4 2024' // If OCR was enabled
519
+ * }
520
+ * ```
521
+ */
522
+ export interface OfficeAttachment {
523
+ /**
524
+ * The category of the attachment.
525
+ * Helps identify what kind of content this represents.
526
+ * @example 'image' for photos and diagrams, 'chart' for embedded charts
527
+ */
528
+ type: "image" | "chart";
529
+ /**
530
+ * The MIME type of the attachment data.
531
+ * Indicates the file format and how the data should be interpreted.
532
+ * @example 'image/png', 'image/jpeg', 'image/svg+xml'
533
+ */
534
+ mimeType: OfficeMimeType;
535
+ /**
536
+ * The attachment content encoded as Base64.
537
+ * This is the actual binary data of the image/chart/etc. encoded for text transmission.
538
+ * Can be used directly in HTML img tags with data URIs or decoded to binary.
539
+ * @example "iVBORw0KGgoAAAANSUhEUgAA..." (truncated)
540
+ */
541
+ data: string;
542
+ /**
543
+ * A unique name for this attachment file.
544
+ * May be derived from the source file or auto-generated.
545
+ * Used to link `ImageMetadata` nodes to their corresponding attachments.
546
+ * @example "image1.png", "chart2.emf", "picture3.jpg"
547
+ */
548
+ name: string;
549
+ /**
550
+ * The file extension (without the dot).
551
+ * Derived from the MIME type or original filename.
552
+ * @example "png", "jpg", "svg"
553
+ */
554
+ extension: string;
555
+ /**
556
+ * Text extracted from the image using Optical Character Recognition (OCR).
557
+ * Only present when:
558
+ * - `config.ocr` is true
559
+ * - `config.extractAttachments` is true
560
+ * - The attachment is an image containing text
561
+ * Uses Tesseract.js with the language specified in `config.ocrLanguage`.
562
+ * @example "Annual Revenue: $1.2M"
563
+ */
564
+ ocrText?: string;
565
+ /**
566
+ * Alt text or description associated with the image in the document.
567
+ * Extracted from the document markup (e.g., wp:docPr descr attribute in DOCX).
568
+ * @example "A chart showing sales growth"
569
+ */
570
+ altText?: string;
571
+ /**
572
+ * Structured data extracted from a chart attachment.
573
+ * Only present if the attachment is a chart and data extraction was successful.
574
+ * Contains series names, values, labels, and titles.
575
+ * @example { title: "Sales Chart", series: [...], categories: [...] }
576
+ */
577
+ chartData?: ChartData;
578
+ }
579
+ /**
580
+ * Metadata for the parsed file.
581
+ */
582
+ export interface OfficeMetadata {
583
+ /** The title of the document. */
584
+ title?: string;
585
+ /** The author of the document. */
586
+ author?: string;
587
+ /** User who last modified the document. */
588
+ lastModifiedBy?: string;
589
+ /** Creation date. */
590
+ created?: Date;
591
+ /** Last modification date. */
592
+ modified?: Date;
593
+ /** Description/Comments. */
594
+ description?: string;
595
+ /** Subject/Topic. */
596
+ subject?: string;
597
+ /** Number of pages (if available). */
598
+ pages?: number;
599
+ /** Document-wide default formatting settings (font, size, color). */
600
+ formatting?: Partial<TextFormatting>;
601
+ /** Style map for styles in the document. */
602
+ styleMap?: Record<string, Partial<TextFormatting>>;
603
+ /**
604
+ * User-defined custom properties embedded in the document.
605
+ * Sources by format:
606
+ * - DOCX/XLSX/PPTX: `docProps/custom.xml` (Office custom document properties)
607
+ * - ODT/ODP/ODS: `meta:user-defined` elements in `meta.xml`
608
+ * - PDF: non-standard entries in the PDF Info dictionary
609
+ * RTF does not support custom properties; the `\info` group is not extracted.
610
+ * Values are typed as string, number, boolean, or Date where the source format provides type information.
611
+ */
612
+ customProperties?: Record<string, string | number | boolean | Date>;
613
+ }
614
+ /**
615
+ * The Abstract Syntax Tree (AST) returned by the parser.
616
+ * This is the root data structure representing the entire parsed document.
617
+ *
618
+ * The AST provides a format-agnostic representation of the document that can be easily
619
+ * processed, transformed, or converted to other formats. It preserves the document's
620
+ * structure, content, formatting, and metadata while abstracting away format-specific details.
621
+ *
622
+ * @example
623
+ * ```typescript
624
+ * const ast = await OfficeParser.parseOffice('document.docx', {
625
+ * extractAttachments: true,
626
+ * includeRawContent: false
627
+ * });
628
+ *
629
+ * console.log(ast.type); // 'docx'
630
+ * console.log(ast.metadata.author); // 'John Doe'
631
+ * console.log(ast.content.length); // Number of top-level content nodes
632
+ * console.log(ast.toText()); // Plain text representation
633
+ * ```
634
+ */
635
+ export interface OfficeParserAST {
636
+ /**
637
+ * The type of the parsed file.
638
+ * Indicates which parser was used and what format the input was in.
639
+ * @example 'docx', 'xlsx', 'pptx', 'rtf', 'pdf', 'odt', 'odp', 'ods'
640
+ */
641
+ type: SupportedFileType;
642
+ /**
643
+ * Document metadata extracted from the file properties.
644
+ * Includes information like author, title, creation date, etc.
645
+ * Availability depends on the file format and whether metadata was present in the source.
646
+ * @example { author: 'John Smith', title: 'Annual Report', created: new Date('2024-01-01') }
647
+ */
648
+ metadata: OfficeMetadata;
649
+ /**
650
+ * The hierarchical content structure of the document.
651
+ * This is an array of top-level content nodes. Each node can have children, creating a tree.
652
+ * For different file types:
653
+ * - DOCX: Array of paragraphs, headings, tables, etc.
654
+ * - XLSX: Array of sheets, each containing rows
655
+ * - PPTX: Array of slides, each containing content nodes
656
+ * - PDF: Array of pages, each containing paragraphs
657
+ * @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
658
+ */
659
+ content: OfficeContentNode[];
660
+ /**
661
+ * Attachments extracted from the document (images, charts, embedded files).
662
+ * Only populated when `config.extractAttachments` is true.
663
+ * Each attachment includes:
664
+ * - Base64-encoded data
665
+ * - MIME type
666
+ * - Optional OCR text (if `config.ocr` is true)
667
+ * @example [{ type: 'image', mimeType: 'image/png', data: 'base64...', name: 'image1.png' }]
668
+ */
669
+ attachments: OfficeAttachment[];
670
+ /**
671
+ * Converts the entire AST to plain text.
672
+ * This method flattens the document structure and returns just the text content,
673
+ * stripping out all formatting, metadata, and structure.
674
+ *
675
+ * The text is concatenated using the delimiter specified in `config.newlineDelimiter` (default: '\n').
676
+ *
677
+ * @returns A plain text representation of the document
678
+ * @example
679
+ * ```typescript
680
+ * const text = ast.toText();
681
+ * console.log(text); // "Hello world\nChapter 1\n..."
682
+ * ```
683
+ */
684
+ toText(): string;
685
+ }
686
+ /**
687
+ * Main parser class providing office document parsing functionality.
688
+ *
689
+ * This class contains a single static method `parseOffice` that serves as the
690
+ * universal entry point for parsing any supported office document format.
691
+ */
692
+ export declare class OfficeParser {
693
+ /**
694
+ * Parses an office document and returns a structured AST.
695
+ *
696
+ * This method:
697
+ * 1. Accepts a file path, Buffer, or ArrayBuffer
698
+ * 2. Detects the file type (from extension or content)
699
+ * 3. Routes to the appropriate format-specific parser
700
+ * 4. Returns a unified AST structure
701
+ *
702
+ * **File Type Detection:**
703
+ * - If a file path is provided, uses the file extension
704
+ * - If a Buffer is provided, uses magic bytes detection (file-type library)
705
+ *
706
+ * **Supported Formats and Routes:**
707
+ * - `.docx` → WordParser (OOXML)
708
+ * - `.xlsx` → ExcelParser (OOXML)
709
+ * - `.pptx` → PowerPointParser (OOXML)
710
+ * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
711
+ * - `.pdf` → PdfParser (PDF.js)
712
+ * - `.rtf` → RtfParser (custom RTF parser)
713
+ *
714
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the document
715
+ * @param config - Optional configuration object (defaults applied for all omitted options)
716
+ * @returns A promise resolving to the parsed OfficeParserAST
717
+ * @throws {Error} If file doesn't exist, format is unsupported, or parsing fails
718
+ *
719
+ * @example
720
+ * ```typescript
721
+ * // Parse a DOCX file
722
+ * const ast = await OfficeParser.parseOffice('report.docx', {
723
+ * extractAttachments: true,
724
+ * includeRawContent: false
725
+ * });
726
+ *
727
+ * // Parse a Buffer with OCR enabled
728
+ * const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
729
+ * const ast = await OfficeParser.parseOffice(buffer, {
730
+ * ocr: true,
731
+ * ocrLanguage: 'eng+fra'
732
+ * });
733
+ *
734
+ * // Extract text
735
+ * const text = ast.toText();
736
+ * ```
737
+ */
738
+ static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
739
+ /**
740
+ * Terminates all active OCR workers and cleans up resources.
741
+ *
742
+ * This should be called when the application is shutting down or when OCR
743
+ * is no longer needed to prevent memory leaks and orphaned worker processes.
744
+ *
745
+ * @returns A promise that resolves when all workers have been terminated
746
+ */
747
+ static terminateOcr(): Promise<void>;
748
+ }
749
+ export declare const parseOffice: typeof OfficeParser.parseOffice;
750
+ export declare const terminateOcr: typeof OfficeParser.terminateOcr;
751
+
752
+ export {
753
+ OfficeParser as default,
754
+ };
755
+
756
+ export {};