officeparser 5.2.2 → 6.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,615 @@
1
+ /**
2
+ * Configuration options for the OfficeParser.
3
+ */
4
+ export interface OfficeParserConfig {
5
+ /**
6
+ * Flag to show all the logs to console in case of an error irrespective of your own handling.
7
+ * Default is false.
8
+ */
9
+ outputErrorToConsole?: boolean;
10
+ /**
11
+ * The delimiter used for every new line in places that allow multiline text like word.
12
+ * Default is \n.
13
+ */
14
+ newlineDelimiter?: string;
15
+ /**
16
+ * Flag to ignore notes from parsing in files like powerpoint.
17
+ * Default is false. It includes notes in the parsed text by default.
18
+ */
19
+ ignoreNotes?: boolean;
20
+ /**
21
+ * Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint.
22
+ * Default is false. It puts each notes right after its main slide content.
23
+ * If ignoreNotes is set to true, this flag is also ignored.
24
+ * @note This flag currently does not affect RTF files; RTF footnotes/endnotes are always collected and appended at the end of the content.
25
+ */
26
+ putNotesAtLast?: boolean;
27
+ /**
28
+ * Flag to extract attachments like images, charts, etc.
29
+ * Default is false.
30
+ */
31
+ extractAttachments?: boolean;
32
+ /**
33
+ * Flag to include raw content (XML for XML-based formats, RTF for RTF) in the AST.
34
+ * Default is false.
35
+ */
36
+ includeRawContent?: boolean;
37
+ /**
38
+ * Flag to enable OCR for images.
39
+ * Default is false.
40
+ */
41
+ ocr?: boolean;
42
+ /**
43
+ * Language for OCR.
44
+ * Default is 'eng'.
45
+ *
46
+ * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
47
+ * The OCR engine will then attempt to recognize text in any of the specified languages.
48
+ *
49
+ * See the list of supported languages and their codes here:
50
+ * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
51
+ */
52
+ ocrLanguage?: string;
53
+ /**
54
+ * The URL/path to the PDF.js worker script.
55
+ *
56
+ * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
57
+ * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.4.530/build/pdf.worker.min.mjs`.
58
+ * You can override this with your own local path or a different CDN link.
59
+ */
60
+ pdfWorkerSrc?: string;
61
+ }
62
+ /**
63
+ * Supported file types for parsing.
64
+ */
65
+ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf';
66
+ /**
67
+ * Types of content nodes in the AST.
68
+ */
69
+ export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page';
70
+ /**
71
+ * Supported MIME types for attachments.
72
+ */
73
+ export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation';
74
+ /**
75
+ * Text formatting options available for text content.
76
+ * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
77
+ * All properties are optional and only present when the formatting is explicitly applied.
78
+ */
79
+ export interface TextFormatting {
80
+ /**
81
+ * Whether the text is bold.
82
+ * Corresponds to `<w:b/>` in OOXML, `\b` in RTF.
83
+ * @example true for **bold text**, false or undefined for normal weight
84
+ */
85
+ bold?: boolean;
86
+ /**
87
+ * Whether the text is italic.
88
+ * Corresponds to `<w:i/>` in OOXML, `\i` in RTF.
89
+ * @example true for *italic text*, false or undefined for normal style
90
+ */
91
+ italic?: boolean;
92
+ /**
93
+ * Whether the text is underlined.
94
+ * Corresponds to `<w:u/>` in OOXML, `\ul` in RTF.
95
+ * @example true for underlined text, false or undefined for no underline
96
+ */
97
+ underline?: boolean;
98
+ /**
99
+ * Whether the text has a strikethrough.
100
+ * Corresponds to `<w:strike/>` in OOXML, `\strike` in RTF.
101
+ * @example true for ~~struck through~~ text
102
+ */
103
+ strikethrough?: boolean;
104
+ /**
105
+ * Text color in hex format (#RRGGBB).
106
+ * Extracted from color tables in RTF or XML color attributes in OOXML.
107
+ * @example "#ff0000" for red, "#00ff00" for green, "#0000ff" for blue
108
+ */
109
+ color?: string;
110
+ /**
111
+ * Background/highlight color in hex format (#RRGGBB).
112
+ * Represents the background color or text highlighting.
113
+ * @example "#ffff00" for yellow highlight, "#d3d3d3" for light gray
114
+ */
115
+ backgroundColor?: string;
116
+ /**
117
+ * Font size with units.
118
+ * Most parsers append 'pt' (points), but ODF may use other units like 'in' (inches) or 'cm'.
119
+ * @example "12pt" for 12pt, "14pt" for 14pt, "0.5in" for 0.5 inches
120
+ */
121
+ size?: string;
122
+ /**
123
+ * Font family/typeface name.
124
+ * Extracted from font tables in RTF or font definitions in OOXML.
125
+ * @example "Arial", "Times New Roman", "Calibri", "Ubuntu Mono"
126
+ */
127
+ font?: string;
128
+ /**
129
+ * Whether the text is subscript (e.g., H₂O).
130
+ * Corresponds to `\sub` in RTF, `<w:vertAlign w:val="subscript"/>` in OOXML.
131
+ * Mutually exclusive with superscript.
132
+ * @example true for subscript text like H₂O
133
+ */
134
+ subscript?: boolean;
135
+ /**
136
+ * Whether the text is superscript (e.g., E=mc²).
137
+ * Corresponds to `\super` in RTF, `<w:vertAlign w:val="superscript"/>` in OOXML.
138
+ * Mutually exclusive with subscript.
139
+ * @example true for superscript text like x²
140
+ */
141
+ superscript?: boolean;
142
+ /**
143
+ * The alignment of the text.
144
+ * Common in spreadsheet cells or paragraph styles.
145
+ * @example "center", "right"
146
+ */
147
+ alignment?: 'left' | 'center' | 'right' | 'justify';
148
+ }
149
+ /**
150
+ * Metadata for a slide in PowerPoint.
151
+ */
152
+ export interface SlideMetadata {
153
+ /** The slide number (1-based). */
154
+ slideNumber: number;
155
+ /**
156
+ * The unique ID of the note associated with this slide (if any).
157
+ * @example "slide-note-1"
158
+ */
159
+ noteId?: string;
160
+ /** The style of the slide. */
161
+ style?: string;
162
+ }
163
+ /**
164
+ * Metadata for a sheet in Excel.
165
+ */
166
+ export interface SheetMetadata {
167
+ /** The name of the sheet. */
168
+ sheetName: string;
169
+ /** The style of the sheet. */
170
+ style?: string;
171
+ }
172
+ /**
173
+ * Metadata for a heading.
174
+ */
175
+ export interface HeadingMetadata {
176
+ /** The heading level (e.g., 1 for H1). */
177
+ level: number;
178
+ /** The alignment of the heading. */
179
+ alignment?: 'left' | 'center' | 'right' | 'justify';
180
+ /** The style of the heading. */
181
+ style?: string;
182
+ }
183
+ /**
184
+ * Metadata for a paragraph.
185
+ */
186
+ export interface ParagraphMetadata {
187
+ /** The alignment of the paragraph. */
188
+ alignment?: 'left' | 'center' | 'right' | 'justify';
189
+ /** The style of the paragraph. */
190
+ style?: string;
191
+ }
192
+ /**
193
+ * Metadata for a list item.
194
+ */
195
+ export interface ListMetadata {
196
+ /**
197
+ * The type of list: 'ordered' (numbered) or 'unordered' (bulleted).
198
+ * @example 'ordered' for numbered lists, 'unordered' for bulleted lists
199
+ */
200
+ listType: 'ordered' | 'unordered';
201
+ /**
202
+ * The nesting level (indent level) of the list item, starting from 0.
203
+ * @example 0 for top-level items, 1 for first nested level
204
+ */
205
+ indentation: number;
206
+ /**
207
+ * Text alignment of the list item.
208
+ * @example 'left', 'center', 'right', 'justify'
209
+ */
210
+ alignment: 'left' | 'center' | 'right' | 'justify';
211
+ /**
212
+ * The list ID from the Word document's numbering definition.
213
+ * Used to identify which list definition this item belongs to.
214
+ * @example '1', '2' for different list definitions
215
+ */
216
+ listId: string;
217
+ /**
218
+ * The zero-based index of this item within its list.
219
+ * Continues incrementing even across paragraph interruptions for the same listId.
220
+ * @example 0, 1, 2, 3 for sequential list items
221
+ */
222
+ itemIndex: number;
223
+ /**
224
+ * The style name of the list item.
225
+ * @example "ListParagraph"
226
+ */
227
+ style?: string;
228
+ }
229
+ /**
230
+ * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
231
+ * Contains positional information about where the cell appears in the table.
232
+ */
233
+ export interface CellMetadata {
234
+ /**
235
+ * The row index of the cell (0-based).
236
+ * @example 0 for the first row, 1 for the second row, etc.
237
+ */
238
+ row: number;
239
+ /**
240
+ * The column index of the cell (0-based).
241
+ * @example 0 for column A, 1 for column B, etc.
242
+ */
243
+ col: number;
244
+ /**
245
+ * The number of rows this cell spans (merges).
246
+ * @example 2 if the cell is merged with the one below it.
247
+ */
248
+ rowSpan?: number;
249
+ /**
250
+ * The number of columns this cell spans (merges).
251
+ * @example 2 if the cell is merged with the one to its right.
252
+ */
253
+ colSpan?: number;
254
+ /** The style of the cell. */
255
+ style?: string;
256
+ }
257
+ /**
258
+ * Metadata for a chart node in the document.
259
+ * Links the chart node to its corresponding attachment in the attachments array.
260
+ */
261
+ export interface ChartMetadata {
262
+ /**
263
+ * The name of the attachment that contains the actual chart data.
264
+ * Use this to look up the full chart data from the attachments array.
265
+ * @example "chart1.xml"
266
+ */
267
+ attachmentName: string;
268
+ }
269
+ /**
270
+ * Metadata for an image node in the document.
271
+ * Links the image node to its corresponding attachment in the attachments array.
272
+ */
273
+ export interface ImageMetadata {
274
+ /**
275
+ * The name of the attachment that contains the actual image data.
276
+ * Use this to look up the full image data from the attachments array.
277
+ * @example "image1.png"
278
+ */
279
+ attachmentName: string;
280
+ /**
281
+ * Alt text (alternative text) describing the image.
282
+ * Extracted from image properties in the document.
283
+ * @example "Company logo"
284
+ */
285
+ altText?: string;
286
+ }
287
+ /**
288
+ * Metadata for PDF page nodes.
289
+ * Indicates which page of the PDF this content came from.
290
+ */
291
+ export interface PageMetadata {
292
+ /**
293
+ * The page number (1-based) from the PDF document.
294
+ * @example 1 for the first page, 2 for the second page, etc.
295
+ */
296
+ pageNumber: number;
297
+ }
298
+ /**
299
+ * Metadata for text nodes that contain hyperlinks.
300
+ * Used to track hyperlinks in text runs.
301
+ */
302
+ export interface TextMetadata {
303
+ /** Style name of the text */
304
+ style?: string;
305
+ /**
306
+ * The hyperlink URL (for external links) or anchor reference (for internal links).
307
+ * @example "https://example.com" or "#_Toc123456"
308
+ */
309
+ link?: string;
310
+ /**
311
+ * Type of hyperlink.
312
+ * - 'internal': Link to a bookmark/anchor within the same document
313
+ * - 'external': Link to an external URL
314
+ */
315
+ linkType?: 'internal' | 'external';
316
+ }
317
+ /**
318
+ * Metadata for note nodes (footnotes/endnotes).
319
+ * Used in ODT and DOCX files to track notes.
320
+ */
321
+ export interface NoteMetadata {
322
+ /**
323
+ * Type of note: 'footnote' or 'endnote'.
324
+ */
325
+ noteType?: 'footnote' | 'endnote';
326
+ /**
327
+ * The unique ID of the note from the source document.
328
+ * @example "1", "2"
329
+ */
330
+ noteId?: string;
331
+ }
332
+ /**
333
+ * Union type for content metadata.
334
+ */
335
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | undefined;
336
+ /**
337
+ * Represents a node in the document content tree.
338
+ * This is the core building block of the parsed document structure.
339
+ * Content nodes can be nested to represent hierarchical document structures
340
+ * (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
341
+ *
342
+ * @example
343
+ * // A simple paragraph with formatted text
344
+ * {
345
+ * type: 'paragraph',
346
+ * text: 'Hello world',
347
+ * children: [
348
+ * { type: 'text', text: 'Hello ', formatting: { bold: true } },
349
+ * { type: 'text', text: 'world', formatting: { italic: true } }
350
+ * ]
351
+ * }
352
+ *
353
+ * @example
354
+ * // A heading with metadata
355
+ * {
356
+ * type: 'heading',
357
+ * text: 'Chapter 1',
358
+ * metadata: { level: 1 },
359
+ * children: [...]
360
+ * }
361
+ */
362
+ export interface OfficeContentNode {
363
+ /**
364
+ * The type of the node.
365
+ * Determines how the node should be interpreted and rendered.
366
+ * Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
367
+ */
368
+ type: OfficeContentNodeType;
369
+ /**
370
+ * The complete text content of the node and all its children combined.
371
+ * For container nodes (paragraph, heading), this is the concatenation of all child text.
372
+ * For leaf nodes (text), this is the actual text content.
373
+ * @example "Hello world" for a paragraph containing "Hello " and "world"
374
+ */
375
+ text?: string;
376
+ /**
377
+ * Child nodes that make up this node's content.
378
+ * Used for hierarchical structures:
379
+ * - Paragraphs contain text runs with different formatting
380
+ * - Tables contain rows
381
+ * - Rows contain cells
382
+ * - Cells contain paragraphs
383
+ * @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
384
+ */
385
+ children?: OfficeContentNode[];
386
+ /**
387
+ * Text formatting applied to this node.
388
+ * Only applicable to text-containing nodes.
389
+ * For container nodes like paragraphs, formatting typically appears on child text nodes.
390
+ * @example { bold: true, size: "12", font: "Arial" }
391
+ */
392
+ formatting?: TextFormatting;
393
+ /**
394
+ * Type-specific metadata providing additional context about the node.
395
+ * The metadata structure depends on the node type:
396
+ * - Headings: { level: 1 }
397
+ * - Lists: { listType: 'ordered', indentation: 0 }
398
+ * - Cells: { row: 0, col: 0 }
399
+ * - Slides: { slideNumber: 1 }
400
+ * @example { level: 1 } for a heading
401
+ */
402
+ metadata?: ContentMetadata;
403
+ /**
404
+ * The raw source content for this node.
405
+ * - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
406
+ * - For RTF: contains the raw RTF markup
407
+ * - For PDF: typically not available
408
+ * Only populated when `config.includeRawContent` is true.
409
+ * Useful for debugging or when you need access to format-specific features.
410
+ * @example "<w:p><w:r><w:t>Hello</w:t></w:r></w:p>" for DOCX
411
+ */
412
+ rawContent?: string;
413
+ }
414
+ /**
415
+ * Structured information extracted from a chart.
416
+ */
417
+ export interface ChartData {
418
+ /** Chart title (if any) */
419
+ title?: string;
420
+ /** X-axis title (for continuous or categorical axes) */
421
+ xAxisTitle?: string;
422
+ /** Y-axis title (for value or continuous axes) */
423
+ yAxisTitle?: string;
424
+ /**
425
+ * Collections of data points.
426
+ * For bar/line charts, each dataset is one 'line' or group of bars.
427
+ * For pie charts, there is typically only one dataset.
428
+ */
429
+ dataSets: {
430
+ /** Name of this data group (e.g., 'Sales 2023') */
431
+ name?: string;
432
+ /** Actual numeric or string values for this group */
433
+ values: string[];
434
+ /** Specific labels for each point in this dataset (if defined per point) */
435
+ pointLabels: string[];
436
+ }[];
437
+ /**
438
+ * Labels for the chart facets (e.g., 'Jan', 'Feb', 'Mar' on X-axis).
439
+ * These typically correspond to the data points in each dataSet.
440
+ */
441
+ labels: string[];
442
+ /** Every text node discovered in the chart XML (for keyword search/raw extraction) */
443
+ rawTexts: string[];
444
+ }
445
+ /**
446
+ * Represents an attachment extracted from the document (image, chart, etc.).
447
+ * Attachments are binary resources embedded in the document.
448
+ * Only populated when `config.extractAttachments` is true.
449
+ *
450
+ * @example
451
+ * ```typescript
452
+ * {
453
+ * type: 'image',
454
+ * mimeType: 'image/png',
455
+ * data: 'iVBORw0KGgoAAAANSUhEUgAA...', // Base64
456
+ * name: 'chart1.png',
457
+ * extension: 'png',
458
+ * ocrText: 'Sales Chart Q4 2024' // If OCR was enabled
459
+ * }
460
+ * ```
461
+ */
462
+ export interface OfficeAttachment {
463
+ /**
464
+ * The category of the attachment.
465
+ * Helps identify what kind of content this represents.
466
+ * @example 'image' for photos and diagrams, 'chart' for embedded charts
467
+ */
468
+ type: 'image' | 'chart';
469
+ /**
470
+ * The MIME type of the attachment data.
471
+ * Indicates the file format and how the data should be interpreted.
472
+ * @example 'image/png', 'image/jpeg', 'image/svg+xml'
473
+ */
474
+ mimeType: OfficeMimeType;
475
+ /**
476
+ * The attachment content encoded as Base64.
477
+ * This is the actual binary data of the image/chart/etc. encoded for text transmission.
478
+ * Can be used directly in HTML img tags with data URIs or decoded to binary.
479
+ * @example "iVBORw0KGgoAAAANSUhEUgAA..." (truncated)
480
+ */
481
+ data: string;
482
+ /**
483
+ * A unique name for this attachment file.
484
+ * May be derived from the source file or auto-generated.
485
+ * Used to link `ImageMetadata` nodes to their corresponding attachments.
486
+ * @example "image1.png", "chart2.emf", "picture3.jpg"
487
+ */
488
+ name: string;
489
+ /**
490
+ * The file extension (without the dot).
491
+ * Derived from the MIME type or original filename.
492
+ * @example "png", "jpg", "svg"
493
+ */
494
+ extension: string;
495
+ /**
496
+ * Text extracted from the image using Optical Character Recognition (OCR).
497
+ * Only present when:
498
+ * - `config.ocr` is true
499
+ * - `config.extractAttachments` is true
500
+ * - The attachment is an image containing text
501
+ * Uses Tesseract.js with the language specified in `config.ocrLanguage`.
502
+ * @example "Annual Revenue: $1.2M"
503
+ */
504
+ ocrText?: string;
505
+ /**
506
+ * Alt text or description associated with the image in the document.
507
+ * Extracted from the document markup (e.g., wp:docPr descr attribute in DOCX).
508
+ * @example "A chart showing sales growth"
509
+ */
510
+ altText?: string;
511
+ /**
512
+ * Structured data extracted from a chart attachment.
513
+ * Only present if the attachment is a chart and data extraction was successful.
514
+ * Contains series names, values, labels, and titles.
515
+ * @example { title: "Sales Chart", series: [...], categories: [...] }
516
+ */
517
+ chartData?: ChartData;
518
+ }
519
+ /**
520
+ * Metadata for the parsed file.
521
+ */
522
+ export interface OfficeMetadata {
523
+ /** The title of the document. */
524
+ title?: string;
525
+ /** The author of the document. */
526
+ author?: string;
527
+ /** User who last modified the document. */
528
+ lastModifiedBy?: string;
529
+ /** Creation date. */
530
+ created?: Date;
531
+ /** Last modification date. */
532
+ modified?: Date;
533
+ /** Description/Comments. */
534
+ description?: string;
535
+ /** Subject/Topic. */
536
+ subject?: string;
537
+ /** Number of pages (if available). */
538
+ pages?: number;
539
+ /** Document-wide default formatting settings (font, size, color). */
540
+ formatting?: Partial<TextFormatting>;
541
+ /** Style map for styles in the document. */
542
+ styleMap?: Record<string, Partial<TextFormatting>>;
543
+ }
544
+ /**
545
+ * The Abstract Syntax Tree (AST) returned by the parser.
546
+ * This is the root data structure representing the entire parsed document.
547
+ *
548
+ * The AST provides a format-agnostic representation of the document that can be easily
549
+ * processed, transformed, or converted to other formats. It preserves the document's
550
+ * structure, content, formatting, and metadata while abstracting away format-specific details.
551
+ *
552
+ * @example
553
+ * ```typescript
554
+ * const ast = await OfficeParser.parseOffice('document.docx', {
555
+ * extractAttachments: true,
556
+ * includeRawContent: false
557
+ * });
558
+ *
559
+ * console.log(ast.type); // 'docx'
560
+ * console.log(ast.metadata.author); // 'John Doe'
561
+ * console.log(ast.content.length); // Number of top-level content nodes
562
+ * console.log(ast.toText()); // Plain text representation
563
+ * ```
564
+ */
565
+ export interface OfficeParserAST {
566
+ /**
567
+ * The type of the parsed file.
568
+ * Indicates which parser was used and what format the input was in.
569
+ * @example 'docx', 'xlsx', 'pptx', 'rtf', 'pdf', 'odt', 'odp', 'ods'
570
+ */
571
+ type: SupportedFileType;
572
+ /**
573
+ * Document metadata extracted from the file properties.
574
+ * Includes information like author, title, creation date, etc.
575
+ * Availability depends on the file format and whether metadata was present in the source.
576
+ * @example { author: 'John Smith', title: 'Annual Report', created: new Date('2024-01-01') }
577
+ */
578
+ metadata: OfficeMetadata;
579
+ /**
580
+ * The hierarchical content structure of the document.
581
+ * This is an array of top-level content nodes. Each node can have children, creating a tree.
582
+ * For different file types:
583
+ * - DOCX: Array of paragraphs, headings, tables, etc.
584
+ * - XLSX: Array of sheets, each containing rows
585
+ * - PPTX: Array of slides, each containing content nodes
586
+ * - PDF: Array of pages, each containing paragraphs
587
+ * @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
588
+ */
589
+ content: OfficeContentNode[];
590
+ /**
591
+ * Attachments extracted from the document (images, charts, embedded files).
592
+ * Only populated when `config.extractAttachments` is true.
593
+ * Each attachment includes:
594
+ * - Base64-encoded data
595
+ * - MIME type
596
+ * - Optional OCR text (if `config.ocr` is true)
597
+ * @example [{ type: 'image', mimeType: 'image/png', data: 'base64...', name: 'image1.png' }]
598
+ */
599
+ attachments: OfficeAttachment[];
600
+ /**
601
+ * Converts the entire AST to plain text.
602
+ * This method flattens the document structure and returns just the text content,
603
+ * stripping out all formatting, metadata, and structure.
604
+ *
605
+ * The text is concatenated using the delimiter specified in `config.newlineDelimiter` (default: '\n').
606
+ *
607
+ * @returns A plain text representation of the document
608
+ * @example
609
+ * ```typescript
610
+ * const text = ast.toText();
611
+ * console.log(text); // "Hello world\nChapter 1\n..."
612
+ * ```
613
+ */
614
+ toText(): string;
615
+ }
package/dist/types.js ADDED
@@ -0,0 +1,2 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
@@ -0,0 +1,7 @@
1
+ /// <reference types="node" />
2
+ import { ChartData } from "../types";
3
+ /**
4
+ * Universal chart data extractor that selects logic based on XML content.
5
+ * @param xmlBuffer Chart XML buffer
6
+ */
7
+ export declare const extractChartData: (xmlBuffer: Buffer) => ChartData;